{"id":"W4399871949","doi":"10.1007/s10664-024-10501-4","title":"Systematic Evaluation of Deep Learning Models for Log-based Failure Prediction","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"HORIZON EUROPE Framework Programme; Canada Research Chairs; Natural Sciences and Engineering Research Council of Canada; European Commission; Science Foundation Ireland; Université du Luxembourg","keywords":"Computer science; Machine learning; Artificial intelligence; Artificial neural network; Algorithm; Convolutional neural network; Deep learning; Encoder; Predictive modelling; Data mining","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003726257,0.001768218,0.0007091052,0.001242353,0.0003029644,0.0007141958,0.001844427,0.00111006,0.001218326],"category_scores_gemma":[0.00959413,0.0005066569,0.001096237,0.0007957489,0.0005330985,0.001513455,0.000954934,0.00143502,0.0004117301],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001560368,"about_ca_system_score_gemma":0.0009528602,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01189023,"about_ca_topic_score_gemma":0.01218626,"domain_scores_codex":[0.9984054,0.0007233886,0.000148418,0.0003793485,0.0002290559,0.000114332],"domain_scores_gemma":[0.9931599,0.004257709,0.0004201424,0.0008872288,0.001124064,0.0001509742],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","study_design_scores_codex":[0.0009275912,0.0006766955,0.01820247,0.0005151284,0.0003982491,0.0001148063,0.00005372637,0.8623008,0.003508616,0.0007490637,0.004890249,0.1076625],"study_design_scores_gemma":[0.00003233395,0.0002154856,0.002033008,0.00002319763,0.00003648755,0.00001812919,0.00001714569,0.9938117,0.003180577,0.0003102134,0.0003123535,0.000009351405],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9316487,0.002117273,0.05203638,0.0007002354,0.0002240451,0.0002929029,0.006158285,0.004491653,0.002330564],"genre_scores_gemma":[0.9628586,0.000282299,0.02529515,0.0001056503,0.00002909746,0.0001642432,0.01054657,0.00007952136,0.0006387698],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01189023,"threshold_uncertainty_score":0.02364206,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03017146570369021,"score_gpt":0.2758273760452596,"score_spread":0.2456559103415694,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}