{"id":"W4324299382","doi":"10.48550/arxiv.2303.07230","title":"Systematic Evaluation of Deep Learning Models for Log-based Failure Prediction","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission; Science Foundation Ireland; Université du Luxembourg","keywords":"Computer science; Artificial intelligence; Deep learning; Machine learning; Modular design; Encoder; Convolutional neural network; Embedding; Artificial neural network; Dependability; Metric (unit); Data mining; Autoencoder","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002673117,0.0002234735,0.0004707626,0.0003197962,0.0001518611,0.00005053912,0.0008920389,0.00033953,0.000003319214],"category_scores_gemma":[0.0002790022,0.0002249853,0.0002939039,0.0005134924,0.00005040486,0.0003845313,0.0003780327,0.0002951845,0.0000180688],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003604424,"about_ca_system_score_gemma":0.0003079862,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004830782,"about_ca_topic_score_gemma":0.00003268505,"domain_scores_codex":[0.997798,0.0004542935,0.000428427,0.0007947253,0.0002934606,0.0002310992],"domain_scores_gemma":[0.9969936,0.0003576022,0.000590764,0.001003229,0.0009875351,0.00006729148],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001185835,0.00003151067,0.00133265,0.01103744,0.00008209062,0.000001657695,0.0002637706,0.984143,0.00000512598,0.002951292,0.00002083985,0.0001187178],"study_design_scores_gemma":[0.0005737534,0.00007338822,0.0003397587,0.002100221,0.000281701,6.63013e-7,0.0001453879,0.972382,0.00005204658,0.02386799,0.000001983109,0.0001811205],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09386374,0.00006602063,0.9031645,0.00002773555,0.0007009872,0.001711942,0.00001308697,0.0003963315,0.00005565481],"genre_scores_gemma":[0.9979991,0.00001382941,0.001716066,0.000005531077,0.00004295953,0.00003253434,0.00004465957,0.00001748886,0.0001277709],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9041354,"threshold_uncertainty_score":0.9174635,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1194605287504361,"score_gpt":0.2168916450397203,"score_spread":0.0974311162892842,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}