{"id":"W4414993542","doi":"10.1145/3771283","title":"LLM meets ML: Data-efficient Anomaly Detection on Unstable Logs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Anomaly detection; Inference; Anomaly (physics); Artificial neural network; Key (lock); Software; Cache; Ensemble learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002708225,0.001877792,0.001420603,0.002436306,0.0006382896,0.001555168,0.003204364,0.001361185,0.001581998],"category_scores_gemma":[0.01240944,0.0005388622,0.0009329263,0.00144813,0.0005837476,0.004040902,0.002167657,0.002052704,0.001921159],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008393524,"about_ca_system_score_gemma":0.001700229,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007663546,"about_ca_topic_score_gemma":0.01278014,"domain_scores_codex":[0.9980255,0.0004365074,0.0001674831,0.0005709477,0.0006037651,0.0001956873],"domain_scores_gemma":[0.995257,0.001787347,0.0003126631,0.001557281,0.0009004718,0.0001852835],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000793575,0.0007744943,0.03413707,0.000368118,0.0002444093,0.000346753,0.0002256518,0.1189071,0.01396894,0.002324306,0.03909512,0.7888144],"study_design_scores_gemma":[0.0000297762,0.0001155103,0.002414584,0.00001474352,0.00001651262,0.0001161891,0.00007714598,0.9836513,0.006450299,0.00388506,0.003204855,0.00002392927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1745263,0.00196412,0.6920414,0.001119051,0.0003848417,0.0003264395,0.005786232,0.1203727,0.003479003],"genre_scores_gemma":[0.6542233,0.0003678425,0.3272654,0.0004423463,0.0001351541,0.0002657968,0.01327669,0.001035592,0.002988031],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007663546,"threshold_uncertainty_score":0.01523787,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07009888762071217,"score_gpt":0.3147679512299077,"score_spread":0.2446690636091955,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}