{"id":"W4389495064","doi":"10.2139/ssrn.4658344","title":"Aml: An Accuracy Metric Model for Effective Evaluation of Log Parsing Techniques","year":2023,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Concordia University","funders":"","keywords":"Metric (unit); Parsing; Computer science; Statistics; Mathematics; Natural language processing; Engineering; Operations management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01130693,0.002068087,0.002021659,0.004476093,0.0007792112,0.004415778,0.004268236,0.002539283,0.00403262],"category_scores_gemma":[0.04896649,0.0006945772,0.001263707,0.003068466,0.001025436,0.007579088,0.002778822,0.002263626,0.00224512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002755402,"about_ca_system_score_gemma":0.003010832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005563004,"about_ca_topic_score_gemma":0.005600838,"domain_scores_codex":[0.9835336,0.004895784,0.001570138,0.001299123,0.008103259,0.0005981086],"domain_scores_gemma":[0.9699708,0.01426627,0.002186166,0.005845882,0.007076017,0.000654791],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001893099,0.0006693208,0.01237236,0.0008528474,0.0003708322,0.0001849268,0.0002893134,0.27757,0.01834431,0.04801134,0.03696883,0.6024727],"study_design_scores_gemma":[0.00003746263,0.0001846168,0.000970543,0.00003627009,0.0000638506,0.00008290871,0.00002608836,0.9687219,0.008448419,0.01742153,0.00396813,0.00003820916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01617225,0.0004634895,0.9614444,0.0004401319,0.0001205183,0.0002385691,0.001863953,0.0175525,0.001704225],"genre_scores_gemma":[0.3541522,0.0004080248,0.6329803,0.0003316134,0.0002758841,0.0007118766,0.005622141,0.002417119,0.003100818],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01130693,"threshold_uncertainty_score":0.05979747,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05219061129629513,"score_gpt":0.3627940116875584,"score_spread":0.3106034003912633,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}