{"id":"W4415583954","doi":"10.1145/3755881.3755908","title":"Enhancement Report Approval Prediction: A Comparative Study of Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Natural language; Language model; Key (lock); Field (mathematics); Feature (linguistics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002053802,0.0002550909,0.0006099091,0.0001459933,0.0007216579,0.00007655271,0.0005166791,0.0001624278,0.002386167],"category_scores_gemma":[0.0001128677,0.0002453309,0.0001433427,0.001234884,0.0005605355,0.0004579751,0.0002876752,0.0002590685,0.00005067171],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002124027,"about_ca_system_score_gemma":0.0005183034,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006623195,"about_ca_topic_score_gemma":0.01105475,"domain_scores_codex":[0.9958236,0.000488102,0.001448703,0.0007051692,0.001008867,0.0005255142],"domain_scores_gemma":[0.9980711,0.0002016488,0.0004557278,0.0006319304,0.0005274212,0.0001122028],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001804759,0.01016886,0.002421736,0.00005959169,0.0006259,0.00007957673,0.7101933,0.002293174,0.0004518696,0.2672273,0.004311827,0.001986466],"study_design_scores_gemma":[0.0003744758,0.0007984581,0.0001397121,0.0001028854,0.0002006599,0.000001350697,0.9332663,0.03484368,0.0201111,0.005968076,0.003932552,0.0002607099],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5403385,0.0001906727,0.05352906,0.0002382315,0.001330144,0.001874246,0.00001711046,0.00005902076,0.4024231],"genre_scores_gemma":[0.9618225,0.00002991391,0.0002895689,0.0000741843,0.0001712874,0.000116671,0.000005818902,0.000007032052,0.037483],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4214841,"threshold_uncertainty_score":0.9999999,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09149538898364608,"score_gpt":0.4320685945447368,"score_spread":0.3405732055610908,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}