{"id":"W4416033657","doi":"10.18653/v1/2025.winlp-main.37","title":"Reference-Guided Verdict: LLMs-as-Judges in Automatic Evaluation of Free-Form QA","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada; Research Nova Scotia","keywords":"Process (computing); Automation; Identification (biology); Set (abstract data type); Measure (data warehouse)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01629305,0.001231666,0.001592868,0.002139217,0.001436233,0.003201304,0.003291148,0.003526303,0.01365771],"category_scores_gemma":[0.06778133,0.0005942967,0.0006910695,0.0008402882,0.001473243,0.003935744,0.005023778,0.00284054,0.007473405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001483986,"about_ca_system_score_gemma":0.002541105,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005503122,"about_ca_topic_score_gemma":0.01135159,"domain_scores_codex":[0.9675906,0.02199638,0.001467984,0.002809162,0.004929434,0.001206477],"domain_scores_gemma":[0.960143,0.02262694,0.001286002,0.006231889,0.008200411,0.001511672],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.005793632,0.000665605,0.01006678,0.001243562,0.0002870517,0.0008844121,0.003671325,0.05570908,0.02721121,0.0406237,0.1611549,0.6926886],"study_design_scores_gemma":[0.0004389846,0.0006415134,0.004352362,0.0002735274,0.0001122558,0.0004791376,0.0009664603,0.8657039,0.0395474,0.05371813,0.03356687,0.0001994615],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2390579,0.003191333,0.6249752,0.0032967,0.00127104,0.0009054104,0.005044929,0.08065299,0.04160452],"genre_scores_gemma":[0.8420695,0.0001354896,0.140019,0.0005359513,0.000190472,0.0001897538,0.004273081,0.003668665,0.008918223],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01629305,"threshold_uncertainty_score":0.08616692,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04708086877994515,"score_gpt":0.361978809971884,"score_spread":0.3148979411919389,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}