{"id":"W4416033657","doi":"10.18653/v1/2025.winlp-main.37","title":"Reference-Guided Verdict: LLMs-as-Judges in Automatic Evaluation of Free-Form QA","year":2025,"lang":"","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada; Research Nova Scotia","keywords":"Process (computing); Automation; Identification (biology); Set (abstract data type); Measure (data warehouse)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.004192525,0.0003669162,0.0005895282,0.001001565,0.0001307105,0.000269848,0.002897643,0.0003931033,0.0007077634],"category_scores_gemma":[0.002314574,0.0003281154,0.0001126817,0.002642942,0.0001827032,0.001132984,0.001470875,0.0005306229,0.00003810938],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005910023,"about_ca_system_score_gemma":0.001815537,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001223268,"about_ca_topic_score_gemma":0.0004245056,"domain_scores_codex":[0.9954394,0.0004308128,0.001284853,0.0007746379,0.001565162,0.0005051558],"domain_scores_gemma":[0.9961689,0.0002627783,0.000454774,0.001829422,0.001206149,0.00007793806],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001462464,0.0002670025,0.00122707,0.0006555917,0.00005517021,0.000008538736,0.00239328,0.00002425027,0.002588765,0.2003773,0.006002667,0.7863857],"study_design_scores_gemma":[0.001089631,0.0001294496,0.002997376,0.001314487,0.00009517273,0.000006758763,0.0001403418,0.4740531,0.07222654,0.4475659,0.00006977806,0.0003114043],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4739499,0.0304745,0.2043963,0.0120436,0.001981164,0.004572826,0.00002044884,0.002258286,0.2703029],"genre_scores_gemma":[0.8373829,0.00007367737,0.1608062,0.0004843212,0.00001804488,0.00006419799,0.000005870653,0.00001100646,0.001153864],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7860743,"threshold_uncertainty_score":0.9999171,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04708086877994515,"score_gpt":0.361978809971884,"score_spread":0.3148979411919389,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}