{"id":"W6979329680","doi":"","title":"Support Evaluation for the TREC 2024 RAG Track: Comparing Human versus LLM Judges","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Institute for Information and Communications Technology Promotion; Natural Sciences and Engineering Research Council of Canada; National Institute of Standards and Technology; Ministry of Science and ICT, South Korea","keywords":"Scratch; Human error; Language model; Documentation","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03884235,0.0009410242,0.001035242,0.00247929,0.001960203,0.003315098,0.001661525,0.002107836,0.00281606],"category_scores_gemma":[0.1604418,0.0003631503,0.0007303057,0.001396848,0.001196043,0.002683502,0.002894841,0.001548065,0.002117104],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001256027,"about_ca_system_score_gemma":0.001384955,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004180469,"about_ca_topic_score_gemma":0.00685893,"domain_scores_codex":[0.955305,0.02895582,0.002961082,0.003944596,0.00778374,0.001049742],"domain_scores_gemma":[0.817485,0.132364,0.008513779,0.01208535,0.02549659,0.004055155],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01197013,0.002555982,0.1891933,0.003865482,0.001231392,0.001504748,0.07525199,0.01685964,0.05586216,0.004013392,0.1061785,0.5315133],"study_design_scores_gemma":[0.003349658,0.01515489,0.4715006,0.0013906,0.001191644,0.003310169,0.04676498,0.2144641,0.07846992,0.01558657,0.1471912,0.001625627],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.948597,0.001455396,0.02384187,0.001342592,0.0005542757,0.0008812468,0.00243061,0.003109943,0.01778708],"genre_scores_gemma":[0.966388,0.0001941593,0.02548217,0.0005213544,0.0001941978,0.0005576811,0.003443985,0.0003943812,0.00282398],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03884235,"threshold_uncertainty_score":0.2054204,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1847326692448298,"score_gpt":0.3754335193551211,"score_spread":0.1907008501102913,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}