{"id":"W4412376941","doi":"10.1145/3726302.3730165","title":"Assessing Support for the TREC 2024 RAG Track: A Large-Scale Comparative Study of LLM and Human Evaluations","year":2025,"lang":"en","type":"article","venue":"","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; National Institute of Standards and Technology","keywords":"Computer science; Track (disk drive); Scale (ratio); Information retrieval; Artificial intelligence; Operating system; Cartography; Geography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06726322,0.001056391,0.001206274,0.005503639,0.002258966,0.003996826,0.002048787,0.002125887,0.002671446],"category_scores_gemma":[0.2385278,0.0003395806,0.0008336513,0.002980071,0.00184013,0.003086775,0.003267785,0.001776233,0.001818918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002014666,"about_ca_system_score_gemma":0.001782325,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005233805,"about_ca_topic_score_gemma":0.008632528,"domain_scores_codex":[0.9037781,0.06121673,0.005615464,0.006357939,0.02148478,0.00154693],"domain_scores_gemma":[0.6271861,0.2931436,0.01807903,0.01749085,0.03862507,0.005475387],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.009669437,0.002125558,0.2580071,0.006366707,0.002051808,0.001931665,0.05079376,0.01481201,0.03667196,0.00356466,0.1242776,0.4897278],"study_design_scores_gemma":[0.001696712,0.009634541,0.670253,0.001669736,0.001017453,0.002847108,0.02220157,0.1055453,0.0446121,0.007612502,0.1316268,0.001283118],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9346996,0.003831734,0.02153626,0.001792896,0.0008892572,0.001150009,0.005518428,0.003598055,0.02698387],"genre_scores_gemma":[0.9735801,0.0003572998,0.01453168,0.0005632067,0.0002746469,0.0006127157,0.006136873,0.0006467263,0.003296762],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06726322,"threshold_uncertainty_score":0.3557262,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05437345212556912,"score_gpt":0.4637000279621464,"score_spread":0.4093265758365773,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}