{"id":"W4408767489","doi":"10.32388/ig5abm","title":"Review of: \"MTRAG: A Multi-Turn Conversational Benchmark for Evaluating Retrieval-Augmented Generation Systems\"","year":2025,"lang":"en","type":"peer-review","venue":"","topic":"Power Systems and Technologies","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Benchmark (surveying); Computer science; Turn-taking; Information retrieval; Turn (biochemistry); Artificial intelligence; Communication; Geography; Psychology; Conversation; Biology; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001592363,0.0004305788,0.00127204,0.0002193966,0.00005904634,0.00003241278,0.00036047,0.0004318029,0.000183338],"category_scores_gemma":[0.001472365,0.0003786733,0.0003339813,0.0004070419,0.0000275178,0.00008280128,0.0000667796,0.0002951798,0.000009025162],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002807881,"about_ca_system_score_gemma":0.0002562876,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005756255,"about_ca_topic_score_gemma":0.0000229272,"domain_scores_codex":[0.997313,0.00008515566,0.001414631,0.0004094615,0.000495929,0.0002818322],"domain_scores_gemma":[0.997776,0.0002281704,0.000438178,0.0006002187,0.0009160179,0.00004137391],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000001281851,0.000009308345,8.312624e-7,0.2385773,0.0002258209,3.614917e-7,0.000003744335,0.000137394,0.0001831265,0.0002787537,0.7586821,0.001899966],"study_design_scores_gemma":[0.0003638411,0.00004518428,0.000001480501,0.1532257,0.0004188037,0.00000452257,0.00001021636,0.07390478,0.000317436,0.000005770562,0.7713701,0.0003321896],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.000004037523,0.9297464,0.05124076,0.001933456,0.007422197,0.004606759,0.001326629,0.0005148588,0.003204876],"genre_scores_gemma":[0.0005697311,0.8500372,0.01555747,0.00185751,0.0008037607,0.002334635,0.01445803,0.0001594664,0.1142222],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.1110173,"threshold_uncertainty_score":0.9998665,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08660005250844185,"score_gpt":0.3502156399937281,"score_spread":0.2636155874852862,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}