{"id":"W4410950883","doi":"10.32388/6k4anl","title":"Review of: \"MedAgentBench: A Realistic Virtual EHR Environment to Benchmark Medical LLM Agents\"","year":2025,"lang":"en","type":"peer-review","venue":"","topic":"Electronic Health Records Systems","field":"Health Professions","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Benchmark (surveying); Computer science; Virtual patient; Data science; Medicine; Geography; Medical education; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03970384,0.000726285,0.001509148,0.008073668,0.001909334,0.006803705,0.003748,0.003041231,0.0323123],"category_scores_gemma":[0.2266836,0.0006246042,0.001428098,0.007404531,0.00183388,0.00588816,0.005424038,0.001888583,0.0183699],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004567941,"about_ca_system_score_gemma":0.02079001,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005878889,"about_ca_topic_score_gemma":0.01175753,"domain_scores_codex":[0.9623042,0.0144337,0.004852137,0.001255143,0.01643612,0.0007187374],"domain_scores_gemma":[0.7303199,0.08004342,0.01503153,0.01170327,0.1543596,0.008542233],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008754994,0.00005014502,0.0005329449,0.008476884,0.00007481997,0.00007556711,0.0003141568,0.0004878722,0.0002451077,0.004996614,0.7714565,0.2132019],"study_design_scores_gemma":[0.00003679441,0.00007258006,0.0009333086,0.00760898,0.00006164754,0.00009638514,0.0001851014,0.0004123311,0.0003040668,0.001195427,0.9890546,0.00003871936],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"other","genre_scores_codex":[0.008573473,0.2263569,0.05840538,0.4161106,0.1045914,0.005519449,0.02193916,0.008498332,0.1500054],"genre_scores_gemma":[0.1054604,0.3764934,0.08322828,0.1567001,0.04014437,0.009383421,0.05822383,0.009046725,0.1613194],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.03970384,"threshold_uncertainty_score":0.2099764,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1043235747251707,"score_gpt":0.4978021440717395,"score_spread":0.3934785693465688,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}