{"id":"W4407042205","doi":"10.2196/66478","title":"Novel Evaluation Metric and Quantified Performance of ChatGPT-4 Patient Management Simulations for Early Clinical Education: Experimental Study","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Metric (unit); Computer science; Engineering; Operations management; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00620759,0.0008826433,0.000610006,0.001157861,0.0003623687,0.001076826,0.001059352,0.0009534459,0.00305425],"category_scores_gemma":[0.05100328,0.0002838398,0.0004918046,0.0005990983,0.000751601,0.001183227,0.001310239,0.0006557063,0.0005279792],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000847263,"about_ca_system_score_gemma":0.0008007067,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001465515,"about_ca_topic_score_gemma":0.0009428737,"domain_scores_codex":[0.9939235,0.003278626,0.0006417236,0.0007553668,0.001134927,0.0002658543],"domain_scores_gemma":[0.9472501,0.03705977,0.003876655,0.003052439,0.006791838,0.001969222],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.02200503,0.0166187,0.2169759,0.002523663,0.0006387605,0.0007479113,0.007027434,0.1660023,0.1166248,0.004754225,0.004343954,0.4417374],"study_design_scores_gemma":[0.0009891307,0.05441262,0.1669381,0.0002144263,0.0005394458,0.0008340777,0.002096069,0.6802402,0.08762384,0.002222901,0.003581265,0.0003079341],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9565493,0.0001329394,0.03914472,0.00006243867,0.00006050649,0.0007903461,0.0003072791,0.000405325,0.002547125],"genre_scores_gemma":[0.9667487,0.00005687038,0.0318086,0.00002964219,0.00001755907,0.0005519649,0.0002531636,0.00003588782,0.0004976084],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00620759,"threshold_uncertainty_score":0.03282923,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4811093885014956,"score_gpt":0.6459946774200589,"score_spread":0.1648852889185632,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}