{"id":"W4386457251","doi":"10.2196/50514","title":"Assessment of Resident and AI Chatbot Performance on the University of Toronto Family Medicine Residency Progress Test: Comparative Study","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":58,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hospital for Sick Children; University of Toronto","funders":"","keywords":"McNemar's test; Test (biology); Medicine; Pace; Multiple choice; Medical education; Chatbot; Family medicine; Cohort; Psychology; Artificial intelligence; Computer science; Internal medicine; Significant difference; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003699534,0.0004616925,0.0005350125,0.001600751,0.0005167575,0.0006039796,0.0007468179,0.0006507179,0.001622877],"category_scores_gemma":[0.01680162,0.0002314895,0.0005008426,0.0008150794,0.0006982081,0.000807766,0.0009556932,0.0004275077,0.0006936888],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001741951,"about_ca_system_score_gemma":0.0009929186,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02934907,"about_ca_topic_score_gemma":0.04555633,"domain_scores_codex":[0.9973347,0.0006887866,0.0002557311,0.0003691956,0.001071713,0.0002799319],"domain_scores_gemma":[0.9864866,0.004346926,0.00248505,0.0005702178,0.003887017,0.002224221],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001254282,0.0008669298,0.9771881,0.00008079369,0.000171184,0.0001981998,0.004207152,0.0002936744,0.0009934272,0.00003037233,0.00059007,0.01412579],"study_design_scores_gemma":[0.00001774487,0.001773629,0.9960455,0.000008903558,0.00002742656,0.000102318,0.001044137,0.0004565443,0.0003245279,0.000006669817,0.0001839457,0.000008486489],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9992248,0.00004971325,0.00005579777,0.00001649895,0.000004474299,0.00002298085,0.0001029056,0.000007449521,0.000515395],"genre_scores_gemma":[0.9991416,0.00004087584,0.0001326941,0.00001572532,0.000005847757,0.00002352187,0.0002481055,0.000003434514,0.0003881634],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9982581,"threshold_uncertainty_score":0.05835646,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.15186098784238,"score_gpt":0.5025409286234617,"score_spread":0.3506799407810817,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}