{"id":"W4383068683","doi":"10.2196/preprints.50514","title":"Assessment of Resident and AI Chatbot Performance on the University of Toronto Family Medicine Residency Progress Test: Comparative Study (Preprint)","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); McNemar's test; Medicine; Chatbot; Medical education; Pace; Family medicine; Psychology; Artificial intelligence; Computer science; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005629355,0.0005005971,0.0005462921,0.001934902,0.0006593798,0.0009056207,0.0008594647,0.0008295366,0.002780108],"category_scores_gemma":[0.02907686,0.00031513,0.0005994927,0.001114416,0.0007861321,0.001160122,0.001066487,0.0007058689,0.001375587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001997984,"about_ca_system_score_gemma":0.001155218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03368263,"about_ca_topic_score_gemma":0.05125003,"domain_scores_codex":[0.9962891,0.001217104,0.0004757659,0.0005435889,0.001123227,0.0003511837],"domain_scores_gemma":[0.9642256,0.01775385,0.004154855,0.001423781,0.008786912,0.003654841],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.006594398,0.004964587,0.8985447,0.0003174935,0.0003595591,0.0004572024,0.02587742,0.0008859131,0.004086206,0.0001416491,0.003640729,0.05413011],"study_design_scores_gemma":[0.00006262115,0.004753369,0.9894533,0.00001729066,0.00004171412,0.00008695661,0.003438005,0.0007380007,0.000836364,0.0000179603,0.0005300928,0.00002443418],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9987398,0.00003880917,0.00009446178,0.00002826186,0.0000115021,0.00005856926,0.000225623,0.00001514947,0.0007877579],"genre_scores_gemma":[0.9977165,0.00004246749,0.0002970733,0.00003808221,0.00001813388,0.00009552699,0.0005795746,0.00000943376,0.001203116],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03368263,"threshold_uncertainty_score":0.06697309,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3075699642974704,"score_gpt":0.4837462154936909,"score_spread":0.1761762511962205,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}