{"id":"W4386023761","doi":"10.1111/1471-0528.17641","title":"Performance of <scp>ChatGPT</scp> in medical examinations: A systematic review and a meta‐analysis","year":2023,"lang":"en","type":"review","venue":"BJOG An International Journal of Obstetrics & Gynaecology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":100,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Jewish General Hospital","funders":"","keywords":"Scopus; Web of science; English language; Meta-analysis; Conversation; Systematic review; MEDLINE; Multiple choice; Computer science; Medical education; Psychology; Medicine; Mathematics education; Pathology; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003164338,0.0002792172,0.004899289,0.003659383,0.00003330937,0.00002314525,0.0006365711,0.0004385273,0.000360568],"category_scores_gemma":[0.0289213,0.0002048833,0.0009602067,0.002388792,0.0001115722,0.0002091661,0.00008466011,0.0007542484,0.0000226162],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005022854,"about_ca_system_score_gemma":0.001431974,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005261195,"about_ca_topic_score_gemma":0.00004299952,"domain_scores_codex":[0.9942785,0.0006101157,0.003488769,0.0002831243,0.001099464,0.0002400574],"domain_scores_gemma":[0.986751,0.008206674,0.002683101,0.0003104099,0.001786617,0.0002622645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.000004041392,0.0006620155,0.001539383,0.7818016,0.07724814,0.0006619396,0.000703029,0.00001084533,3.003442e-8,0.0002595165,0.0001187068,0.1369908],"study_design_scores_gemma":[0.0002816127,0.001539053,0.002025573,0.0628258,0.9162096,0.002064094,0.0008142974,0.0007102934,0.00000213598,0.0001494732,0.01311478,0.0002632873],"study_design_candidate":"meta_analysis","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0007959996,0.9965405,0.00003545277,0.0001915829,0.001311548,0.0009605021,0.00002449266,0.000009579471,0.0001303251],"genre_scores_gemma":[0.002945621,0.9959929,0.0001534352,0.0002949922,0.00009769825,0.0001371097,0.00006942759,0.00002945442,0.0002793723],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8389615,"threshold_uncertainty_score":0.9792585,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2378566884582639,"score_gpt":0.4765983405220842,"score_spread":0.2387416520638203,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}