{"id":"W4386023761","doi":"10.1111/1471-0528.17641","title":"Performance of <scp>ChatGPT</scp> in medical examinations: A systematic review and a meta‐analysis","year":2023,"lang":"en","type":"review","venue":"BJOG An International Journal of Obstetrics & Gynaecology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":100,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Jewish General Hospital","funders":"","keywords":"Scopus; Web of science; English language; Meta-analysis; Conversation; Systematic review; MEDLINE; Multiple choice; Computer science; Medical education; Psychology; Medicine; Mathematics education; Pathology; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0237418,0.002167921,0.01454094,0.01085015,0.0009093531,0.003396796,0.002551537,0.002556846,0.0036516],"category_scores_gemma":[0.07911783,0.001040102,0.02161799,0.01319133,0.001179426,0.002541783,0.00194249,0.00130384,0.0002987834],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003931822,"about_ca_system_score_gemma":0.007090675,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006213102,"about_ca_topic_score_gemma":0.01804624,"domain_scores_codex":[0.9792276,0.007719597,0.007831885,0.001668093,0.0030949,0.0004579281],"domain_scores_gemma":[0.9297089,0.05343253,0.009917914,0.001877755,0.004540191,0.0005227108],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"meta_analysis","study_design_scores_codex":[0.0009186793,0.00002803111,0.005364305,0.8149873,0.161545,0.000135627,0.0002053948,0.0001707896,0.0002418168,0.0001401414,0.0008074508,0.01545544],"study_design_scores_gemma":[0.0006032977,0.0003721481,0.01234759,0.2415902,0.7400185,0.0002715073,0.0002372691,0.0002402106,0.0003127762,0.0003150498,0.003638533,0.00005291942],"study_design_candidate":"meta_analysis","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.006047159,0.991119,0.0004762509,0.0003194901,0.0001580204,0.0007879816,0.0008119012,0.00002490709,0.0002552939],"genre_scores_gemma":[0.1648162,0.8247181,0.002929387,0.001375486,0.0004523296,0.003856228,0.001432167,0.00003715736,0.0003828565],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.0237418,"threshold_uncertainty_score":0.1255601,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2378566884582639,"score_gpt":0.4765983405220842,"score_spread":0.2387416520638203,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}