{"id":"W4399135677","doi":"10.1136/fmch-2023-002626","title":"Performance of generative pre-trained transformers (GPTs) in Certification Examination of the College of Family Physicians of Canada","year":2024,"lang":"en","type":"article","venue":"Family Medicine and Community Health","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Jewish General Hospital; Canadian Institute for Advanced Research; McGill University; Saskatchewan Health Authority; McGill University Health Centre; Mila - Quebec Artificial Intelligence Institute; Saskatchewan Health; University of Saskatchewan","funders":"Fonds de Recherche du Québec - Santé; Natural Sciences and Engineering Research Council of Canada","keywords":"Certification; Logistic regression; Medical school; Medicine; Medical education; Family medicine; Mathematics; Internal medicine; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001280194,0.00008778995,0.0003846841,0.0001480262,0.0001028095,5.541617e-7,0.00009575941,0.00005086422,0.000004218722],"category_scores_gemma":[0.0001366179,0.00006354692,0.00002937467,0.0006548563,0.0002653257,0.00005426611,0.00001098076,0.0003395224,4.263953e-8],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001560431,"about_ca_system_score_gemma":0.001662349,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.4936427,"about_ca_topic_score_gemma":0.2089507,"domain_scores_codex":[0.9983343,0.0003969933,0.0007800541,0.00008789403,0.0002665544,0.0001342398],"domain_scores_gemma":[0.9987808,0.0004490623,0.0002258852,0.0002444639,0.0002494881,0.00005022853],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005971573,0.000990947,0.04115486,0.02014157,0.0001151435,9.456714e-7,0.2640236,0.000203431,0.1297139,0.00176929,0.002424246,0.5388649],"study_design_scores_gemma":[0.0002099825,0.001975178,0.8257549,0.003823986,0.00004705522,0.000002483539,0.1256187,0.01094914,0.03119069,0.0001566471,0.0002037216,0.00006741378],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9924406,0.001398923,0.00008689837,0.004413764,0.000209213,0.0005388999,0.00006025984,0.000004567876,0.0008468091],"genre_scores_gemma":[0.9982194,0.001156121,0.00004552335,0.0004439764,0.00003534694,0.00000989252,0.00003266233,0.000007241485,0.00004981435],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7846001,"threshold_uncertainty_score":0.8054839,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1669237254155756,"score_gpt":0.3973337135532156,"score_spread":0.2304099881376399,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}