{"id":"W4399135677","doi":"10.1136/fmch-2023-002626","title":"Performance of generative pre-trained transformers (GPTs) in Certification Examination of the College of Family Physicians of Canada","year":2024,"lang":"en","type":"article","venue":"Family Medicine and Community Health","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Jewish General Hospital; Canadian Institute for Advanced Research; McGill University; Saskatchewan Health Authority; McGill University Health Centre; Mila - Quebec Artificial Intelligence Institute; Saskatchewan Health; University of Saskatchewan","funders":"Fonds de Recherche du Québec - Santé; Natural Sciences and Engineering Research Council of Canada","keywords":"Certification; Logistic regression; Medical school; Medicine; Medical education; Family medicine; Mathematics; Internal medicine; Political science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02154715,0.000597427,0.000584439,0.001355476,0.0007831898,0.001620453,0.001416764,0.0006727329,0.003950031],"category_scores_gemma":[0.123354,0.0004454177,0.0008136495,0.0009686895,0.001348888,0.001069056,0.002131019,0.0009308649,0.001462905],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00586003,"about_ca_system_score_gemma":0.01130122,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1206691,"about_ca_topic_score_gemma":0.1526414,"domain_scores_codex":[0.9828841,0.009362658,0.001115975,0.001870121,0.003804195,0.0009629338],"domain_scores_gemma":[0.8753332,0.07683377,0.01320644,0.005785598,0.02273927,0.006101699],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002161131,0.0007850215,0.7534749,0.0004975752,0.0001213967,0.0002878643,0.01444865,0.005125588,0.002130304,0.0006098616,0.008213436,0.2121443],"study_design_scores_gemma":[0.0002178202,0.004031586,0.9294168,0.0003281037,0.0001324565,0.0005361162,0.007084453,0.04074284,0.006704761,0.001084262,0.009506183,0.0002145434],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.990484,0.0001512904,0.00310074,0.0005085361,0.00005298018,0.0006182807,0.0007078868,0.0004073081,0.003969007],"genre_scores_gemma":[0.9927434,0.0001085088,0.004415334,0.0001524936,0.00002095014,0.0002322155,0.0007376305,0.00004185937,0.001547716],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.99414,"threshold_uncertainty_score":0.2399335,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1669237254155756,"score_gpt":0.3973337135532156,"score_spread":0.2304099881376399,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}