{"id":"W6987936627","doi":"","title":"Use of Large Language Models for Family medicine education and examination","year":2025,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Fonds de Recherche du Québec - Santé; Natural Sciences and Engineering Research Council of Canada","keywords":"MEDLINE; Curriculum; Feature (linguistics)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02420626,0.001161418,0.0006499509,0.002202149,0.001120988,0.004434677,0.001626039,0.001100581,0.007354336],"category_scores_gemma":[0.1372745,0.0007474622,0.0009375241,0.0009405633,0.001207332,0.006554772,0.004851838,0.00176613,0.002029018],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002553428,"about_ca_system_score_gemma":0.003180788,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00520375,"about_ca_topic_score_gemma":0.007068519,"domain_scores_codex":[0.9699753,0.0253363,0.0009722297,0.001522537,0.001849674,0.0003440186],"domain_scores_gemma":[0.7440655,0.225676,0.006805864,0.01499652,0.006593179,0.001862888],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003060155,0.002716396,0.05507815,0.002072551,0.0003570337,0.000898583,0.0257914,0.01654609,0.02196123,0.01130466,0.01447083,0.8457431],"study_design_scores_gemma":[0.001483324,0.006084446,0.0745694,0.003656828,0.001382981,0.004367727,0.02289627,0.6260155,0.05467367,0.07103328,0.1325529,0.001283671],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5541148,0.001639976,0.3942781,0.006129927,0.0003627108,0.002742745,0.002062544,0.01589441,0.02277483],"genre_scores_gemma":[0.7954686,0.0003772067,0.1981356,0.0006915511,0.0001049201,0.001111095,0.001457649,0.0006134401,0.002039932],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9757937,"threshold_uncertainty_score":0.1280165,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1299960903705907,"score_gpt":0.4004185184247862,"score_spread":0.2704224280541955,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}