{"id":"W4391941699","doi":"10.1080/0142159x.2024.2316223","title":"Twelve tips for Natural Language Processing in medical education program evaluation","year":2024,"lang":"en","type":"article","venue":"Medical Teacher","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Centre for Addiction and Mental Health","funders":"","keywords":"Workflow; Computer science; Preprocessor; Process (computing); Medical education; Data science; Artificial intelligence; Medicine; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002379486,0.00008756327,0.0001023575,0.00009837178,0.00003246927,0.0001292446,0.0004500241,0.0001516123,0.000287832],"category_scores_gemma":[0.001511918,0.00006896246,0.00004179887,0.0002772821,0.00003356991,0.0002584795,0.0000830199,0.0004062214,0.00001651983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001280028,"about_ca_system_score_gemma":0.001702799,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002802247,"about_ca_topic_score_gemma":0.00005320858,"domain_scores_codex":[0.9979187,0.00009958931,0.0002363153,0.0003592287,0.001164246,0.0002219847],"domain_scores_gemma":[0.9995571,0.00003744956,0.00002277938,0.0001824744,0.0000552057,0.0001449514],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001325676,0.0001072998,0.0001597445,0.00006729337,0.000002600281,0.000008216589,0.002018537,0.000001784323,0.000003066719,0.001261554,0.001314417,0.9950542],"study_design_scores_gemma":[0.0002230131,0.00001788439,0.0001492666,0.0003087603,0.000005757099,0.00002392527,0.0001077954,0.9833174,0.000005979579,0.001206381,0.01454962,0.000084215],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1622178,0.0284162,0.7497938,0.04486093,0.004188056,0.003473256,3.854755e-7,0.001190791,0.005858845],"genre_scores_gemma":[0.9820914,0.0000045785,0.015004,0.0004982192,0.0006764924,0.0009448359,0.00001467798,0.00001118622,0.0007545645],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.99497,"threshold_uncertainty_score":0.3151559,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0370885945004089,"score_gpt":0.4096117849988083,"score_spread":0.3725231904983994,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}