{"id":"W4406427891","doi":"10.2196/62779","title":"Detecting Artificial Intelligence–Generated Versus Human-Written Medical Student Essays: Semirandomized Controlled Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Randomized controlled trial; Psychology; Computer science; Medicine; World Wide Web; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03863779,0.001477427,0.001809573,0.001852077,0.00258721,0.001776326,0.001728152,0.002951442,0.008011622],"category_scores_gemma":[0.09264043,0.001309707,0.0009909444,0.0009297237,0.004934777,0.001829718,0.001561239,0.001916999,0.001965959],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002120741,"about_ca_system_score_gemma":0.003325722,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006296157,"about_ca_topic_score_gemma":0.001327283,"domain_scores_codex":[0.9573193,0.02783781,0.003914771,0.004924272,0.004566069,0.001437725],"domain_scores_gemma":[0.8889257,0.06829249,0.01755749,0.01249941,0.009983256,0.002741717],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"nonrandomized_trial","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.344908,0.3602656,0.09008358,0.003911929,0.001010737,0.002230698,0.06231437,0.002328542,0.02783328,0.003904588,0.005124817,0.09608394],"study_design_scores_gemma":[0.07897841,0.7510348,0.1117501,0.0005714118,0.0006089734,0.0004779436,0.01586434,0.005704293,0.01648473,0.003337399,0.01488776,0.0002999583],"study_design_candidate":"nonrandomized_trial","study_design_consensus":"nonrandomized_trial","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8998648,0.0001096875,0.003854556,0.000129425,0.0001979101,0.09416182,0.000356691,0.00006262178,0.001262379],"genre_scores_gemma":[0.7316204,0.0001146146,0.01267912,0.0005558959,0.0003084031,0.2524743,0.0003560388,0.00002933071,0.001861892],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03863779,"threshold_uncertainty_score":0.2043386,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1007353124654197,"score_gpt":0.509232481821237,"score_spread":0.4084971693558173,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}