{"id":"W4414005603","doi":"10.2196/76702","title":"Development of a Clinical Clerkship Mentor Using Generative AI and Evaluation of Its Effectiveness in a Medical Student Trial Compared to Student Mentors: 2-Part Comparative Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Likert scale; Medical education; Rubric; CLARITY; Context (archaeology); Psychology; Scale (ratio); Medicine; Mathematics education","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01026887,0.0001805005,0.0008502335,0.0003835941,0.00008135704,0.0000154137,0.0001666225,0.0002121105,0.0002109887],"category_scores_gemma":[0.002722458,0.0001573319,0.00006073389,0.0006612567,0.00011675,0.00008206555,0.0001015546,0.0003562904,0.000003878325],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006371677,"about_ca_system_score_gemma":0.01216212,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001776464,"about_ca_topic_score_gemma":0.000579964,"domain_scores_codex":[0.9929345,0.001736011,0.002179472,0.0004776234,0.002452246,0.000220171],"domain_scores_gemma":[0.9971803,0.0005982785,0.0003421612,0.0002206834,0.001204927,0.0004536881],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01695661,0.05068562,0.4484362,0.0007503093,0.0006164724,0.000003647438,0.2062176,0.00007400327,0.0004828833,0.0003340307,0.0006518001,0.2747908],"study_design_scores_gemma":[0.01585319,0.002127307,0.8609414,0.004595167,0.0003953722,0.000003243857,0.09854662,0.01151404,0.005293358,0.0000980429,0.000374128,0.0002581986],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9884446,0.0002491122,0.0002375065,0.001527502,0.001903381,0.007516319,0.000001101067,0.000009801051,0.0001106217],"genre_scores_gemma":[0.9972633,0.00001592777,0.0003400025,0.0004766366,0.0003307468,0.001520797,0.00002633549,0.000007796916,0.00001852097],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4125051,"threshold_uncertainty_score":0.993438,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4342178331578377,"score_gpt":0.6669062045329408,"score_spread":0.2326883713751031,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}