{"id":"W4415443834","doi":"10.2196/81718","title":"Automated Evaluation of Reflection and Feedback Quality in Workplace-Based Assessments by Using Natural Language Processing: Cross-Sectional Competency-Based Medical Education Study","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Formative assessment; Reflection (computer programming); Narrative; Quality (philosophy); Quality assurance; Natural language; Quality assessment","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.013595,0.0004975453,0.0003780842,0.001478429,0.0003347295,0.0009122699,0.0006319658,0.0006760043,0.0005390101],"category_scores_gemma":[0.04191689,0.0003479606,0.000760746,0.0007478544,0.0005259519,0.001092124,0.001045185,0.0007622369,0.0003426581],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006737874,"about_ca_system_score_gemma":0.0009423349,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003908189,"about_ca_topic_score_gemma":0.005645958,"domain_scores_codex":[0.9942501,0.003159697,0.0005661433,0.001031856,0.0007708883,0.0002213065],"domain_scores_gemma":[0.9431754,0.02893459,0.01021022,0.004147376,0.01197661,0.001555739],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002991426,0.001054939,0.9743115,0.00005448228,0.000178632,0.00006827096,0.001916019,0.001632499,0.001147125,0.00004044773,0.0002202704,0.01907674],"study_design_scores_gemma":[0.00004412242,0.001738888,0.9572847,0.00003591927,0.0001172873,0.0002064151,0.002280689,0.03461385,0.00307551,0.0001562799,0.0004007002,0.00004568231],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9975408,0.0000200627,0.002122308,0.00001453019,0.000002937499,0.00006361462,0.000137177,0.00002261163,0.00007587268],"genre_scores_gemma":[0.9959415,0.00002225592,0.003410361,0.0000158575,0.000004648281,0.0001076176,0.000407265,0.000006450219,0.00008400082],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.986405,"threshold_uncertainty_score":0.0718981,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03673680330527152,"score_gpt":0.5284062895740835,"score_spread":0.491669486268812,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}