{"id":"W4408031439","doi":"10.1080/10401334.2025.2461991","title":"Beyond Student Evaluations of Teaching and Educator Portfolios: A Multisource, Longitudinal System for Evaluating Teaching","year":2025,"lang":"en","type":"article","venue":"Teaching and Learning in Medicine","topic":"Reflective Practices in Education","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Columbia College","funders":"","keywords":"Medical education; Mathematics education; Teaching method; Psychology; Computer science; Pedagogy; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03239127,0.0006359263,0.000755789,0.006549663,0.001230378,0.002443459,0.0008677486,0.0008319241,0.001427294],"category_scores_gemma":[0.06132142,0.0002135118,0.0005484353,0.003934321,0.0006274536,0.00297372,0.002809147,0.0009771846,0.0008508664],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001243843,"about_ca_system_score_gemma":0.00291054,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002700651,"about_ca_topic_score_gemma":0.005146245,"domain_scores_codex":[0.9818163,0.01173922,0.002312767,0.0007614355,0.00299269,0.0003776874],"domain_scores_gemma":[0.9318424,0.01937779,0.01358581,0.008889989,0.02291859,0.003385324],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008470875,0.001008832,0.7672681,0.0001924702,0.0001308207,0.00008405111,0.005071511,0.0007699658,0.001982856,0.0009492488,0.002368601,0.2193265],"study_design_scores_gemma":[0.0001157243,0.00445928,0.9499356,0.0003229948,0.0001959044,0.0004254454,0.006814232,0.01567567,0.00975433,0.003578048,0.008528088,0.0001945527],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9397346,0.0004900796,0.04120116,0.0005178017,0.0001198757,0.002083694,0.003005848,0.0005715819,0.01227533],"genre_scores_gemma":[0.9580719,0.0002291661,0.0321346,0.0001106666,0.0000626179,0.00381282,0.002622591,0.00006781499,0.002887864],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9676087,"threshold_uncertainty_score":0.1713035,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04901732876771051,"score_gpt":0.4991829524736445,"score_spread":0.450165623705934,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}