{"id":"W3108103715","doi":"10.4300/jgme-d-20-00145.1","title":"Gender Effects in Assessment of Clinical Teaching: Does Concordance Matter?","year":2020,"lang":"en","type":"article","venue":"Journal of Graduate Medical Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Wilson Centre; University of Toronto","funders":"","keywords":"Concordance; Gender bias; Medicine; Rating scale; Ceiling effect; Family medicine; Psychology; Demography; Clinical psychology; Internal medicine; Social psychology; Alternative medicine; Developmental psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003554212,0.0001282142,0.0006585058,0.0002259093,0.00002990596,0.000009603453,0.0002184422,0.0001665673,0.0003373375],"category_scores_gemma":[0.01670271,0.000086751,0.0001512864,0.0004096909,0.000181308,0.0001422766,0.0000307009,0.001686383,0.00001124541],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001384744,"about_ca_system_score_gemma":0.004633301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001747292,"about_ca_topic_score_gemma":9.980176e-7,"domain_scores_codex":[0.9959512,0.0003703473,0.001959626,0.0001873917,0.001361178,0.0001702745],"domain_scores_gemma":[0.9976227,0.000324562,0.0009970788,0.0001832355,0.0005430911,0.0003293656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001619917,0.002847413,0.6785539,0.001018879,0.0001008081,0.00003711978,0.001591698,0.000003851872,0.0002926616,0.00201738,0.0856392,0.2277351],"study_design_scores_gemma":[0.002521243,0.0006736048,0.9827681,0.001179813,0.0001242653,0.0001284863,0.001225751,0.00275339,0.0001412775,0.001074551,0.007296856,0.0001127034],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9040024,0.0001042511,0.00567867,0.08532333,0.003169895,0.0002660111,4.074872e-7,0.000007693845,0.001447373],"genre_scores_gemma":[0.9580057,0.00009128875,0.02186976,0.01822314,0.001722526,0.00001030702,0.00001135907,0.00001648188,0.00004944937],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3042141,"threshold_uncertainty_score":0.99158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06940317322382675,"score_gpt":0.4759551126829898,"score_spread":0.406551939459163,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}