{"id":"W4386066764","doi":"10.5811/westjem.58153","title":"Gender and Inconsistent Evaluations: A Mixed-methods Analysis of Feedback for Emergency Medicine Residents","year":2023,"lang":"en","type":"article","venue":"Western Journal of Emergency Medicine","topic":"Diversity and Career in Medicine","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"National Institutes of Health; National Center for Advancing Translational Sciences; University of Southern California","keywords":"Medicine; Demography; Family medicine; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01511946,0.0001829708,0.0008883355,0.001314203,0.0002999899,0.000003936585,0.000567845,0.0001155477,0.003890881],"category_scores_gemma":[0.005610277,0.0001396989,0.0002832713,0.002864832,0.0004846269,0.0002165497,0.00009241729,0.0001826483,0.000004437245],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006517523,"about_ca_system_score_gemma":0.0001712189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006322264,"about_ca_topic_score_gemma":0.0008795063,"domain_scores_codex":[0.9952088,0.0006821433,0.001638154,0.0002471735,0.001847414,0.0003763234],"domain_scores_gemma":[0.9960956,0.0005560662,0.001104787,0.0002468544,0.001614165,0.0003825328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003533881,0.0002036026,0.6096358,0.0005951024,0.006266771,0.00002523188,0.122027,0.0003173893,0.003672474,0.002658319,0.234307,0.01993789],"study_design_scores_gemma":[0.003197702,0.00168938,0.7754914,0.0005888352,0.01227841,0.000005277852,0.1717531,0.0003500866,0.00004428157,0.009045116,0.0252388,0.0003176169],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9282737,0.01501608,0.01328845,0.01526878,0.01962395,0.0007888047,0.00004048475,0.00005346394,0.007646324],"genre_scores_gemma":[0.9356903,0.04782899,0.001822241,0.00027387,0.003782803,0.00002129194,0.00006897444,0.00004120229,0.01047031],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2090682,"threshold_uncertainty_score":0.9970197,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2103788623131993,"score_gpt":0.5010196562997677,"score_spread":0.2906407939865684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}