{"id":"W3174008823","doi":"10.1097/sla.0000000000005015","title":"Gender Bias in the Evaluation of Surgical Performance","year":2021,"lang":"en","type":"article","venue":"Annals of Surgery","topic":"Diversity and Career in Medicine","field":"Social Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"National Cancer Institute; National Institutes of Health","keywords":"Medicine; Checklist; Inter-rater reliability; Cronbach's alpha; Gender bias; Respondent; Clinical psychology; Rating scale; Psychometrics; Psychology; Social psychology; Developmental psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01160178,0.00002469874,0.0001035156,0.00004557442,0.00004903625,0.000006681584,0.0001064504,0.00003298454,0.0003770369],"category_scores_gemma":[0.000536372,0.00001904426,0.0000595013,0.0003519118,0.0001420957,0.00008920117,0.00001613066,0.00004664011,0.000003605698],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000004824815,"about_ca_system_score_gemma":0.0002992781,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002005163,"about_ca_topic_score_gemma":0.00006800373,"domain_scores_codex":[0.9981189,0.000473268,0.0001334783,0.00005667599,0.001118085,0.00009955959],"domain_scores_gemma":[0.998861,0.0005625057,0.00005586863,0.00008374392,0.0004187015,0.00001823471],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00003466372,0.0002296407,0.8195875,0.0000758906,0.00003390433,0.00004230148,0.03112141,0.00009267747,0.00002508071,0.005179072,0.04264891,0.1009289],"study_design_scores_gemma":[0.0002985488,0.00001796891,0.8478118,0.0001492275,0.00004431147,0.000003225632,0.07777331,0.00009619155,0.002417365,0.00183079,0.06944724,0.0001100323],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9568107,0.0007915814,3.348314e-7,0.003456898,0.0001317348,0.00003274237,0.000001716305,0.000002048336,0.03877224],"genre_scores_gemma":[0.9974747,0.002048538,0.000002043337,0.0003384093,0.00005921324,0.000001171698,0.000004219266,8.36399e-7,0.00007085943],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1008189,"threshold_uncertainty_score":0.412829,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7118481104725602,"score_gpt":0.4509731255451161,"score_spread":0.2608749849274442,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}