{"id":"W4411445833","doi":"10.7759/cureus.86366","title":"Adjusting for Resident Rater Leniency or Severity Improves the Reliability of Routine Resident Evaluations of Faculty Anesthesiologists","year":2025,"lang":"en","type":"article","venue":"Cureus","topic":"Radiology practices and education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Medicine; Inter-rater reliability; Reliability (semiconductor); Family medicine; Physical therapy; Rating scale; Statistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001437015,0.0000846133,0.0002457089,0.00003259703,0.0001071476,0.000005342541,0.000128188,0.00009046637,0.0000667974],"category_scores_gemma":[0.00689045,0.00004700276,0.00007914109,0.0001449578,0.0001576033,0.00008943567,0.00003628672,0.0001102506,0.000001027576],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008070967,"about_ca_system_score_gemma":0.0004262349,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005922461,"about_ca_topic_score_gemma":0.0001105633,"domain_scores_codex":[0.9989621,0.0001482254,0.0004327292,0.000203009,0.000116914,0.0001370079],"domain_scores_gemma":[0.9980747,0.0007216614,0.0002844251,0.000442058,0.0004509538,0.00002621676],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.007071196,0.001335675,0.8264797,0.002934834,0.0004730403,0.000005985524,0.01271815,0.0005298704,0.06400954,0.001975725,0.06079167,0.02167463],"study_design_scores_gemma":[0.0006433327,0.0004429407,0.9818968,0.0001006897,0.0003390044,0.00000918933,0.002441086,0.0006164055,0.01120255,0.0009661685,0.00128784,0.00005402535],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9916517,0.0005102746,0.00019476,0.0061286,0.0002438444,0.0008108897,0.00001195176,0.00001413633,0.0004338008],"genre_scores_gemma":[0.9954802,0.0001049295,0.0015681,0.0001522549,0.00007205938,0.00008317389,0.00003293579,0.000004610443,0.002501757],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1554171,"threshold_uncertainty_score":0.824901,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0823975803660913,"score_gpt":0.437334998881693,"score_spread":0.3549374185156017,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}