{"id":"W2044059462","doi":"10.1097/acm.0b013e31822a6cf8","title":"Rater-Based Assessments as Social Judgments: Rethinking the Etiology of Rater Errors","year":2011,"lang":"en","type":"review","venue":"Academic Medicine","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":194,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Northern British Columbia","funders":"","keywords":"Categorical variable; Psychology; Categorization; Impression formation; Inter-rater reliability; Social psychology; Construct (python library); Cognitive psychology; Social perception; Rating scale; Computer science; Perception; Developmental psychology; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4511673,0.001638655,0.004595663,0.01004639,0.001354689,0.009585964,0.007055001,0.002789495,0.001269822],"category_scores_gemma":[0.6504135,0.001643176,0.00215461,0.01178842,0.01218326,0.01308951,0.00440528,0.006270309,0.000987304],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006037916,"about_ca_system_score_gemma":0.009083715,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004272409,"about_ca_topic_score_gemma":0.005821215,"domain_scores_codex":[0.5092735,0.324151,0.03881156,0.0151426,0.1115256,0.001095674],"domain_scores_gemma":[0.1563227,0.7026173,0.04290565,0.02460744,0.07292757,0.0006193588],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001627359,0.00005104032,0.01441548,0.03252663,0.00177567,0.0002440347,0.01555455,0.001002736,0.0008672102,0.05699442,0.01273462,0.8636709],"study_design_scores_gemma":[0.0002632595,0.0008635504,0.1042098,0.180281,0.005461524,0.004543035,0.01545938,0.01332601,0.008862281,0.2353039,0.430334,0.001092154],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.006753442,0.8852514,0.07687549,0.02048205,0.002651354,0.0005141883,0.0001708653,0.0001541546,0.007147026],"genre_scores_gemma":[0.2306315,0.6492692,0.1026073,0.008568697,0.004599982,0.001691103,0.0002520228,0.0003349581,0.00204527],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.5488328,"threshold_uncertainty_score":0.6768085,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6437870321086094,"score_gpt":0.5584685302281239,"score_spread":0.08531850188048551,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}