{"id":"W2467932880","doi":"10.1177/0013164416654740","title":"An Unbiased Estimate of Global Interrater Agreement","year":2016,"lang":"en","type":"article","venue":"Educational and Psychological Measurement","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Trois-Rivières; University of Ottawa","funders":"","keywords":"Inter-rater reliability; Statistics; Agreement; Econometrics; Psychology; Mathematics; Rating scale; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00468477,0.0001685083,0.0002420851,0.00004880848,0.00009732899,0.0000705069,0.0005302479,0.00006647132,0.004082111],"category_scores_gemma":[0.001237599,0.0000809371,0.00008718234,0.0002394589,0.0002409615,0.0001792417,0.00004938157,0.00005394743,0.000220565],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001358931,"about_ca_system_score_gemma":0.00005972539,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001225435,"about_ca_topic_score_gemma":0.00002549133,"domain_scores_codex":[0.99567,0.0003640442,0.0007689554,0.0006351526,0.002324226,0.0002376765],"domain_scores_gemma":[0.9979936,0.0002339856,0.0002208966,0.0005532801,0.0007717469,0.0002264475],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004415165,0.005118027,0.6080528,0.00001731195,0.0000770604,0.000001501954,0.0002491693,0.00002459881,0.06034634,0.03082471,0.02655412,0.2682928],"study_design_scores_gemma":[0.0005476729,0.0005817458,0.8357299,0.00006297764,0.00001024668,0.00000398904,0.00009014054,0.00001093242,0.0006666507,0.1592554,0.002900657,0.0001396689],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9668057,0.0001687776,0.006221743,0.02041754,0.0008634564,0.0002825336,0.00003129165,0.00001404692,0.005194914],"genre_scores_gemma":[0.9975572,0.00001178912,0.001617866,0.0004762325,0.0001247494,0.0000405321,0.000001764257,0.00000279572,0.000167041],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2681531,"threshold_uncertainty_score":0.9968283,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4022887300727808,"score_gpt":0.47602807594491,"score_spread":0.07373934587212921,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}