{"id":"W2155243985","doi":"10.20982/tqmp.08.1.p023","title":"Computing Inter-Rater Reliability for Observational Data: An Overview and Tutorial","year":2012,"lang":"en","type":"article","venue":"Tutorials in Quantitative Methods for Psychology","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":3862,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute on Alcohol Abuse and Alcoholism","keywords":"Observational study; Reliability (semiconductor); Computer science; Syntax; Consistency (knowledge bases); Class (philosophy); Statistics; Multiple comparisons problem; Inter-rater reliability; Econometrics; Mathematics; Artificial intelligence; Power (physics); Rating scale","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06979387,0.003653719,0.003127687,0.01618263,0.001209161,0.003807294,0.003599142,0.00278731,0.01322493],"category_scores_gemma":[0.1407642,0.002858866,0.003396574,0.01500267,0.001947439,0.008113082,0.003334882,0.004965792,0.01857237],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001506085,"about_ca_system_score_gemma":0.003299826,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001694989,"about_ca_topic_score_gemma":0.002094904,"domain_scores_codex":[0.9320188,0.04221538,0.008408046,0.003793425,0.0131046,0.0004597739],"domain_scores_gemma":[0.8394294,0.123764,0.006926848,0.006772066,0.02234066,0.0007670119],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00009429502,0.0001778791,0.002026607,0.006763141,0.0003309888,0.0002702106,0.00205932,0.003273993,0.005289653,0.0449449,0.06619325,0.8685756],"study_design_scores_gemma":[0.0001649573,0.000791866,0.01365564,0.01097351,0.0004659924,0.005431902,0.001721629,0.05035447,0.01997095,0.2296043,0.6657985,0.001066321],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0004352524,0.00898443,0.9843138,0.0006793733,0.000236491,0.0006642518,0.0004387389,0.001761443,0.002486196],"genre_scores_gemma":[0.003187956,0.01208352,0.9792595,0.0002245173,0.0004559548,0.002216479,0.0007712265,0.0007829313,0.001017835],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9302061,"threshold_uncertainty_score":0.3691097,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8909602875101015,"score_gpt":0.7046596688682544,"score_spread":0.1863006186418471,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}