{"id":"W2122573908","doi":"10.1177/0013164415574086","title":"A Ratio Test of Interrater Agreement With High Specificity","year":2015,"lang":"en","type":"article","venue":"Educational and Psychological Measurement","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Trois-Rivières; University of Ottawa","funders":"","keywords":"Inter-rater reliability; Agreement; Psychology; Validation test; Test validity; Test (biology); Psychometrics; Statistics; Clinical psychology; Developmental psychology; Rating scale; Mathematics; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08930669,0.001636676,0.002146473,0.00701717,0.00134272,0.004106593,0.002810194,0.002833202,0.01219856],"category_scores_gemma":[0.482363,0.0007860598,0.002245834,0.005844935,0.004403617,0.006411538,0.00395574,0.002677314,0.004363929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001103234,"about_ca_system_score_gemma":0.002143837,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008028134,"about_ca_topic_score_gemma":0.0007070383,"domain_scores_codex":[0.8368983,0.09302918,0.01229528,0.02226278,0.03376995,0.001744606],"domain_scores_gemma":[0.4197406,0.5005758,0.02049641,0.02779237,0.0298749,0.001519936],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002502841,0.0006215707,0.1604556,0.003281934,0.005132086,0.001136322,0.004428725,0.02090991,0.007461214,0.1411245,0.02979428,0.6231511],"study_design_scores_gemma":[0.001088425,0.00583754,0.1595997,0.00240981,0.003218025,0.01390871,0.004993294,0.3505156,0.02343443,0.3554716,0.07838488,0.001138113],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07318521,0.001435754,0.8728276,0.001066629,0.0008310318,0.002265176,0.001961529,0.002011209,0.04441579],"genre_scores_gemma":[0.7420636,0.0004019749,0.2474964,0.0007377267,0.0004470583,0.002977834,0.001145745,0.0004992673,0.004230266],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.08930669,"threshold_uncertainty_score":0.4723046,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4746679401283067,"score_gpt":0.4088811515647052,"score_spread":0.06578678856360154,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}