{"id":"W3114083521","doi":"10.20982/tqmp.16.5.p467","title":"Inter-Rater Agreement, Data Reliability, and The Crisis of Confidence in Psychological Research","year":2020,"lang":"en","type":"article","venue":"The Quantitative Methods for Psychology","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Inter-rater reliability; Psychology; Reliability (semiconductor); Clinical psychology; Developmental psychology; Physics; Thermodynamics; Rating scale","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"reproducibility","study_design":"theoretical_or_conceptual","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":["metaresearch"],"domain":"reproducibility","study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.8387451,0.001799153,0.007015957,0.01546658,0.005104586,0.01577303,0.008078054,0.006691587,0.001704806],"category_scores_gemma":[0.9238808,0.003811107,0.004380585,0.01375589,0.03368133,0.01776054,0.0148539,0.0138871,0.0006059209],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009839998,"about_ca_system_score_gemma":0.01423313,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004437231,"about_ca_topic_score_gemma":0.004660935,"domain_scores_codex":[0.06646398,0.7664704,0.08493546,0.02189667,0.05853639,0.001697102],"domain_scores_gemma":[0.02201076,0.854207,0.04673659,0.0438639,0.03213997,0.001041771],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003215576,0.0003647398,0.09788165,0.04866394,0.01785164,0.00116161,0.12079,0.006069737,0.003248306,0.2924888,0.0275626,0.3807015],"study_design_scores_gemma":[0.0008937297,0.001708193,0.06041504,0.05066368,0.003546747,0.002547541,0.019767,0.02334938,0.006280587,0.7634743,0.06561746,0.001736383],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06673088,0.09273695,0.6982208,0.1078177,0.009659583,0.004226127,0.001160146,0.0008705286,0.01857731],"genre_scores_gemma":[0.6463474,0.01016357,0.3160843,0.01345116,0.002893394,0.009181008,0.000430095,0.0005272831,0.000921757],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1612549,"threshold_uncertainty_score":0.1988561,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9696765528338096,"score_gpt":0.7788110470329636,"score_spread":0.190865505800846,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}