{"id":"W2167816954","doi":"10.2466/pr0.101.3.1001-1010","title":"Four Multi-Item Interrater Agreement Options: Comparisons and Outcomes","year":2007,"lang":"en","type":"article","venue":"Psychological Reports","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Inter-rater reliability; Intraclass correlation; Psychology; Agreement; Statistics; Psychometrics; Clinical psychology; Rating scale; Developmental psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01139015,0.0002128928,0.0004071374,0.0001532252,0.0001922827,0.0002151846,0.0003278616,0.0001133818,0.0009971563],"category_scores_gemma":[0.004059266,0.0001211985,0.0001719268,0.0002932073,0.0002047211,0.0001533051,0.0001882124,0.0002436316,0.0002158637],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000469582,"about_ca_system_score_gemma":0.000007256285,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001326875,"about_ca_topic_score_gemma":0.00006245333,"domain_scores_codex":[0.9954126,0.0001900471,0.001553884,0.0009201657,0.001508237,0.0004150136],"domain_scores_gemma":[0.9973308,0.0008487334,0.0004244425,0.0008895033,0.0002433889,0.0002630815],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0000334806,0.0006902377,0.9470958,0.000002615125,0.00002532262,0.0003064477,0.0001366826,0.00001282505,0.0006065703,0.0002052433,0.01349168,0.03739312],"study_design_scores_gemma":[0.0002833764,0.0001492317,0.9604405,0.00001307127,0.00001004169,0.0001227808,0.0003398104,0.00007295467,0.00008718921,0.006526907,0.03178417,0.0001699378],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9337136,0.0001694603,0.05297678,0.002450367,0.00126441,0.0004450097,0.000001868886,0.00005907413,0.008919481],"genre_scores_gemma":[0.9840152,0.00002269816,0.01204209,0.001227862,0.00004853228,0.00002161067,0.000001897622,0.000005223526,0.002614943],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0503016,"threshold_uncertainty_score":0.9999161,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4302605755541647,"score_gpt":0.4916184474274782,"score_spread":0.06135787187331354,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}