{"id":"W2587471964","doi":"10.1007/s11145-017-9724-6","title":"Writing evaluation: rater and task effects on the reliability of writing scores for children in Grades 3 and 4","year":2017,"lang":"en","type":"article","venue":"Reading and Writing","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":44,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Eunice Kennedy Shriver National Institute of Child Health and Human Development; National Institute of Child Health and Human Development; National Institutes of Health; McMaster University","keywords":"Psychology; Psycholinguistics; Task (project management); Reliability (semiconductor); Inter-rater reliability; Literacy; Developmental psychology; Mathematics education; Cognitive psychology; Pedagogy; Cognition; Rating scale; Psychiatry","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05008892,0.0006845,0.0008819817,0.001300407,0.001026882,0.001323383,0.0006775861,0.0008459783,0.001166452],"category_scores_gemma":[0.1610583,0.0008065377,0.001601258,0.0008552693,0.001359284,0.001278584,0.001789845,0.001410375,0.0005101886],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009604183,"about_ca_system_score_gemma":0.0007227623,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003761051,"about_ca_topic_score_gemma":0.00830262,"domain_scores_codex":[0.95732,0.02658263,0.005219081,0.003365969,0.006406616,0.001105697],"domain_scores_gemma":[0.7205809,0.2120812,0.01949398,0.01708052,0.02851296,0.002250426],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.009847585,0.0007011297,0.8751912,0.0003729515,0.001871029,0.0003296336,0.02591654,0.001328252,0.02542439,0.0005348495,0.001878492,0.05660404],"study_design_scores_gemma":[0.0001818165,0.001429868,0.9839624,0.00009009449,0.000448608,0.0003161675,0.001359775,0.002424643,0.008092792,0.0001858312,0.001424429,0.000083561],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9930609,0.0003850939,0.003418445,0.00006290199,0.000106905,0.0001583574,0.0001679697,0.00006280028,0.002576713],"genre_scores_gemma":[0.9960461,0.00008106619,0.002370406,0.0000493332,0.0000266224,0.0002046488,0.0003051789,0.0001138734,0.0008027143],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05008892,"threshold_uncertainty_score":0.2648987,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03800838606232904,"score_gpt":0.363701433064812,"score_spread":0.325693047002483,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}