{"id":"W4403899111","doi":"10.5539/jel.v14n2p74","title":"Scoring Difficulty in Summary Writing Assessment: Toward the Reconstruction of Analytic Rubric","year":2024,"lang":"en","type":"article","venue":"Journal of Education and Learning","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Japan Society for the Promotion of Science","keywords":"Rubric; Psychology; Mathematics education; Evaluation methods","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001635784,0.00003919381,0.00009425863,0.0001752064,0.0001558237,0.0001359287,0.0000691813,0.00002621534,0.00006739669],"category_scores_gemma":[0.0001740094,0.00002874722,0.00004461856,0.0003512416,0.00005456858,0.0002865824,0.00001107891,0.0003582901,5.310278e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009336218,"about_ca_system_score_gemma":0.0003551876,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001256164,"about_ca_topic_score_gemma":0.00003074279,"domain_scores_codex":[0.9992026,0.0001955393,0.0002618447,0.00005893717,0.0001992742,0.00008176768],"domain_scores_gemma":[0.9994053,0.0002813173,0.0001761081,0.00002307246,0.00008350679,0.00003072636],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000005264519,0.00005485173,0.6850118,0.00004892096,0.00003448782,0.000001650948,0.05005682,0.0001071006,0.0005162152,0.003244137,0.0002306604,0.2606881],"study_design_scores_gemma":[0.0001319676,0.00004447032,0.6565338,0.0008875855,0.00004310839,0.00001854902,0.3362296,0.0007213951,0.00000832156,0.0003551447,0.004960407,0.00006571333],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9830742,0.001783599,0.00008527492,0.002071677,0.0008423335,0.00004425856,6.382867e-8,0.00000549497,0.01209314],"genre_scores_gemma":[0.9975076,0.00102319,0.0003566397,0.00001556432,0.0004193584,9.479759e-7,3.821737e-7,0.000002932725,0.0006734333],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2861727,"threshold_uncertainty_score":0.1556612,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0306549708698953,"score_gpt":0.3844111303993394,"score_spread":0.3537561595294441,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}