{"id":"W1797460826","doi":"","title":"Putting Rubrics to the Test: The Effect of Rubric-Referenced Peer Assessment on EFL Learners’ Evaluation of Speaking","year":2013,"lang":"en","type":"article","venue":"Journal of academic and applied studies","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Rubric; Formative assessment; Peer assessment; Psychology; Presupposition; Test (biology); Mathematics education; Peer feedback; Pedagogy; Computer science; Linguistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007503666,0.0001128572,0.0003264719,0.00006833572,0.0003868455,0.00003410713,0.0003164245,0.00006696191,0.0000173867],"category_scores_gemma":[0.001446532,0.00005376704,0.00006808653,0.0002557083,0.0001978955,0.00009362892,0.000107203,0.0004581747,0.000002831277],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007181471,"about_ca_system_score_gemma":0.00006773703,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002997517,"about_ca_topic_score_gemma":0.000009684816,"domain_scores_codex":[0.9974374,0.0002794664,0.0004463455,0.0001023878,0.001555467,0.0001788992],"domain_scores_gemma":[0.9963809,0.002447677,0.0006585619,0.00007766084,0.0003929681,0.00004224239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001515971,0.0001541134,0.3654803,0.0001451578,0.001334657,7.607844e-7,0.2398491,0.002144053,0.02133068,0.0115542,0.03191891,0.3259365],"study_design_scores_gemma":[0.002995047,0.00156215,0.5938902,0.0006794518,0.001236186,0.000003154069,0.3816053,0.0004199563,0.004530948,0.007145797,0.005562459,0.0003693429],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9786103,0.0005669112,0.00003212805,0.008382608,0.0001755203,0.0005949338,0.000001006969,0.000003660326,0.01163287],"genre_scores_gemma":[0.9984835,0.0007010775,0.0002007747,0.000113956,0.0003370497,0.00002174356,1.480892e-7,0.000005160949,0.0001365672],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3255671,"threshold_uncertainty_score":0.2975342,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08647179667909898,"score_gpt":0.4307727206229517,"score_spread":0.3443009239438527,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}