{"id":"W1999929076","doi":"10.1080/10627197.2011.584042","title":"Voices From Test-Takers: Further Evidence for Language Assessment Validation and Use","year":2011,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":56,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"Guangdong University of Foreign Studies; Ministry of Education, India; Ministry of Earth Sciences","keywords":"Test (biology); Language assessment; Psychology; Test validity; Scale (ratio); Coding (social sciences); Test score; Computer science; Psychometrics; Mathematics education; Standardized test; Developmental psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1679558,0.0006862718,0.001460696,0.003745166,0.005046748,0.007357724,0.003083837,0.0028186,0.002667543],"category_scores_gemma":[0.5089094,0.0008900915,0.001389184,0.002010464,0.009811517,0.00712669,0.01106563,0.004901371,0.0006093996],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002931892,"about_ca_system_score_gemma":0.003802686,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004104786,"about_ca_topic_score_gemma":0.00371341,"domain_scores_codex":[0.6754895,0.2317346,0.02039349,0.008941623,0.05831548,0.005125324],"domain_scores_gemma":[0.2965513,0.5815055,0.04345421,0.02513368,0.04700278,0.006352395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0003438988,0.0001300095,0.07841828,0.0005830973,0.0001187508,0.000871143,0.8657131,0.000051526,0.002517341,0.00100734,0.0007533315,0.04949212],"study_design_scores_gemma":[0.0001141921,0.00116446,0.1429631,0.002713463,0.0001657072,0.003143332,0.8113228,0.001001511,0.007190667,0.003345808,0.02659477,0.0002802153],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9781256,0.002971441,0.006102022,0.007367013,0.0001826752,0.0001152881,0.0001072719,0.00004695568,0.004981703],"genre_scores_gemma":[0.9945524,0.0007247289,0.00183059,0.001846136,0.0001106038,0.0001243077,0.00008408133,0.00006191035,0.0006652317],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1679558,"threshold_uncertainty_score":0.8882459,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1441623853519503,"score_gpt":0.4552257976525222,"score_spread":0.3110634123005719,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}