{"id":"W7161994237","doi":"10.82308/45504","title":"In the service of the stakeholder: a critical, mixed-method program of research in high-stakes language assessment","year":2011,"lang":"en","type":"dissertation","venue":"","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Certification; Stakeholder; Language assessment; Language proficiency; Perception; Task (project management); Service (business); Field (mathematics)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006148646,0.0001620465,0.000356447,0.0002675882,0.00018542,0.00008730188,0.001394053,0.0002726598,0.0004755096],"category_scores_gemma":[0.0002660995,0.000096981,0.00009619072,0.001952752,0.000261461,0.0001514443,0.00007371409,0.0007328272,0.000004676844],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000850428,"about_ca_system_score_gemma":0.0006446353,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.07215026,"about_ca_topic_score_gemma":0.2640242,"domain_scores_codex":[0.9946705,0.002310426,0.0005172209,0.0002987173,0.001741784,0.0004613553],"domain_scores_gemma":[0.9973505,0.001554684,0.0001749212,0.0004343467,0.0004485669,0.00003702773],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001306188,0.004075334,0.03949874,0.001425985,0.00008267741,0.00002318306,0.4430131,0.000001689604,0.001097557,0.4940521,0.002914831,0.01368415],"study_design_scores_gemma":[0.0003555201,0.0001148053,0.2338833,0.0003759823,0.00003378461,1.293845e-7,0.7591874,0.00001123219,0.0004935815,0.004293876,0.001095149,0.0001552275],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7828332,0.000168977,0.000004418476,0.002243186,0.0005527579,0.001903309,0.00001536583,0.00001989521,0.2122589],"genre_scores_gemma":[0.9891369,0.00008797697,0.00350445,0.00008519189,0.00009748709,0.0003506122,0.00004282962,0.00001957943,0.006674937],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4897582,"threshold_uncertainty_score":0.9340284,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1571692724479921,"score_gpt":0.5265548475892885,"score_spread":0.3693855751412963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}