{"id":"W7161994237","doi":"10.82308/45504","title":"In the service of the stakeholder: a critical, mixed-method program of research in high-stakes language assessment","year":2011,"lang":"en","type":"dissertation","venue":"","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Certification; Stakeholder; Language assessment; Language proficiency; Perception; Task (project management); Service (business); Field (mathematics)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2888697,0.001720486,0.002185728,0.005000775,0.01170895,0.00752389,0.006277825,0.004879502,0.003094197],"category_scores_gemma":[0.2434704,0.002015834,0.00223379,0.004762845,0.007368493,0.005815657,0.00728259,0.004165662,0.0007783811],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01341889,"about_ca_system_score_gemma":0.02675769,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008573334,"about_ca_topic_score_gemma":0.02545261,"domain_scores_codex":[0.7056612,0.255951,0.00819006,0.01074058,0.01501788,0.004439199],"domain_scores_gemma":[0.6178423,0.290166,0.01194977,0.03326951,0.04359033,0.00318223],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.003924573,0.02155482,0.02279218,0.008762754,0.001146924,0.001601879,0.6349645,0.002542207,0.01693631,0.0225236,0.004815527,0.2584349],"study_design_scores_gemma":[0.0128782,0.06043186,0.05506828,0.01229209,0.002316079,0.001425765,0.6123653,0.012885,0.04495507,0.04426467,0.1401317,0.0009859336],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6280779,0.002202921,0.1193506,0.003447433,0.0007551981,0.2366251,0.0004810001,0.0003594339,0.008700529],"genre_scores_gemma":[0.4364684,0.0006872438,0.3182676,0.004304061,0.0002558714,0.2363847,0.0002320785,0.0001810317,0.003219086],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2888697,"threshold_uncertainty_score":0.8769503,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1571692724479921,"score_gpt":0.5265548475892885,"score_spread":0.3693855751412963,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}