{"id":"W3098210275","doi":"10.37213/cjal.2020.30649","title":"Investigating the Alignment Between the CELPIP-General Reading Test and the Canadian Language Benchmarks: A Content Validation Study","year":2020,"lang":"en","type":"article","venue":"Canadian Journal of Applied Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Computer science; Content validity; Language assessment; Language proficiency; Scale (ratio); Test validity; Index (typography); Psychology; Natural language processing; Mathematics education; Psychometrics; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001469495,0.0001554358,0.0002150548,0.00008923186,0.0008485111,0.0007206371,0.001264676,0.00005386952,0.000003809442],"category_scores_gemma":[0.003511393,0.0000810923,0.00004204872,0.0003213481,0.0002534488,0.00003831108,0.00008681544,0.0005764995,9.912129e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002061891,"about_ca_system_score_gemma":0.001182114,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.09390143,"about_ca_topic_score_gemma":0.1140182,"domain_scores_codex":[0.9986681,0.0001069815,0.0004214427,0.0001759244,0.0003151229,0.0003124368],"domain_scores_gemma":[0.9979823,0.0005196314,0.0003597277,0.000294525,0.0002510675,0.0005927067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00002489409,0.00003217714,0.04492695,0.00008773217,0.0004578338,0.0006274129,0.2979965,0.0004607175,0.00109487,0.6109385,0.01503625,0.0283161],"study_design_scores_gemma":[0.02123034,0.004270349,0.05819826,0.001924121,0.004339085,0.0008775999,0.1621976,0.09845949,0.06259581,0.4101935,0.1683002,0.007413548],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.702715,0.01178829,0.03865509,0.1989613,0.003921005,0.009389591,0.0002736981,0.0003725156,0.03392347],"genre_scores_gemma":[0.9790066,0.000002891847,0.01680822,0.003130259,0.001020205,0.000008100258,0.000003710124,0.00001202951,0.00000799843],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2762915,"threshold_uncertainty_score":0.9121323,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03015350777276179,"score_gpt":0.2519293768007262,"score_spread":0.2217758690279645,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}