{"id":"W1645137987","doi":"10.18806/tesl.v18i2.909","title":"Using the Canadian Language Benchmarks (CLB) to Benchmark College Programs/Courses and Language Proficiency Tests","year":2001,"lang":"en","type":"article","venue":"TESL Canada Journal","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Benchmarking; Test of English as a Foreign Language; Benchmark (surveying); Computer science; Language assessment; Test (biology); Process (computing); Language proficiency; Mathematics education; Foreign language; Order (exchange); Natural language processing; Psychology; Programming language","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005251698,0.0001960801,0.0001794845,0.0001705696,0.002228399,0.0006138564,0.0002795577,0.00004232199,0.01853336],"category_scores_gemma":[0.0001686805,0.0001366031,0.00004441502,0.0001197716,0.0001122712,0.0001497152,0.00003503248,0.0006138171,0.000002750515],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003722945,"about_ca_system_score_gemma":0.001416399,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.9736115,"about_ca_topic_score_gemma":0.9986203,"domain_scores_codex":[0.9984453,0.0001109713,0.0002383477,0.0001959544,0.0003810073,0.0006284339],"domain_scores_gemma":[0.9990365,0.0000864804,0.0001103093,0.0001936701,0.00009954776,0.0004734721],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008239348,0.0002040087,0.02345546,0.0001092874,0.0004035386,0.01227742,0.5893062,0.0004290666,0.0007880229,0.02595509,0.1689701,0.1780194],"study_design_scores_gemma":[0.0005840406,0.0002267469,0.006047484,0.0002227742,0.0001086605,0.002960009,0.3389149,0.0008364546,0.00002696313,0.0000716258,0.6492457,0.0007545779],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9582464,0.009835767,0.000003391105,0.001980913,0.0005397162,0.0002496339,0.00004655658,0.00002632871,0.02907135],"genre_scores_gemma":[0.9829116,0.000003761978,0.000147433,0.0132819,0.001003905,0.00000624407,0.00001583584,0.00002578335,0.002603562],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4802757,"threshold_uncertainty_score":0.9990706,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02763543203184957,"score_gpt":0.2606801595379964,"score_spread":0.2330447275061468,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}