{"id":"W3098210275","doi":"10.37213/cjal.2020.30649","title":"Investigating the Alignment Between the CELPIP-General Reading Test and the Canadian Language Benchmarks: A Content Validation Study","year":2020,"lang":"en","type":"article","venue":"Canadian Journal of Applied Linguistics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Computer science; Content validity; Language assessment; Language proficiency; Scale (ratio); Test validity; Index (typography); Psychology; Natural language processing; Mathematics education; Psychometrics; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04761833,0.0005368569,0.0007330226,0.00648854,0.005743155,0.004443394,0.003759817,0.0009355217,0.001814276],"category_scores_gemma":[0.2038252,0.0005287447,0.0007198032,0.009112138,0.005678259,0.00198396,0.004404314,0.002296749,0.0003318536],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.05214524,"about_ca_system_score_gemma":0.10332,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.8626668,"about_ca_topic_score_gemma":0.8894689,"domain_scores_codex":[0.955479,0.008976103,0.003019326,0.003673051,0.0266168,0.002235619],"domain_scores_gemma":[0.79247,0.04913483,0.01431442,0.01138501,0.1276872,0.005008542],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007341724,0.001149628,0.6882163,0.0008057039,0.000276622,0.0004154709,0.09544756,0.002185206,0.003306299,0.01511715,0.01003663,0.1823093],"study_design_scores_gemma":[0.00009474813,0.0004002531,0.947364,0.0003563328,0.0001122044,0.00008791162,0.02493392,0.002551902,0.002883216,0.001418422,0.01967017,0.0001269605],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9628626,0.0003324865,0.006160242,0.001147385,0.0001018603,0.002117676,0.002016392,0.0000997181,0.02516178],"genre_scores_gemma":[0.9809761,0.0001848171,0.01148388,0.0004033409,0.00001968696,0.002293994,0.002159877,0.00008038702,0.002397846],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1373332,"threshold_uncertainty_score":0.378342,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03015350777276179,"score_gpt":0.2519293768007262,"score_spread":0.2217758690279645,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}