{"id":"W1982491995","doi":"10.3138/cmlr.1723.359","title":"Relating a Reading Comprehension Test to the CEFR Levels: A Case of Standard Setting in Practice with Focus on Judges and Items","year":2013,"lang":"en","type":"article","venue":"Canadian Modern Language Review/ La Revue canadienne des langues vivantes","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Context (archaeology); Focus (optics); Set (abstract data type); Reading (process); Psychology; Reading comprehension; Language assessment; Computer science; Comprehension; Mathematics education; Linguistics; Medical education; Natural language processing; Medicine; History","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1654132,0.0006766639,0.001068426,0.005150731,0.01069922,0.007492677,0.006112,0.008434825,0.002054312],"category_scores_gemma":[0.399516,0.001295967,0.0008678036,0.00531844,0.01581365,0.005627257,0.01169182,0.01165465,0.0004188905],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01232351,"about_ca_system_score_gemma":0.01011729,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01680537,"about_ca_topic_score_gemma":0.01733159,"domain_scores_codex":[0.6718169,0.224456,0.02399518,0.01576044,0.05633068,0.007640786],"domain_scores_gemma":[0.5719719,0.331446,0.02668319,0.02348941,0.04109805,0.005311508],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.0004584259,0.0007040311,0.1298662,0.0007294954,0.00009195415,0.05112738,0.5433907,0.001948843,0.005220173,0.1122352,0.01100837,0.1432193],"study_design_scores_gemma":[0.0003218764,0.00191906,0.1654714,0.005173616,0.0003500846,0.06619672,0.3996626,0.02921136,0.03479056,0.1709335,0.1249701,0.0009992498],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8074923,0.001676442,0.08346946,0.04718952,0.001032498,0.0007968285,0.000123502,0.0002698449,0.05794961],"genre_scores_gemma":[0.9753644,0.0001580761,0.02014764,0.002209288,0.0001057179,0.0002519854,0.00003393534,0.0000517167,0.001677336],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1654132,"threshold_uncertainty_score":0.8747993,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01766055693986884,"score_gpt":0.2359235001311821,"score_spread":0.2182629431913132,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}