{"id":"W2180011677","doi":"10.1093/deafed/env049","title":"Comprehension of Written Grammar Test: Reliability and Known-Groups Validity Study With Hearing and Deaf and Hard-of-Hearing Students","year":2015,"lang":"en","type":"article","venue":"The Journal of Deaf Studies and Deaf Education","topic":"Hearing Impairment and Communication","field":"Psychology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Psychology; Test (biology); Grammar; Comprehension; Audiology; Reliability (semiconductor); Sentence; Test validity; Vocabulary; Developmental psychology; Psychometrics; Linguistics; Natural language processing; Computer science; Medicine","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002338935,0.0001366391,0.0003806173,0.00008921337,0.0002019687,0.00003210105,0.0001318271,0.0000429871,0.000001190351],"category_scores_gemma":[0.0002423397,0.0000881478,0.00001978078,0.0001167125,0.0002282816,0.0001505978,0.0002500604,0.0002124525,2.734157e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003667636,"about_ca_system_score_gemma":0.00005509615,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004316587,"about_ca_topic_score_gemma":0.00006324128,"domain_scores_codex":[0.9985896,0.0003837874,0.0004713643,0.0001526394,0.0002670036,0.0001355781],"domain_scores_gemma":[0.9981057,0.0006688015,0.0003990343,0.0003093444,0.0004135864,0.0001035396],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003815569,0.0007087615,0.9660473,0.00008812702,0.0001526594,5.134169e-7,0.0241061,0.000004449494,0.000160209,0.00004999225,0.0001546967,0.008145566],"study_design_scores_gemma":[0.0012694,0.002285758,0.9501998,0.0001960538,0.0002438258,0.00008194964,0.04468737,0.00001663611,0.0000338578,0.0007888901,0.000111724,0.00008474927],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9892142,0.009582234,0.00002474544,0.0005812308,0.0001677965,0.0003556207,7.382441e-7,0.000006071876,0.00006737148],"genre_scores_gemma":[0.9972842,0.001917648,0.0006540457,0.00002966457,0.00006128479,0.000006817746,4.874648e-7,0.00000976932,0.00003611895],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02058128,"threshold_uncertainty_score":0.3594563,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1353991389432234,"score_gpt":0.3983073291637506,"score_spread":0.2629081902205272,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}