{"id":"W2135663628","doi":"10.1177/0265532208097336","title":"Cognitive diagnostic assessment of L2 reading comprehension ability: Validity arguments for Fusion Model application to <i>LanguEdge</i> assessment","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":175,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reading comprehension; Psychology; Cognition; Test (biology); Profiling (computer programming); Dependability; Comprehension; Reading (process); Cognitive psychology; Computer science; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0486821,0.0007586219,0.001072731,0.004358161,0.0009653732,0.003803343,0.002019352,0.001754677,0.002137555],"category_scores_gemma":[0.2508255,0.0003459126,0.001366724,0.002192287,0.004630828,0.004936526,0.005208767,0.00225757,0.0003793943],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00236771,"about_ca_system_score_gemma":0.00219577,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003331222,"about_ca_topic_score_gemma":0.001666186,"domain_scores_codex":[0.9650055,0.02072003,0.001915497,0.002848978,0.008765366,0.0007446035],"domain_scores_gemma":[0.7637811,0.1912163,0.008795157,0.0173636,0.01699496,0.001848906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003420413,0.0008683964,0.4591054,0.0006854163,0.000779748,0.0006137151,0.01071717,0.02013513,0.007388768,0.07253753,0.003212869,0.4205355],"study_design_scores_gemma":[0.0003997541,0.002202476,0.1764929,0.0005914565,0.0005031759,0.001437463,0.005373914,0.6001076,0.01589247,0.1918741,0.004769902,0.0003548207],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7499437,0.0008948687,0.2258897,0.005164768,0.0001694286,0.0006953829,0.0004554502,0.000531794,0.01625488],"genre_scores_gemma":[0.9653413,0.00006738762,0.03382828,0.0002036996,0.00003523198,0.0002243116,0.00009728307,0.00001957974,0.0001829591],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0486821,"threshold_uncertainty_score":0.2574586,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4824119830634888,"score_gpt":0.5159648486704076,"score_spread":0.03355286560691884,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}