{"id":"W2096343999","doi":"10.1191/0265532203lt248oa","title":"Does item-level DIF manifest itself in scale-level analyses? Implications for translating language tests","year":2003,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Differential item functioning; Equivalence (formal languages); Psychology; Scale (ratio); Measurement invariance; Item response theory; Item analysis; Statistics; Test (biology); Psychometrics; Developmental psychology; Linguistics; Structural equation modeling; Mathematics; Confirmatory factor analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1741679,0.001668009,0.00250879,0.003758,0.002100031,0.006615139,0.003071767,0.002287974,0.003827184],"category_scores_gemma":[0.665731,0.0009926538,0.002300151,0.004827591,0.01132217,0.01101788,0.003984601,0.004101844,0.0009048782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003611376,"about_ca_system_score_gemma":0.003728779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002854875,"about_ca_topic_score_gemma":0.002577238,"domain_scores_codex":[0.7531396,0.2154706,0.009110659,0.005530565,0.01497175,0.001776866],"domain_scores_gemma":[0.389396,0.5342252,0.01715731,0.03871311,0.01906081,0.001447509],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002117026,0.0009511592,0.2104652,0.002371002,0.0007993317,0.002997644,0.04360043,0.01866808,0.005589718,0.1542625,0.005460616,0.5527174],"study_design_scores_gemma":[0.0006893126,0.002445092,0.1322249,0.002315832,0.0004634399,0.002301843,0.04180646,0.09224727,0.01549298,0.6971639,0.01245412,0.0003947009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4390763,0.003787065,0.4727606,0.04856186,0.001222171,0.001585674,0.0006147871,0.001363012,0.03102848],"genre_scores_gemma":[0.8519405,0.0006638839,0.1421973,0.002825819,0.0002853731,0.0008046106,0.0001775433,0.0002399375,0.0008651118],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1741679,"threshold_uncertainty_score":0.9210988,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6773012489253456,"score_gpt":0.5390986227521983,"score_spread":0.1382026261731473,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}