{"id":"W1763331641","doi":"10.1177/0013164415584576","title":"It Might Not Make a Big DIF","year":2015,"lang":"en","type":"article","venue":"Educational and Psychological Measurement","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":130,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Differential item functioning; Item response theory; Statistics; Polytomous Rasch model; Computer science; Monte Carlo method; Econometrics; Type I and type II errors; Statistical hypothesis testing; Machine learning; Data mining; Artificial intelligence; Psychometrics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02659532,0.001002635,0.001404838,0.002353047,0.001871793,0.003930259,0.001294903,0.00190273,0.009022163],"category_scores_gemma":[0.1333622,0.0005695508,0.001387743,0.002763754,0.004509213,0.005939245,0.003245976,0.003280876,0.002697587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001817093,"about_ca_system_score_gemma":0.002064252,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002639211,"about_ca_topic_score_gemma":0.003881221,"domain_scores_codex":[0.9768153,0.0121103,0.002364609,0.003834195,0.003855458,0.001019923],"domain_scores_gemma":[0.9281565,0.04206962,0.005700678,0.01432898,0.008164345,0.001579868],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000886063,0.0005265265,0.2772764,0.001386972,0.001227916,0.001122885,0.01002279,0.00484191,0.003109658,0.2879284,0.03024185,0.3814286],"study_design_scores_gemma":[0.0002643008,0.0009543953,0.1568173,0.001069598,0.0008158433,0.003130318,0.01112898,0.02436795,0.004302763,0.7105446,0.08626635,0.0003375698],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.507188,0.003613276,0.3201028,0.04222179,0.003926119,0.001132477,0.002983561,0.0009312081,0.1179008],"genre_scores_gemma":[0.9268292,0.0006785416,0.05959286,0.006753273,0.0004100253,0.0005861601,0.0008752696,0.0001615466,0.004113044],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02659532,"threshold_uncertainty_score":0.1406511,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9118226372485674,"score_gpt":0.5444501445426981,"score_spread":0.3673724927058692,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}