{"id":"W4376626995","doi":"10.58379/rshg8366","title":"DIF investigations across groups of gender and academic background in a large-scale high-stakes language test ","year":2015,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Queen's University","keywords":"Test (biology); Differential item functioning; Psychology; Quality (philosophy); Reliability (semiconductor); Scale (ratio); Gender bias; Social psychology; Applied psychology; Mathematics education; Medical education; Item response theory; Developmental psychology; Psychometrics; Medicine","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04040404,0.0004531023,0.0006980777,0.003450269,0.001270335,0.001071426,0.0007638202,0.0004919579,0.001464676],"category_scores_gemma":[0.1081698,0.0001985538,0.0009474493,0.001651765,0.00162525,0.00119332,0.002145661,0.0005769636,0.0002655243],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00105017,"about_ca_system_score_gemma":0.001098234,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002136123,"about_ca_topic_score_gemma":0.00292848,"domain_scores_codex":[0.9715961,0.01478039,0.003445428,0.001990824,0.006804809,0.001382522],"domain_scores_gemma":[0.9158825,0.04578331,0.01125835,0.007485604,0.01783516,0.001755119],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004037235,0.0002618251,0.943424,0.00008044475,0.0001515599,0.0001201539,0.01263108,0.0001380861,0.001441122,0.0007012446,0.0002735658,0.04037327],"study_design_scores_gemma":[0.00003085711,0.0007357008,0.982946,0.00006945791,0.00007142502,0.0002558158,0.01026507,0.001386553,0.002000309,0.001024997,0.001177388,0.0000363734],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9965592,0.00008832804,0.002013211,0.00009053443,0.00002433701,0.0001003533,0.00005718681,0.000006016672,0.001060787],"genre_scores_gemma":[0.9984264,0.00002846117,0.001099587,0.00005161251,0.00001413347,0.00008854158,0.00008057636,0.000004211841,0.000206378],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04040404,"threshold_uncertainty_score":0.2136796,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.119758425185111,"score_gpt":0.46502313377722,"score_spread":0.3452647085921091,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}