{"id":"W4365813887","doi":"10.1080/08957347.2023.2201703","title":"Multi-Group Generalizations of SIBTEST and Crossing-SIBTEST","year":2023,"lang":"en","type":"article","venue":"Applied Measurement in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Type I and type II errors; Differential item functioning; Statistics; Mathematics; Monte Carlo method; Logistic regression; Set (abstract data type); Population; Group (periodic table); Statistical power; Applied mathematics; Item response theory; Computer science; Psychometrics; Demography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07060982,0.001179574,0.001801027,0.004774427,0.001081421,0.001809053,0.001971327,0.001255082,0.006555918],"category_scores_gemma":[0.2286084,0.0005244266,0.002053539,0.004070001,0.003230242,0.003623162,0.003174519,0.003479569,0.0008834377],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007330428,"about_ca_system_score_gemma":0.0009908255,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006824225,"about_ca_topic_score_gemma":0.000710897,"domain_scores_codex":[0.9537348,0.03232123,0.00227607,0.004701456,0.00641697,0.0005494834],"domain_scores_gemma":[0.8039418,0.1509097,0.008712269,0.02882602,0.00672376,0.0008863772],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007251514,0.0004186223,0.1199749,0.001062124,0.001967234,0.0009160223,0.008062877,0.0359359,0.00312272,0.3862312,0.009397864,0.4321854],"study_design_scores_gemma":[0.0001494253,0.00178273,0.08669873,0.0003880616,0.0003973828,0.002458573,0.002209799,0.1421762,0.005314602,0.7414986,0.01661181,0.0003140419],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08645695,0.0004992857,0.9027918,0.0002924191,0.0002011733,0.0006685231,0.000743565,0.0008493776,0.00749685],"genre_scores_gemma":[0.6727064,0.000415015,0.3208479,0.0004243528,0.0001853425,0.002512836,0.001223851,0.0002941591,0.001390141],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.07060982,"threshold_uncertainty_score":0.3734249,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5905972101310687,"score_gpt":0.4788309874553456,"score_spread":0.1117662226757231,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}