{"id":"W2100487390","doi":"10.1177/0022022105284492","title":"Evaluating the Effectiveness of Two-Stage Testing on English and French Versions of a Science Achievement Test","year":2006,"lang":"en","type":"article","venue":"Journal of Cross-Cultural Psychology","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Test (biology); Ethnic group; Achievement test; Psychology; Differential item functioning; Sample (material); Mathematics education; Standardized test; Developmental psychology; Item response theory; Psychometrics; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04052384,0.001286121,0.0008359863,0.001704223,0.0006153663,0.001252685,0.001420622,0.001323748,0.0009971586],"category_scores_gemma":[0.1606656,0.0004637166,0.001429745,0.0008405741,0.0005347461,0.001646243,0.001036569,0.0007597901,0.0003566718],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001919712,"about_ca_system_score_gemma":0.002058727,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01055895,"about_ca_topic_score_gemma":0.01831963,"domain_scores_codex":[0.9530426,0.03240041,0.00291578,0.002214497,0.00855738,0.0008693304],"domain_scores_gemma":[0.7320135,0.226745,0.01391388,0.005888164,0.01828966,0.00314982],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.02286604,0.008490247,0.5597416,0.0006021847,0.001300411,0.000474421,0.004276587,0.004878997,0.01117431,0.0003595127,0.001272816,0.3845629],"study_design_scores_gemma":[0.001751727,0.07115935,0.85647,0.0002990135,0.001492401,0.0008651128,0.001731277,0.0255733,0.03523742,0.0002866338,0.004802995,0.0003308051],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9944071,0.0004402696,0.002919991,0.0001609371,0.00004328964,0.0003239693,0.0001148768,0.0000962304,0.001493392],"genre_scores_gemma":[0.9805694,0.0002861387,0.01728099,0.0001668878,0.00004704409,0.0002496098,0.0004815726,0.0000301204,0.0008882345],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04052384,"threshold_uncertainty_score":0.2143131,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4485049263862267,"score_gpt":0.5900866364199733,"score_spread":0.1415817100337465,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}