{"id":"W2978741138","doi":"10.1177/0013164419878861","title":"A Propensity Score Method for Investigating Differential Item Functioning in Performance Assessment","year":2019,"lang":"en","type":"article","venue":"Educational and Psychological Measurement","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Propensity score matching; Differential item functioning; Matching (statistics); Context (archaeology); Psychology; Proxy (statistics); Language proficiency; Aptitude; Set (abstract data type); Test (biology); Computer science; Psychometrics; Statistics; Natural language processing; Cognitive psychology; Item response theory; Machine learning; Developmental psychology; Mathematics; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07131661,0.001307565,0.001842754,0.006095574,0.001315323,0.001812364,0.001952699,0.00171977,0.006222636],"category_scores_gemma":[0.1642554,0.0004667531,0.002588377,0.006221991,0.001677319,0.001935495,0.003180856,0.002273526,0.001253531],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009724184,"about_ca_system_score_gemma":0.002212475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001831379,"about_ca_topic_score_gemma":0.001041783,"domain_scores_codex":[0.9328659,0.05144475,0.003017652,0.005788137,0.006180018,0.0007034746],"domain_scores_gemma":[0.8926273,0.07358639,0.008486122,0.01953643,0.004950298,0.0008134057],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001463212,0.001105983,0.258122,0.0006993976,0.004426701,0.0004070619,0.001668968,0.01789333,0.002744087,0.1033756,0.01043174,0.5976618],"study_design_scores_gemma":[0.001192277,0.00427792,0.2591018,0.0005909908,0.00266752,0.001714491,0.00145706,0.3764861,0.007535659,0.280224,0.06423457,0.0005177132],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03138028,0.0002995472,0.963048,0.0002605221,0.0001364237,0.002123991,0.0009945903,0.0004867509,0.001269985],"genre_scores_gemma":[0.407565,0.0003630512,0.5757676,0.0003935051,0.0002674127,0.01009921,0.002718986,0.0002188125,0.002606371],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.07131661,"threshold_uncertainty_score":0.3771628,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7918302834411759,"score_gpt":0.5152538123684256,"score_spread":0.2765764710727503,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}