{"id":"W4210694249","doi":"10.1016/j.stueduc.2022.101126","title":"Item wording effects in self-report measures and reading achievement: Does removing careless respondents help?","year":2022,"lang":"en","type":"article","venue":"Studies In Educational Evaluation","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Psychology; Mathematics education; Academic achievement; Achievement test; Standardized test; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.241551,0.001278105,0.002030071,0.002149851,0.00216518,0.002463259,0.003321736,0.003544767,0.003129437],"category_scores_gemma":[0.4629652,0.001459272,0.004516035,0.002532452,0.00386732,0.006360759,0.002083638,0.003431642,0.0006470223],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001394966,"about_ca_system_score_gemma":0.00271786,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004975556,"about_ca_topic_score_gemma":0.01014684,"domain_scores_codex":[0.8102607,0.1362987,0.02551362,0.008747214,0.01752269,0.001657162],"domain_scores_gemma":[0.342504,0.5609976,0.02859274,0.04600734,0.01991794,0.001980381],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002858169,0.001851829,0.695691,0.001948952,0.002746594,0.0002647895,0.0142723,0.0005270544,0.003095735,0.003544669,0.005233288,0.2679655],"study_design_scores_gemma":[0.0006522964,0.002739404,0.9588671,0.002549889,0.003280166,0.0006680186,0.004505038,0.002246629,0.009854341,0.00549429,0.008910364,0.0002324809],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9271633,0.007903477,0.04878245,0.006741693,0.001259785,0.001231604,0.0006427927,0.000289084,0.005985887],"genre_scores_gemma":[0.9501703,0.00105609,0.0415628,0.002758728,0.0002712315,0.001252824,0.0007810126,0.0001726229,0.001974265],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.758449,"threshold_uncertainty_score":0.9353026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4604167025598831,"score_gpt":0.5511538330204064,"score_spread":0.09073713046052334,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}