{"id":"W2584951375","doi":"10.5539/ies.v10n2p84","title":"The Psychological Effect of Errors in Standardized Language Test Items on EFL Students’ Responses to the Following Item","year":2017,"lang":"en","type":"article","venue":"International Education Studies","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Psychology; Test (biology); Affect (linguistics); Social psychology; Stratified sampling; Test of English as a Foreign Language; Mathematics education; Personality; Language assessment; Statistics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.002392218,0.0000890357,0.0001499245,0.00007336915,0.00097237,0.0002421464,0.001001962,0.00002816039,0.00002151034],"category_scores_gemma":[0.01403099,0.00004818428,0.00008545371,0.00009937286,0.0001998272,0.0001030655,0.0001497794,0.00009689715,0.00002586676],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001615377,"about_ca_system_score_gemma":0.00005779083,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003047574,"about_ca_topic_score_gemma":0.001031033,"domain_scores_codex":[0.9984164,0.0002610153,0.0002120122,0.0001649184,0.0008041713,0.000141461],"domain_scores_gemma":[0.9960873,0.003242285,0.0001672565,0.0002408022,0.0002348133,0.00002754067],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002263885,0.0001993786,0.9556207,0.000002056528,0.0001291401,0.000001684546,0.02549525,0.000001387483,0.00007341569,0.001164562,0.007345395,0.009740626],"study_design_scores_gemma":[0.0004555783,0.000141387,0.9228879,0.00008534906,0.000012818,1.485948e-7,0.04296931,9.328417e-7,0.00008113779,0.0001247896,0.03317278,0.00006784722],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9662521,0.0003857653,0.000001359526,0.01681194,0.004539581,0.0003794105,0.000005939799,0.0000126714,0.01161123],"genre_scores_gemma":[0.9926299,0.0001974891,0.00002577269,0.0002079988,0.0003052985,0.0001565961,0.000001214982,0.000004563031,0.006471178],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03273279,"threshold_uncertainty_score":0.9942743,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0729122271749704,"score_gpt":0.5470174107587235,"score_spread":0.4741051835837531,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}