{"id":"W4416309139","doi":"10.2139/ssrn.5757617","title":"Do Test Scores Misrepresent Test Results? An Item-by-Item Analysis","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Test (biology); Download; Aggregate (composite); Aggregate data; Ex-ante; Test data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.04958668,0.0007924543,0.001849527,0.004469583,0.0007620134,0.002384741,0.005986365,0.000676178,0.0002111562],"category_scores_gemma":[0.3669701,0.0005860985,0.001345886,0.009228374,0.0002086904,0.0003885575,0.00168986,0.008283379,0.00004838013],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00137461,"about_ca_system_score_gemma":0.005868738,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006307549,"about_ca_topic_score_gemma":0.001199097,"domain_scores_codex":[0.9847581,0.002017716,0.003279568,0.002470921,0.003416605,0.004057083],"domain_scores_gemma":[0.8524011,0.1355773,0.004574293,0.004652488,0.002063449,0.0007313663],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002141792,0.0007616525,0.6487758,0.0000239044,0.00243965,0.00004171119,0.0003200854,0.00436428,0.0002006144,0.002449195,0.02205625,0.3183527],"study_design_scores_gemma":[0.001634608,0.001106365,0.08706318,0.0002359368,0.001599623,0.0002640818,0.004511985,0.01144937,0.0001646024,0.8791866,0.01134343,0.001440186],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5232028,0.1027706,0.2769069,0.01643154,0.008516939,0.00239255,0.004300436,0.0009424158,0.06453577],"genre_scores_gemma":[0.938846,0.01556626,0.009297022,0.0002956841,0.001680231,0.00004317643,0.0001730556,0.00006801099,0.03403059],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8767374,"threshold_uncertainty_score":0.9997671,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2133188991517597,"score_gpt":0.4592982282682351,"score_spread":0.2459793291164755,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}