{"id":"W4416368436","doi":"10.2139/ssrn.5771962","title":"Do Test Scores Misrepresent Test Results? An Item-by-Item Analysis","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Test (biology); Aggregate (composite); Aggregate data; Ex-ante; Test data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","bibliometrics","sts","scholarly_communication","open_science","research_integrity"],"consensus_categories":["metaresearch","metaepi_narrow","research_integrity"],"category_scores_codex":[0.09474405,0.001998978,0.004124429,0.008896057,0.002526599,0.005391168,0.01052216,0.001543674,0.000797598],"category_scores_gemma":[0.4893141,0.001610722,0.003241023,0.02406079,0.0007085816,0.001072553,0.003117536,0.01813738,0.0001389595],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003776671,"about_ca_system_score_gemma":0.01764284,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0013685,"about_ca_topic_score_gemma":0.001850839,"domain_scores_codex":[0.964962,0.005184079,0.007838031,0.005453429,0.006202431,0.01036003],"domain_scores_gemma":[0.6713502,0.3000208,0.01219029,0.00921409,0.004999149,0.002225363],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006044073,0.00188101,0.4899717,0.00005154097,0.005487555,0.00007068883,0.0006031179,0.005108195,0.000352481,0.002512339,0.007857741,0.4854992],"study_design_scores_gemma":[0.005887951,0.005505681,0.1298236,0.001081216,0.00787602,0.0012295,0.02085236,0.0432903,0.0004223567,0.7549241,0.02447044,0.004636534],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3355008,0.192676,0.3726165,0.01799306,0.01389661,0.004170121,0.006346202,0.0008180168,0.05598262],"genre_scores_gemma":[0.8473468,0.09186768,0.01028205,0.0003576391,0.002471933,0.00006760898,0.0002150878,0.0001218811,0.04726934],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7524117,"threshold_uncertainty_score":0.9997525,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1679397505946936,"score_gpt":0.4390831264509314,"score_spread":0.2711433758562378,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}