{"id":"W4416368436","doi":"10.2139/ssrn.5771962","title":"Do Test Scores Misrepresent Test Results? An Item-by-Item Analysis","year":2025,"lang":"","type":"preprint","venue":"SSRN Electronic Journal","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Test (biology); Aggregate (composite); Aggregate data; Ex-ante; Test data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1924541,0.001909686,0.002573925,0.007066307,0.001295498,0.0040846,0.002925612,0.003327306,0.001967299],"category_scores_gemma":[0.5031398,0.001012282,0.003951035,0.01082238,0.004889167,0.006071729,0.00360786,0.003078155,0.0009006007],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001544998,"about_ca_system_score_gemma":0.0012605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001800071,"about_ca_topic_score_gemma":0.002413546,"domain_scores_codex":[0.7300602,0.193148,0.01988423,0.01524316,0.03966595,0.00199846],"domain_scores_gemma":[0.2828778,0.5965468,0.04071816,0.05257447,0.02612318,0.001159565],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008634339,0.0003255293,0.8975884,0.000305677,0.006453,0.0001491044,0.003311672,0.001658947,0.0009127493,0.005282313,0.002960108,0.08018909],"study_design_scores_gemma":[0.0003093005,0.00185403,0.9175304,0.0003684137,0.004895007,0.001072427,0.003673963,0.02565267,0.004047598,0.03392239,0.006442645,0.0002312499],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7567784,0.002780446,0.2209925,0.003268132,0.0007003676,0.001439105,0.00183244,0.0004984672,0.01171015],"genre_scores_gemma":[0.9669447,0.0002787939,0.02927749,0.001102496,0.0002020682,0.0006255555,0.0007199729,0.0001626645,0.0006863543],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8075459,"threshold_uncertainty_score":0.9958479,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1679397505946936,"score_gpt":0.4390831264509314,"score_spread":0.2711433758562378,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}