{"id":"W3013409274","doi":"10.5539/hes.v10n2p107","title":"Should Items and Answer Keys of Small-Scale Exams Be Published?","year":2020,"lang":"en","type":"article","venue":"Higher Education Studies","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Psychology; Statistics; Reliability (semiconductor); Item analysis; Descriptive statistics; Internal consistency; Item response theory; Mathematics education; Psychometrics; Clinical psychology; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07039478,0.000628361,0.000857998,0.002680873,0.0007559687,0.002319908,0.001661881,0.001569781,0.003314711],"category_scores_gemma":[0.5060524,0.0004153214,0.0009193434,0.003007952,0.001278574,0.004886032,0.001435406,0.001353835,0.002285563],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009072579,"about_ca_system_score_gemma":0.001585009,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007749153,"about_ca_topic_score_gemma":0.001405115,"domain_scores_codex":[0.9298173,0.0378189,0.009768507,0.00386953,0.01736998,0.001355881],"domain_scores_gemma":[0.3257174,0.482096,0.1211897,0.03558537,0.03146865,0.003942813],"domain_codex":null,"domain_gemma":"reporting","domain_candidate":"reporting","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001416369,0.000917228,0.705146,0.0009132531,0.0003180924,0.0003142339,0.002727863,0.0007870978,0.003311021,0.000965564,0.002608361,0.2805749],"study_design_scores_gemma":[0.00006286083,0.001957963,0.9792088,0.0005162041,0.0001563844,0.0007991081,0.001569882,0.001975613,0.005247931,0.001028704,0.007409724,0.00006672899],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.985246,0.001051086,0.005464669,0.001779262,0.00027138,0.0003157005,0.0006198044,0.000186544,0.005065516],"genre_scores_gemma":[0.9876047,0.0003173315,0.009693466,0.0003867056,0.0001638844,0.0002818413,0.0005096412,0.00006894486,0.0009733513],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9296052,"threshold_uncertainty_score":0.3722876,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7685150406017406,"score_gpt":0.5240299162515087,"score_spread":0.2444851243502318,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}