{"id":"W3013409274","doi":"10.5539/hes.v10n2p107","title":"Should Items and Answer Keys of Small-Scale Exams Be Published?","year":2020,"lang":"en","type":"article","venue":"Higher Education Studies","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Test (biology); Psychology; Statistics; Reliability (semiconductor); Item analysis; Descriptive statistics; Internal consistency; Item response theory; Mathematics education; Psychometrics; Clinical psychology; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00178103,0.0001299734,0.0004089714,0.0002297593,0.0001201598,0.0001369504,0.0003568804,0.00005681313,0.0004007972],"category_scores_gemma":[0.01213443,0.00008854867,0.0000587686,0.001875283,0.0001834923,0.0002416931,0.0002175721,0.0001114756,0.00001486399],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000136391,"about_ca_system_score_gemma":0.00005863985,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004533493,"about_ca_topic_score_gemma":0.000004370905,"domain_scores_codex":[0.9979934,0.0002200835,0.0005756911,0.0004788408,0.0005610447,0.0001709442],"domain_scores_gemma":[0.9905728,0.007866879,0.0003398742,0.0003553542,0.0007381524,0.0001269581],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00001835308,0.0001123906,0.5691988,0.00004765618,0.00007777428,4.969586e-7,0.004464641,0.00000774581,0.000407178,0.002703373,0.3656534,0.05730817],"study_design_scores_gemma":[0.0001911538,0.000091451,0.6540422,0.00001868496,0.00002359307,0.000001551675,0.01645247,0.00001670435,0.0002629883,0.007586166,0.321159,0.0001540153],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9234954,0.01441402,0.0005614065,0.04059282,0.003659203,0.000165129,0.000007227959,0.00006367991,0.01704116],"genre_scores_gemma":[0.9729886,0.0003258797,0.01543269,0.003767604,0.0005195427,0.00003125965,0.000002020724,0.00001116273,0.006921259],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08484334,"threshold_uncertainty_score":0.9961868,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7685150406017406,"score_gpt":0.5240299162515087,"score_spread":0.2444851243502318,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}