{"id":"W3031601125","doi":"10.4995/head20.2020.11303","title":"Reliability of multiple-choice versus problem-solving student exam scores in higher education: Empirical tests","year":2020,"lang":"en","type":"article","venue":"","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Saint Mary's University","funders":"","keywords":"Reliability (semiconductor); Multiple choice; Consistency (knowledge bases); Internal consistency; Test (biology); Computer science; Contrast (vision); Mathematics education; Empirical research; Confidence interval; Psychology; Statistics; Significant difference; Artificial intelligence; Mathematics; Psychometrics; Clinical psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001755753,0.00008616026,0.0001506304,0.00004554524,0.0001506596,0.00006275099,0.000353564,0.00007258642,0.001333925],"category_scores_gemma":[0.01557486,0.00007820381,0.0000406086,0.0003944678,0.0001417832,0.0005820966,0.00009809706,0.0002174017,0.00004033014],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001587129,"about_ca_system_score_gemma":0.0005933031,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006594279,"about_ca_topic_score_gemma":0.005310848,"domain_scores_codex":[0.9976772,0.0007887034,0.0003547579,0.0002983175,0.0007059253,0.0001751118],"domain_scores_gemma":[0.9943829,0.004912736,0.0001936383,0.0001780257,0.000204089,0.0001285645],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00005021318,0.0004163551,0.9722703,0.00002628465,0.000007675489,3.488331e-7,0.02037611,0.0001991837,0.00009394397,0.002157941,0.002915385,0.001486278],"study_design_scores_gemma":[0.0005494182,0.00009396817,0.9404866,0.00002473077,0.00001492646,3.419216e-8,0.004148778,0.00007147159,0.00002363666,0.0001976783,0.05428316,0.0001056533],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8598861,0.0001044761,0.000006444157,0.06047175,0.0005167032,0.0004480315,6.484868e-7,0.00008706968,0.07847886],"genre_scores_gemma":[0.9931118,0.00001380494,0.004634052,0.0007129749,0.0002777731,0.00002532193,0.000001599362,0.000006990075,0.001215697],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1332258,"threshold_uncertainty_score":0.999579,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3215689350706037,"score_gpt":0.5071320585536043,"score_spread":0.1855631234830006,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}