{"id":"W3031601125","doi":"10.4995/head20.2020.11303","title":"Reliability of multiple-choice versus problem-solving student exam scores in higher education: Empirical tests","year":2020,"lang":"en","type":"article","venue":"","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Saint Mary's University","funders":"","keywords":"Reliability (semiconductor); Multiple choice; Consistency (knowledge bases); Internal consistency; Test (biology); Computer science; Contrast (vision); Mathematics education; Empirical research; Confidence interval; Psychology; Statistics; Significant difference; Artificial intelligence; Mathematics; Psychometrics; Clinical psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1043165,0.0005385692,0.0008380944,0.003042608,0.0006457292,0.001582818,0.001905176,0.001106099,0.002003549],"category_scores_gemma":[0.3430911,0.0004698005,0.0008566582,0.003048164,0.00243729,0.002357448,0.001799691,0.00109643,0.0007818546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006150317,"about_ca_system_score_gemma":0.0007903633,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001237896,"about_ca_topic_score_gemma":0.001822215,"domain_scores_codex":[0.9008233,0.06779486,0.007340272,0.005552648,0.01740176,0.001087093],"domain_scores_gemma":[0.4178901,0.4800578,0.03925394,0.02614184,0.0341421,0.002514086],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006383506,0.0007105354,0.9428775,0.0002220476,0.0003988602,0.00004251657,0.002636598,0.001232191,0.0009033922,0.0007565275,0.000922897,0.0486586],"study_design_scores_gemma":[0.00007010865,0.001298969,0.9896134,0.00009672208,0.0000896322,0.00009050922,0.0009330329,0.004005354,0.002316989,0.0007401657,0.0007086965,0.00003652783],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9881286,0.0003572345,0.006540574,0.0001802672,0.00004708357,0.0001857968,0.0002008946,0.0000446485,0.00431486],"genre_scores_gemma":[0.9948331,0.00008373211,0.00431939,0.00003856987,0.00003638418,0.0001533417,0.0002129087,0.00002292574,0.000299515],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8956835,"threshold_uncertainty_score":0.5516848,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3215689350706037,"score_gpt":0.5071320585536043,"score_spread":0.1855631234830006,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}