{"id":"W2158712864","doi":"10.5206/cjsotl-rcacea.2011.2.4","title":"Examination of the Quality of Multiple-choice Items on Classroom Tests","year":2011,"lang":"en","type":"article","venue":"The Canadian Journal for the Scholarship of Teaching and Learning","topic":"Evaluation of Teaching Practices","field":"Social Sciences","cited_by":120,"is_retracted":false,"has_abstract":true,"ca_institutions":"Brock University","funders":"Brock University","keywords":"Multiple choice; Psychology; Selection (genetic algorithm); Quality (philosophy); Test (biology); Social psychology; Statistical analysis; Statistics; Significant difference; Mathematics; Computer science; Biology; Artificial intelligence; Philosophy","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02254064,0.000585068,0.0008252017,0.003729181,0.0002672631,0.001215014,0.0009121117,0.0005704171,0.001541172],"category_scores_gemma":[0.1525902,0.0003147451,0.0006595172,0.00274132,0.00106975,0.001029513,0.0009115796,0.0008237962,0.0004194884],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005223936,"about_ca_system_score_gemma":0.0004793502,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001422849,"about_ca_topic_score_gemma":0.001735031,"domain_scores_codex":[0.9803361,0.00645276,0.002517455,0.001442126,0.008574748,0.000676755],"domain_scores_gemma":[0.6619108,0.2393023,0.03427389,0.01480328,0.04670841,0.003001234],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000367929,0.000232084,0.9200658,0.0001498074,0.0002504931,0.0001824872,0.002473814,0.001115258,0.007536847,0.0001729861,0.0003406159,0.06711176],"study_design_scores_gemma":[0.00001883169,0.0008111818,0.9869325,0.00003840152,0.0000385868,0.0002822712,0.0007996553,0.003885645,0.006131992,0.0002330187,0.0007910547,0.00003682554],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9925984,0.0001695573,0.005971132,0.00006745372,0.00001593487,0.00005571078,0.0001829781,0.00005255646,0.0008862985],"genre_scores_gemma":[0.9952147,0.00004600305,0.004099794,0.0000208486,0.00001589974,0.00002763284,0.0002425979,0.00002289163,0.000309682],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9985772,"threshold_uncertainty_score":0.1192077,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3066813021309989,"score_gpt":0.440101905591797,"score_spread":0.133420603460798,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}