{"id":"W2890473533","doi":"10.1152/advan.00186.2016","title":"Item statistics derived from three-option versions of multiple-choice questions are usually as robust as four- or five-option versions: implications for exam design","year":2018,"lang":"en","type":"article","venue":"AJP Advances in Physiology Education","topic":"Innovative Teaching Methods","field":"Social Sciences","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Division of Biological Infrastructure; University of California, Irvine; National Science Foundation","keywords":"Multiple choice; Point (geometry); Class (philosophy); Quarter (Canadian coin); Mathematics education; Psychology; Index (typography); Statistics; Computer science; Mathematics; Artificial intelligence; Significant difference","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2808951,0.001588932,0.003311814,0.006130931,0.001145813,0.004607385,0.003702272,0.002306399,0.005403406],"category_scores_gemma":[0.7356948,0.001280319,0.002991118,0.009860846,0.004322227,0.006332521,0.003082007,0.003223082,0.002588096],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001554873,"about_ca_system_score_gemma":0.001053447,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001271047,"about_ca_topic_score_gemma":0.001979015,"domain_scores_codex":[0.5889046,0.3285327,0.02807701,0.01515773,0.03800122,0.001326719],"domain_scores_gemma":[0.08049864,0.8505103,0.01982497,0.03053153,0.01786781,0.0007667897],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0101828,0.001688463,0.3338784,0.004334708,0.009833878,0.0009594106,0.007030848,0.01404165,0.01124731,0.01767228,0.02918366,0.5599466],"study_design_scores_gemma":[0.001456046,0.008357708,0.8074639,0.001980725,0.001689465,0.002470284,0.002767067,0.08170968,0.01845734,0.04019046,0.03235171,0.00110569],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3882436,0.004026684,0.5767828,0.001889857,0.001784941,0.003763297,0.003445725,0.005235464,0.01482757],"genre_scores_gemma":[0.7699345,0.0004616136,0.217934,0.00106364,0.0004427709,0.004512797,0.002370022,0.001769431,0.001511226],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7191049,"threshold_uncertainty_score":0.8867844,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1155585495440885,"score_gpt":0.4424690047889706,"score_spread":0.326910455244882,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}