{"id":"W2911682672","doi":"10.1097/acm.0000000000002627","title":"Ensuring the Quality of Multiple-Choice Tests: An Algorithm to Facilitate Decision Making for Difficult Questions","year":2019,"lang":"en","type":"article","venue":"Academic Medicine","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Quality (philosophy); Management science; Machine learning; Algorithm; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06051924,0.00290972,0.0031444,0.01288605,0.00219695,0.006540997,0.005472679,0.003859587,0.007103865],"category_scores_gemma":[0.2094259,0.001309593,0.002425552,0.004803083,0.001835677,0.006237794,0.005548699,0.00404184,0.003696253],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003310136,"about_ca_system_score_gemma":0.008359513,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004108889,"about_ca_topic_score_gemma":0.006061793,"domain_scores_codex":[0.9456153,0.02585008,0.00849097,0.004381504,0.01430419,0.001357978],"domain_scores_gemma":[0.7856325,0.1286774,0.01887574,0.007025572,0.05474319,0.005045637],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002927747,0.001213215,0.05299501,0.00128302,0.0005011226,0.0003378149,0.001675123,0.01045433,0.008238893,0.007768942,0.02340868,0.8891962],"study_design_scores_gemma":[0.002436984,0.002687133,0.0529207,0.002009731,0.001270114,0.003296413,0.00208476,0.7676249,0.05061727,0.06326832,0.05083627,0.0009473293],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02873881,0.0006240307,0.9501175,0.002420514,0.0001716229,0.006234585,0.0008179193,0.007668499,0.003206617],"genre_scores_gemma":[0.03312973,0.0001186156,0.9643662,0.0001551123,0.00004857329,0.001360394,0.0002650332,0.0001284265,0.0004279497],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9394808,"threshold_uncertainty_score":0.3200601,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1164316970633832,"score_gpt":0.4311357888689092,"score_spread":0.3147040918055259,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}