{"id":"W2112868871","doi":"10.1111/medu.12202","title":"Evaluating the quality of medical multiple‐choice items created with automated processes","year":2013,"lang":"en","type":"article","venue":"Medical Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":41,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Quality (philosophy); Multiple choice; MEDLINE; Medical education; Psychology; Computer science; Medicine; Internal medicine; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08025657,0.001037655,0.001012207,0.003445937,0.000561977,0.002353583,0.001579606,0.001314132,0.003952594],"category_scores_gemma":[0.2563512,0.0005084322,0.001546576,0.002308098,0.001870485,0.002443559,0.002657312,0.001083938,0.00125754],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001023511,"about_ca_system_score_gemma":0.001276063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004006161,"about_ca_topic_score_gemma":0.000701749,"domain_scores_codex":[0.9230646,0.04296599,0.01037001,0.00383258,0.01893441,0.0008324405],"domain_scores_gemma":[0.5137681,0.3832396,0.03510521,0.03024574,0.03612991,0.001511423],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.006382698,0.002795216,0.2821453,0.003733822,0.0008790609,0.0002942567,0.009700359,0.005742733,0.02699969,0.002250863,0.003569071,0.6555069],"study_design_scores_gemma":[0.001860074,0.01905547,0.8577546,0.001704741,0.0007189016,0.001135588,0.003660366,0.02450989,0.06369103,0.005971293,0.01955324,0.0003848991],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8877829,0.0009927142,0.09470252,0.0005402399,0.000185888,0.005430438,0.000631535,0.0007757921,0.008958039],"genre_scores_gemma":[0.8336485,0.0004873054,0.1590509,0.0003323716,0.0001093244,0.003754326,0.0008150164,0.000142754,0.00165957],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9197434,"threshold_uncertainty_score":0.4244424,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5970176373843434,"score_gpt":0.6240804864393285,"score_spread":0.02706284905498502,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}