{"id":"W4415962483","doi":"10.1128/jmbe.00205-25","title":"Question format is the best predictor of item discrimination: a multivariable analysis","year":2025,"lang":"en","type":"article","venue":"Journal of Microbiology and Biology Education","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Trent University","funders":"","keywords":"Recall; Logistic regression; Item bank; Odds; Multivariable calculus; Correlation; Exploratory analysis; Relation (database); Odds ratio","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008451364,0.00005076264,0.0001629607,0.0001970675,0.0002037188,0.00001222881,0.0001452554,0.000130969,0.00004871255],"category_scores_gemma":[0.0001602912,0.00003133339,0.00007323235,0.0003246591,0.0003045206,0.00009665981,0.0000222145,0.0000972434,9.0673e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003976498,"about_ca_system_score_gemma":0.0002811867,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001662563,"about_ca_topic_score_gemma":0.0001367818,"domain_scores_codex":[0.9992751,0.0002654484,0.000277605,0.00006950907,0.00002404498,0.00008822368],"domain_scores_gemma":[0.9990587,0.000262058,0.0003326745,0.000056469,0.0002723466,0.00001779506],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00006913746,0.0002780919,0.9216166,0.00001753453,0.0006615641,6.209894e-8,0.01088555,0.000001165141,0.02624426,0.02825923,0.005901916,0.006064884],"study_design_scores_gemma":[0.001382125,0.0006254325,0.7076036,0.0002530858,0.003679722,0.00002054957,0.1026918,0.00006493853,0.0049296,0.01350948,0.1649653,0.0002743371],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9853106,0.00194205,0.0007669402,0.008670677,0.0009889379,0.0001211244,0.000007985774,0.000002906388,0.002188827],"genre_scores_gemma":[0.9974101,0.0007663526,0.0003705536,0.0002737054,0.0001323122,0.000002989182,0.00001539327,8.976128e-7,0.00102775],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.214013,"threshold_uncertainty_score":0.1566861,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01958637381077954,"score_gpt":0.3725553989119098,"score_spread":0.3529690251011302,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}