{"id":"W4321452091","doi":"10.1080/08957347.2023.2172017","title":"Dissecting Knowledge, Guessing, and Blunder in Multiple Choice Assessments","year":2023,"lang":"en","type":"article","venue":"Applied Measurement in Education","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University; University of Toronto","funders":"Division of Molecular and Cellular Biosciences; National Institute of General Medical Sciences; National Heart, Lung, and Blood Institute","keywords":"Test (biology); Probabilistic logic; Mistake; Psychology; Proxy (statistics); Robustness (evolution); Bayesian probability; Social psychology; Computer science; Artificial intelligence; Machine learning; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4307819,0.002161389,0.002710931,0.007777183,0.001533775,0.005349502,0.003473873,0.003752753,0.003804288],"category_scores_gemma":[0.7810754,0.001577559,0.005781025,0.005944379,0.0082214,0.006776542,0.006416549,0.004268354,0.0004593405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002627885,"about_ca_system_score_gemma":0.002148582,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002353266,"about_ca_topic_score_gemma":0.004157053,"domain_scores_codex":[0.5715744,0.3238786,0.03924826,0.0225588,0.04035273,0.002387182],"domain_scores_gemma":[0.08712095,0.8436829,0.03182122,0.02874847,0.008063287,0.0005630993],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01196509,0.001476282,0.3419473,0.007466592,0.01425037,0.0008803523,0.03576932,0.07935901,0.007740445,0.08375428,0.004709358,0.4106816],"study_design_scores_gemma":[0.00126515,0.007563238,0.2994432,0.003653234,0.003870678,0.001852444,0.005006728,0.3717227,0.04090029,0.2475376,0.01545934,0.001725472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2967034,0.001612787,0.6894945,0.0008090521,0.0001976747,0.003485162,0.0009010329,0.000772073,0.006024348],"genre_scores_gemma":[0.7667019,0.0003020127,0.227626,0.0003682861,0.0000659451,0.003499249,0.0004943039,0.0001222765,0.0008199154],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.569218,"threshold_uncertainty_score":0.7019472,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8152165292421817,"score_gpt":0.575029629601757,"score_spread":0.2401868996404247,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}