{"id":"W4361991005","doi":"10.2196/44084","title":"Scoring Single-Response Multiple-Choice Items: Scoping Review and Comparison of Different Scoring Methods","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Multiple choice; PsycINFO; Metric (unit); Test (biology); MEDLINE; Medicine; Psychology; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1154527,0.002892023,0.008651732,0.07041523,0.001802256,0.005583691,0.004380124,0.002783751,0.008473601],"category_scores_gemma":[0.3663765,0.001800874,0.01013185,0.06786165,0.002775745,0.007313138,0.004634905,0.001661847,0.001434832],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007010979,"about_ca_system_score_gemma":0.02936654,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007920454,"about_ca_topic_score_gemma":0.0188818,"domain_scores_codex":[0.8777933,0.04820928,0.05212596,0.004377752,0.01663085,0.0008629521],"domain_scores_gemma":[0.5386131,0.3655353,0.04247048,0.008408821,0.04393877,0.001033325],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0002368595,0.00004347831,0.002693954,0.8155933,0.003281333,0.0001160298,0.001195034,0.0003200148,0.0001806216,0.001059233,0.004461838,0.1708184],"study_design_scores_gemma":[0.0001177394,0.0001260679,0.005293945,0.9616423,0.009391422,0.0002139976,0.001048461,0.0002358677,0.0003146981,0.0008598753,0.02067634,0.00007926216],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.006172214,0.9495655,0.01172887,0.001704068,0.0005800031,0.01804653,0.007492378,0.0001573214,0.004553022],"genre_scores_gemma":[0.04090038,0.8536237,0.05254811,0.001113118,0.0002384738,0.04420497,0.006406101,0.0001259962,0.0008391403],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8845473,"threshold_uncertainty_score":0.6105795,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5643011162701261,"score_gpt":0.625437994430518,"score_spread":0.06113687816039182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}