{"id":"W2143619291","doi":"10.1177/0146621603254799","title":"A New Look at the Influence of Guessing on the Reliability of Multiple-Choice Tests","year":2003,"lang":"en","type":"article","venue":"Applied Psychological Measurement","topic":"Advanced Statistical Methods and Models","field":"Mathematics","cited_by":59,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Variance (accounting); Reliability (semiconductor); Test (biology); Statistics; Econometrics; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03635688,0.001126134,0.002083933,0.003438599,0.0009726277,0.005469345,0.002841339,0.002304553,0.005035478],"category_scores_gemma":[0.3355232,0.001190444,0.00196572,0.002396918,0.007089938,0.01388765,0.002943872,0.007951532,0.0008967706],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001478232,"about_ca_system_score_gemma":0.00109029,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003920957,"about_ca_topic_score_gemma":0.003524185,"domain_scores_codex":[0.96777,0.01843235,0.001300542,0.002724738,0.009016134,0.0007562079],"domain_scores_gemma":[0.498598,0.4451222,0.009669635,0.02714857,0.01808354,0.001378102],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007197473,0.0002786555,0.07418545,0.0009171611,0.00106798,0.001731428,0.006217902,0.03349894,0.009697593,0.2369936,0.01447639,0.6202152],"study_design_scores_gemma":[0.0001144737,0.001192348,0.1171358,0.00104245,0.0007767976,0.004062324,0.002006994,0.1799316,0.01311008,0.642949,0.03667842,0.0009996843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2008027,0.03857173,0.6678993,0.04971242,0.001902056,0.00008438898,0.0003895318,0.001297344,0.0393407],"genre_scores_gemma":[0.8830085,0.01221034,0.08928097,0.002877403,0.00470623,0.00006965434,0.0001953022,0.0009731274,0.00667848],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03635688,"threshold_uncertainty_score":0.1922759,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2678450942987462,"score_gpt":0.4308893442137286,"score_spread":0.1630442499149824,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}