{"id":"W2056826270","doi":"10.1007/s10459-011-9279-2","title":"Optimization of answer keys for script concordance testing: should we exclude deviant panelists, deviant responses, or neither?","year":2011,"lang":"en","type":"article","venue":"Advances in Health Sciences Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":34,"is_retracted":false,"has_abstract":false,"ca_institutions":"McGill University; Université de Montréal","funders":"","keywords":"Concordance; Reliability (semiconductor); Outlier; Correlation; Test (biology); Statistics; Psychology; Type I and type II errors; Concordance correlation coefficient; Social psychology; Clinical psychology; Medicine; Power (physics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.07261711,0.001916475,0.002257961,0.003050334,0.001723387,0.003740094,0.003369422,0.003091952,0.01212066],"category_scores_gemma":[0.4122924,0.001031183,0.001042003,0.001910269,0.001140906,0.006655233,0.00285847,0.002407449,0.005667002],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008697564,"about_ca_system_score_gemma":0.004532705,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00169198,"about_ca_topic_score_gemma":0.003230426,"domain_scores_codex":[0.9127274,0.05906589,0.01200184,0.005473053,0.008248434,0.002483473],"domain_scores_gemma":[0.5987433,0.3111164,0.01963397,0.02939933,0.035513,0.005594003],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.007107519,0.001371459,0.1141434,0.001939157,0.0003025044,0.0008658933,0.004843093,0.003802801,0.02937807,0.005496431,0.03900794,0.7917417],"study_design_scores_gemma":[0.003389281,0.005185899,0.2055196,0.00505079,0.001488481,0.01346204,0.0100901,0.3805486,0.2028088,0.06421122,0.1067834,0.001461875],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3126179,0.001140217,0.6295249,0.01342552,0.001116487,0.004975455,0.002471086,0.02538941,0.009339021],"genre_scores_gemma":[0.4346859,0.0002105893,0.5565574,0.002129449,0.0002590233,0.001755102,0.001121712,0.001585519,0.00169525],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9273829,"threshold_uncertainty_score":0.3840406,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1905218552026139,"score_gpt":0.4576104457242404,"score_spread":0.2670885905216265,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}