{"id":"W2122398545","doi":"10.1016/j.amjsurg.2012.09.002","title":"Assessing clinical judgment using the Script Concordance test: the importance of using specialty-specific experts to develop the scoring key","year":2012,"lang":"en","type":"article","venue":"The American Journal of Surgery","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Montréal; McGill University","funders":"","keywords":"Concordance; Test (biology); Specialty; Reliability (semiconductor); Key (lock); Medicine; Scoring system; Cronbach's alpha; Medical education; Psychology; Medical physics; Computer science; Family medicine; Clinical psychology; Psychometrics; Surgery; Internal medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06904461,0.0006739069,0.001284784,0.004147945,0.001537137,0.003255722,0.001604803,0.001298114,0.00180537],"category_scores_gemma":[0.2904311,0.0005268501,0.001078353,0.002237692,0.00125685,0.003698543,0.002535426,0.001983658,0.0007772341],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0017902,"about_ca_system_score_gemma":0.004994086,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002633562,"about_ca_topic_score_gemma":0.006684589,"domain_scores_codex":[0.9215164,0.04060353,0.01590202,0.002406561,0.01859257,0.0009788515],"domain_scores_gemma":[0.7282693,0.1550303,0.02579152,0.01311724,0.0727471,0.005044646],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001736169,0.0007861019,0.5996903,0.001470849,0.001119287,0.0007209708,0.008517628,0.003526716,0.009018893,0.004877397,0.01934726,0.3491884],"study_design_scores_gemma":[0.0004361297,0.00346704,0.7855575,0.003052778,0.0009363535,0.008786762,0.01397298,0.1037245,0.03117016,0.01884431,0.02919399,0.0008574969],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.821397,0.00213779,0.1273878,0.00411125,0.002166508,0.002559591,0.000854893,0.000976786,0.03840844],"genre_scores_gemma":[0.8684372,0.0006482542,0.1274798,0.0007573137,0.0002129395,0.0007890784,0.000410895,0.0001768679,0.001087628],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9309554,"threshold_uncertainty_score":0.3651472,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2193784326796066,"score_gpt":0.4300570789545134,"score_spread":0.2106786462749068,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}