{"id":"W2160065345","doi":"10.1186/1472-6920-12-29","title":"Summative assessment of 5thyear medical students’ clinical reasoning by script concordance test: requirements and challenges","year":2012,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":54,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Summative assessment; Concordance; Multidisciplinary approach; Test (biology); Medical education; Curriculum; Educational measurement; Medicine; Academic year; Formative assessment; Mathematics education; Psychology; Internal medicine; Pedagogy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.005863891,0.0001777683,0.0005995808,0.00005515582,0.00005390435,0.00001498843,0.000247356,0.0003484145,0.0007833898],"category_scores_gemma":[0.2644275,0.0001432444,0.00009784739,0.000124572,0.0004593285,0.0001104961,0.0001569359,0.0004973643,0.00002212744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000837287,"about_ca_system_score_gemma":0.002807037,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006745158,"about_ca_topic_score_gemma":0.00001729289,"domain_scores_codex":[0.9955298,0.0003587222,0.001000947,0.0003723385,0.00238205,0.0003561008],"domain_scores_gemma":[0.979668,0.01724442,0.0004137143,0.0004221514,0.0002569871,0.001994735],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00003901588,0.004389301,0.916778,0.000168901,0.0000538683,0.000002222588,0.0002572357,1.340761e-8,0.00000679359,0.001217136,0.017461,0.05962647],"study_design_scores_gemma":[0.002603632,0.0006035327,0.9831194,0.005967491,0.0001524052,0.00003840585,0.001034571,0.0003000882,0.00003103647,0.00007697831,0.00591943,0.0001530085],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9719718,0.01343266,0.000768687,0.007507251,0.001780719,0.0004967357,0.000007355665,0.00005320238,0.003981663],"genre_scores_gemma":[0.9772069,0.01418255,0.004513259,0.002448724,0.001087151,0.00007042629,0.00007570891,0.00002364544,0.0003915998],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2585636,"threshold_uncertainty_score":0.8577569,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0947932579973595,"score_gpt":0.488989824920343,"score_spread":0.3941965669229835,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}