{"id":"W2024540370","doi":"10.1186/1472-6920-12-100","title":"Constructing a question bank based on script concordance approach as a novel assessment methodology in surgical education","year":2012,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Concordance; Test (biology); Item bank; Scale (ratio); Construct validity; Psychology; Medical education; Medicine; Medical physics; Psychometrics; Clinical psychology; Item response theory","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08946011,0.0008996645,0.0009067174,0.00709269,0.001359905,0.003072525,0.001439478,0.0007291763,0.002593265],"category_scores_gemma":[0.166009,0.0005054947,0.001132648,0.004267353,0.001806402,0.003930735,0.00355944,0.001128346,0.0009899725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00245683,"about_ca_system_score_gemma":0.006443454,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009261637,"about_ca_topic_score_gemma":0.001632172,"domain_scores_codex":[0.8568112,0.1102805,0.01237056,0.004303676,0.01518177,0.001052336],"domain_scores_gemma":[0.7610554,0.1426366,0.02212441,0.01570502,0.05606775,0.002410713],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007395409,0.0008958646,0.1451167,0.002411132,0.0002348606,0.0003334199,0.03166033,0.004648132,0.01489143,0.01638072,0.006302563,0.7763851],"study_design_scores_gemma":[0.0008596798,0.009053564,0.4074742,0.005971909,0.001008561,0.004768976,0.055211,0.1962435,0.09646393,0.1047812,0.1169591,0.001204472],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2080114,0.0004235411,0.7712343,0.0008784262,0.0001980588,0.01111045,0.0005670785,0.000929565,0.006647167],"genre_scores_gemma":[0.2270451,0.0002096313,0.7614542,0.0001721424,0.00006442869,0.009691356,0.0005405739,0.00011253,0.0007101181],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9105399,"threshold_uncertainty_score":0.4731159,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1101638321494085,"score_gpt":0.4763523772804762,"score_spread":0.3661885451310677,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}