{"id":"W2024540370","doi":"10.1186/1472-6920-12-100","title":"Constructing a question bank based on script concordance approach as a novel assessment methodology in surgical education","year":2012,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Concordance; Test (biology); Item bank; Scale (ratio); Construct validity; Psychology; Medical education; Medicine; Medical physics; Psychometrics; Clinical psychology; Item response theory","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003871995,0.0001786788,0.000428783,0.000203193,0.00004842788,0.00001639755,0.0001000648,0.0003230286,0.0005297469],"category_scores_gemma":[0.1426434,0.0001543843,0.00009275923,0.0003559817,0.0001808529,0.00008230325,0.00002258178,0.000588475,0.00004041346],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004656653,"about_ca_system_score_gemma":0.01096368,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005985392,"about_ca_topic_score_gemma":0.00003127941,"domain_scores_codex":[0.9972618,0.0006843047,0.0006007688,0.0003915175,0.0006914453,0.000370196],"domain_scores_gemma":[0.9831623,0.01522263,0.0002276231,0.0003633471,0.0001865593,0.0008375806],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003539072,0.01118642,0.6583529,0.0001966129,0.00001413314,0.000003036394,0.0002402091,0.00001659094,0.00004057481,0.09170571,0.001798718,0.2360912],"study_design_scores_gemma":[0.01166136,0.00119858,0.9068334,0.01344247,0.0003382236,0.001667423,0.006950062,0.04626455,0.0003234929,0.001945394,0.008496637,0.0008784755],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8557527,0.0005556747,0.06895769,0.005873216,0.004907635,0.001013783,0.000003278043,0.0001079795,0.06282803],"genre_scores_gemma":[0.8426963,0.00002237411,0.151253,0.003966323,0.001200116,0.0002778602,0.0002781932,0.00001859705,0.0002872754],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2484805,"threshold_uncertainty_score":0.9946433,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1101638321494085,"score_gpt":0.4763523772804762,"score_spread":0.3661885451310677,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}