{"id":"W2148707490","doi":"10.1186/1472-6920-12-121","title":"Temporal stability of objective structured clinical exams: a longitudinal study employing item response theory","year":2012,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Generalizability theory; Objective structured clinical examination; Internal consistency; Competence (human resources); Reliability (semiconductor); Educational measurement; Consistency (knowledge bases); Psychology; Medicine; Medical education; Psychometrics; Statistics; Clinical psychology; Computer science; Mathematics; Social psychology; Developmental psychology; Curriculum; Artificial intelligence; Pedagogy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01516789,0.0002235063,0.0002927732,0.001317553,0.0008216121,0.001173499,0.0006429655,0.0006029078,0.001369151],"category_scores_gemma":[0.04431553,0.0003583987,0.0006230545,0.001494578,0.0005526635,0.001374388,0.001116539,0.001250474,0.0004062802],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006945163,"about_ca_system_score_gemma":0.001140425,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005698962,"about_ca_topic_score_gemma":0.005187719,"domain_scores_codex":[0.9960678,0.002123445,0.0002954638,0.000444305,0.000809847,0.0002592392],"domain_scores_gemma":[0.9688751,0.01129728,0.01001201,0.003543818,0.004684729,0.001586968],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000104031,0.0001692824,0.993765,0.00001015232,0.00005137512,0.00004654196,0.001421216,0.0001023724,0.0001752578,0.0000825192,0.0001244167,0.00394783],"study_design_scores_gemma":[0.000007660418,0.0004130382,0.9966486,0.00001658741,0.00002782529,0.0001328132,0.001148756,0.0009151789,0.0001257468,0.0001113408,0.000441217,0.00001118181],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9986541,0.0001023013,0.0006954058,0.00007414334,0.000005938012,0.00002653038,0.0001500243,0.000004889588,0.0002865961],"genre_scores_gemma":[0.998853,0.00004964653,0.0004967696,0.00002465891,0.000006803115,0.000045626,0.0003476448,0.00000494298,0.0001708393],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01516789,"threshold_uncertainty_score":0.08021647,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.664638114578315,"score_gpt":0.5950031946609595,"score_spread":0.06963491991735549,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}