{"id":"W1972679656","doi":"10.1007/s10459-009-9175-1","title":"Two models of raters in a structured oral examination: does it make a difference?","year":2009,"lang":"en","type":"article","venue":"Advances in Health Sciences Education","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":9,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University; University of Calgary; Medical Council of Canada; Ottawa Hospital; University of Ottawa","funders":"","keywords":"Reliability (semiconductor); Oral examination; Inter-rater reliability; Internal consistency; Objective structured clinical examination; Physical examination; Psychology; Medicine; Clinical psychology; Psychometrics; Developmental psychology; Rating scale; Family medicine; Oral health; Psychiatry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2442981,0.003268459,0.004975247,0.003080359,0.002820439,0.01129474,0.009370131,0.00864632,0.009510638],"category_scores_gemma":[0.4933853,0.003019425,0.008911762,0.00158845,0.007567061,0.01466222,0.005401892,0.01038362,0.004507163],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005874738,"about_ca_system_score_gemma":0.004608754,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02109803,"about_ca_topic_score_gemma":0.01885216,"domain_scores_codex":[0.8082151,0.1540227,0.006058547,0.01581787,0.01076175,0.005123999],"domain_scores_gemma":[0.323384,0.5717365,0.03061686,0.038743,0.02919284,0.006326756],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.02154991,0.002812375,0.6988846,0.001360415,0.01131783,0.001033048,0.018348,0.06715705,0.001687501,0.04106728,0.01174613,0.1230359],"study_design_scores_gemma":[0.003723284,0.004822283,0.255142,0.001179958,0.006161174,0.001573972,0.007817874,0.6023636,0.00270949,0.1062558,0.006617915,0.001632591],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8370654,0.001526293,0.1313651,0.01106484,0.001079912,0.003574289,0.001544932,0.0004546722,0.01232467],"genre_scores_gemma":[0.9418221,0.0004524493,0.04079362,0.002099562,0.0004425829,0.001856258,0.0021657,0.0002611388,0.01010665],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2442981,"threshold_uncertainty_score":0.9319149,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03388845178213149,"score_gpt":0.4330013033188576,"score_spread":0.399112851536726,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}