{"id":"W4401079394","doi":"10.1186/s12909-024-05803-6","title":"Measuring and correcting staff variability in large-scale OSCEs","year":2024,"lang":"en","type":"article","venue":"BMC Medical Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Hotel Dieu Hospital","funders":"","keywords":"Grading (engineering); Session (web analytics); Confidence interval; Medical education; Psychology; Grading scale; Variance (accounting); Consistency (knowledge bases); Scale (ratio); Statistics; Medicine; Computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003867384,0.0001122024,0.0001887707,0.0002642068,0.00006838711,0.00004164397,0.00007145053,0.0001843543,0.001173921],"category_scores_gemma":[0.01817914,0.00009974006,0.00003086418,0.0007766069,0.00009439654,0.0001387324,0.00003566076,0.000540642,0.00004952545],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000283291,"about_ca_system_score_gemma":0.003786764,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007637838,"about_ca_topic_score_gemma":0.0001451915,"domain_scores_codex":[0.9981796,0.0001538851,0.0004417562,0.000376389,0.0006094057,0.0002389729],"domain_scores_gemma":[0.9990153,0.0004258398,0.0000425033,0.0002259694,0.0001355615,0.0001548765],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00004082005,0.001605649,0.1468903,0.002498958,0.00001803247,0.000005840303,0.006415645,0.000001743996,0.0001064567,0.002973785,0.01640828,0.8230345],"study_design_scores_gemma":[0.002773115,0.0003389135,0.6212206,0.01370743,0.0002632574,0.0008923843,0.03778521,0.2189241,0.0009240451,0.006074004,0.09619267,0.0009042221],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9715623,0.0007581411,0.01428873,0.00504673,0.003399915,0.000357402,9.859997e-7,0.00012672,0.004459083],"genre_scores_gemma":[0.9888081,0.00005083066,0.008510645,0.001005775,0.0008104243,0.0001055149,0.00005492614,0.00001891764,0.0006348527],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8221303,"threshold_uncertainty_score":0.9997392,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02410916767005815,"score_gpt":0.3447051655968144,"score_spread":0.3205959979267562,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}