{"id":"W4385236484","doi":"10.2196/48978","title":"Performance of ChatGPT on the Situational Judgement Test—A Professional Dilemmas–Based Examination for Doctors in the United Kingdom","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute for Health and Care Research","keywords":"Judgement; Test (biology); Situational ethics; Medical education; Psychology; Teamwork; Clinical judgement; Medicine; Applied psychology; Social psychology; Family medicine; Management; Political science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00203398,0.00008527021,0.000101946,0.0002985716,0.0001537079,0.000008865498,0.0001458152,0.0001047882,0.000281637],"category_scores_gemma":[0.003564351,0.00004895467,0.00004048225,0.000973473,0.00008900038,0.00005003509,0.00000987154,0.0002260198,0.00003504247],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001219791,"about_ca_system_score_gemma":0.001566901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001002619,"about_ca_topic_score_gemma":0.0000184549,"domain_scores_codex":[0.9983321,0.0001317368,0.0004145783,0.0001587889,0.0007886375,0.0001741412],"domain_scores_gemma":[0.9971425,0.002143565,0.0001351459,0.0002029722,0.000303996,0.00007181247],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0007358325,0.009332879,0.3674618,0.001670868,0.00003575519,0.000001945253,0.03598685,0.0003292783,0.001004732,0.04376406,0.1209519,0.4187242],"study_design_scores_gemma":[0.0002308817,0.0006939926,0.890793,0.0009762351,0.00002154512,0.000002265003,0.006094429,0.08052392,0.002269371,0.0006868819,0.0176018,0.0001057264],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9219936,0.000004430534,0.00003743264,0.07558584,0.0007525211,0.001439069,0.000003001403,0.00002100196,0.0001631771],"genre_scores_gemma":[0.9920047,0.000009542638,0.00004088147,0.00473584,0.0004967104,0.001820024,0.0005977593,0.000008937838,0.0002855852],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5233312,"threshold_uncertainty_score":0.4267119,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1817766227857167,"score_gpt":0.4728353424075032,"score_spread":0.2910587196217865,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}