{"id":"W4402581026","doi":"10.2196/56859","title":"Performance of ChatGPT in the In-Training Examination for Anesthesiology and Pain Medicine Residents in South Korea: Observational Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Anesthesiology; Observational study; Pain medicine; Medicine; Training (meteorology); Physical therapy; Family medicine; Internal medicine; Anesthesia; Geography","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001892313,0.0004211357,0.0004642953,0.001082758,0.0005508297,0.0005995284,0.0005966256,0.0005444382,0.001295096],"category_scores_gemma":[0.008462691,0.0003870256,0.000522182,0.0008124893,0.0005733168,0.001018254,0.001457572,0.0007272277,0.000342573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007095825,"about_ca_system_score_gemma":0.001130362,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004564755,"about_ca_topic_score_gemma":0.007503369,"domain_scores_codex":[0.9989251,0.0003142926,0.0001894182,0.0001741134,0.0001978409,0.0001992149],"domain_scores_gemma":[0.9940482,0.00119809,0.002435839,0.0002841792,0.0009790872,0.001054554],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001119869,0.0004100217,0.9922234,0.00008529144,0.00003167225,0.0002523314,0.002376694,0.00005438096,0.0003731549,0.00001475531,0.000235947,0.003830407],"study_design_scores_gemma":[0.00001424423,0.0007295477,0.9929526,0.00002972781,0.00003534917,0.0004850325,0.004728827,0.0003733028,0.0002638647,0.00001652065,0.0003573185,0.00001353162],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9997578,0.00003567929,0.00002380755,0.00001613292,0.000001786418,0.00001742505,0.00003635646,0.000001542215,0.0001094477],"genre_scores_gemma":[0.9994875,0.00006820788,0.0001178568,0.00003402715,0.000003113493,0.00002604555,0.0001082616,0.000001793489,0.0001530518],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004564755,"threshold_uncertainty_score":0.01000762,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2320721749646591,"score_gpt":0.4763171061991056,"score_spread":0.2442449312344465,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}