{"id":"W4415534993","doi":"10.1016/j.adro.2025.101929","title":"ChatGPT Versus DeepSeek: Assessing Artificial Intelligence Performance on Radiation Oncology Examination Questions","year":2025,"lang":"en","type":"article","venue":"Advances in Radiation Oncology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Radiation oncology; MEDLINE; Medical physicist; Radiation therapy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001014821,0.0001874624,0.0003652647,0.0008088002,0.0002671485,0.00003065444,0.0001250875,0.0003734095,0.00009797863],"category_scores_gemma":[0.001587937,0.000203147,0.00005496633,0.001055024,0.0001615666,0.0007341835,0.00002087646,0.0005067958,0.0001223774],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002782027,"about_ca_system_score_gemma":0.001173813,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008782536,"about_ca_topic_score_gemma":0.0004391995,"domain_scores_codex":[0.9977012,0.0003605622,0.0008796368,0.0004538222,0.00022236,0.0003824474],"domain_scores_gemma":[0.9974994,0.001555563,0.0003204197,0.0002729531,0.0002512509,0.0001004256],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004820352,0.0004557324,0.006500538,0.00006222645,0.00001137568,0.000006404094,0.001162023,0.005975585,0.0001677063,0.01863979,0.00005899534,0.9664776],"study_design_scores_gemma":[0.001481491,0.01092063,0.2445693,0.0009619106,0.0002843723,0.00005351656,0.0144979,0.2071188,0.05197758,0.02832237,0.4387454,0.00106671],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9004302,0.003136475,0.0272731,0.01261735,0.01047614,0.001326596,0.000004345913,0.0001739567,0.04456183],"genre_scores_gemma":[0.9896036,0.006585759,0.001695591,0.001053808,0.0005669931,0.0002390149,0.0001093607,0.00001497808,0.0001308791],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9654109,"threshold_uncertainty_score":0.8284093,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.113197509616978,"score_gpt":0.5116446984048433,"score_spread":0.3984471887878653,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}