{"id":"W4409832437","doi":"10.1093/oncolo/oyaf038","title":"Medical accuracy of artificial intelligence chatbots in oncology: a scoping review","year":2025,"lang":"en","type":"review","venue":"The Oncologist","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Princess Margaret Cancer Centre; University of Waterloo; University of Toronto","funders":"","keywords":"Readability; Workload; MEDLINE; Medicine; Oncology; Inclusion (mineral); Resource (disambiguation); Internal medicine; Medical education; Psychology; Computer science; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004103856,0.0003161423,0.002860545,0.0002697703,0.00009161939,0.000009898838,0.0006437231,0.0007272381,0.000659861],"category_scores_gemma":[0.01973403,0.00019508,0.0004123572,0.001284752,0.000552104,0.00004118741,0.000192896,0.001273479,0.0001557238],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006958814,"about_ca_system_score_gemma":0.01558121,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006221946,"about_ca_topic_score_gemma":0.0013764,"domain_scores_codex":[0.995234,0.0008275587,0.002640649,0.0004338201,0.0004653227,0.0003986873],"domain_scores_gemma":[0.9902729,0.007570709,0.001020529,0.0007572894,0.0002193237,0.0001592284],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"systematic_review","study_design_scores_codex":[0.00001985866,0.0002243276,0.00000366158,0.1765806,0.00002268522,0.00003738512,0.000109616,2.390775e-7,4.262046e-8,0.0005010522,0.0005903541,0.8219102],"study_design_scores_gemma":[0.00001080554,0.0003093307,0.000001093848,0.7723962,0.0005556272,0.00008819909,0.0002083591,0.00001496882,0.000009471457,0.0005817497,0.2256778,0.0001464106],"study_design_candidate":"systematic_review","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.00000371504,0.9834094,0.0001652602,0.008295512,0.0009659926,0.004349339,0.000005870793,0.00003201825,0.002772867],"genre_scores_gemma":[0.00003091877,0.9965615,0.0001581353,0.001903756,0.0004720784,0.000661881,0.00005276082,0.00001867226,0.0001403246],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.8217638,"threshold_uncertainty_score":0.9899995,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4675575167168528,"score_gpt":0.6145205468759336,"score_spread":0.1469630301590809,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}