{"id":"W4403681525","doi":"10.1001/jamanetworkopen.2024.37711","title":"Performance of Multimodal Artificial Intelligence Chatbots Evaluated on Clinical Oncology Cases","year":2024,"lang":"en","type":"article","venue":"JAMA Network Open","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":43,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Health Network; Princess Margaret Cancer Centre; University of Toronto","funders":"AstraZeneca Canada; AstraZeneca","keywords":"Chatbot; Multimodal therapy; Artificial intelligence; Medicine; Computer science; Internal medicine; Natural language processing; Medical physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00329031,0.0001551968,0.000480374,0.0000911259,0.0001211407,0.00007709259,0.000282101,0.0003058882,0.0006729528],"category_scores_gemma":[0.0008856934,0.0001275783,0.0001157777,0.0005237542,0.0001754557,0.0002055857,0.0001206887,0.0006043855,0.0006324396],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001556186,"about_ca_system_score_gemma":0.001062552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008845432,"about_ca_topic_score_gemma":0.0001510154,"domain_scores_codex":[0.9974518,0.0003021414,0.00116706,0.0004327174,0.000282035,0.0003642275],"domain_scores_gemma":[0.997272,0.001712623,0.0001672614,0.0003709233,0.0002803767,0.0001968367],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002365185,0.0003815754,0.01384817,0.00009789626,0.00007022085,0.00007873905,0.0006978809,0.001906523,0.00003716436,0.00147897,0.007197952,0.9718397],"study_design_scores_gemma":[0.0001746171,0.02062644,0.04855691,0.004698585,0.0002792112,0.0002482558,0.001862638,0.8597628,0.009791623,0.004797407,0.04864147,0.0005599909],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9788848,0.0007460978,0.000194351,0.009858434,0.003439207,0.001077245,0.000005227075,0.00006233373,0.005732313],"genre_scores_gemma":[0.9920458,0.00117559,0.001286002,0.001470764,0.003434535,0.00007261756,0.00004314468,0.00002568594,0.0004459117],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9712797,"threshold_uncertainty_score":0.8128943,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5008204620269682,"score_gpt":0.5705068335257164,"score_spread":0.06968637149874823,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}