{"id":"W4388812715","doi":"10.2196/50442","title":"Assessment of ChatGPT-3.5's Knowledge in Oncology: Comparative Study with ASCO-SEP Benchmarks","year":2023,"lang":"en","type":"article","venue":"JMIR AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; Ottawa Hospital","funders":"","keywords":"Preprint; Oncology; Clinical Oncology; Medicine; Internal medicine; Medical physics; Computer science; Cancer; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004492667,0.00009502356,0.0003407619,0.0002426736,0.0000441403,0.000005458396,0.00006365772,0.00007346238,0.0001911976],"category_scores_gemma":[0.00001986993,0.0000743457,0.00002774065,0.0007442702,0.00006726875,0.00005952739,0.00002646653,0.0002730694,0.00006502095],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002049382,"about_ca_system_score_gemma":0.0006827645,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005306815,"about_ca_topic_score_gemma":0.002781253,"domain_scores_codex":[0.9989508,0.0001127557,0.0003574636,0.0002056673,0.000171836,0.0002014406],"domain_scores_gemma":[0.9992297,0.000200928,0.00008544047,0.0002059969,0.0001975355,0.00008045605],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002431697,0.003639727,0.9127449,0.0001491112,0.0000625123,0.00004213313,0.05828589,0.00007907676,0.0003300097,0.0004748543,0.009456305,0.01449225],"study_design_scores_gemma":[0.0002533206,0.00427645,0.9208657,0.0001573084,0.00002398579,0.000004118298,0.06726712,0.001892076,0.0008208263,0.0001275094,0.004222173,0.0000893828],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9813255,0.00003620604,0.0000174312,0.00224569,0.0002105972,0.001330193,0.000001771364,0.00002966348,0.01480292],"genre_scores_gemma":[0.9987728,0.00001346738,0.00007681724,0.0001230688,0.0001177423,0.0003530205,0.00004086921,0.000008187241,0.000494011],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01744729,"threshold_uncertainty_score":0.303173,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2142872304348355,"score_gpt":0.5726386980995014,"score_spread":0.358351467664666,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}