{"id":"W4388812715","doi":"10.2196/50442","title":"Assessment of ChatGPT-3.5's Knowledge in Oncology: Comparative Study with ASCO-SEP Benchmarks","year":2023,"lang":"en","type":"article","venue":"JMIR AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; Ottawa Hospital","funders":"","keywords":"Preprint; Oncology; Clinical Oncology; Medicine; Internal medicine; Medical physics; Computer science; Cancer; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05116513,0.0003628034,0.001718701,0.008450512,0.0008182095,0.001572637,0.00176916,0.0009969301,0.001581045],"category_scores_gemma":[0.1845102,0.0002491959,0.002867082,0.005311384,0.001454534,0.002509815,0.004557998,0.001306104,0.000260642],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003443161,"about_ca_system_score_gemma":0.005318627,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004407173,"about_ca_topic_score_gemma":0.009291796,"domain_scores_codex":[0.9599029,0.01995145,0.008425189,0.002170956,0.008795086,0.000754417],"domain_scores_gemma":[0.7358677,0.1950706,0.02992661,0.005371629,0.02930154,0.004461977],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003179048,0.001030529,0.6949925,0.02313565,0.003037315,0.0003442193,0.01377374,0.002373295,0.0004964032,0.001002341,0.007173849,0.249461],"study_design_scores_gemma":[0.0006196507,0.00413082,0.9442818,0.01016989,0.002510871,0.0007672487,0.00916476,0.005793914,0.001198239,0.001629536,0.01955029,0.0001828948],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9507127,0.01669211,0.005916342,0.001524461,0.0001816363,0.006307615,0.009427303,0.0001801429,0.009057577],"genre_scores_gemma":[0.9693182,0.003307298,0.01303633,0.0006118163,0.00005936289,0.006076398,0.007118854,0.0000362398,0.0004353667],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05116513,"threshold_uncertainty_score":0.2705903,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2142872304348355,"score_gpt":0.5726386980995014,"score_spread":0.358351467664666,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}