{"id":"W4394845024","doi":"10.1056/aioa2300151","title":"Comparative Evaluation of LLMs in Clinical Oncology","year":2024,"lang":"en","type":"article","venue":"NEJM AI","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":97,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Sunnybrook Health Science Centre","funders":"National Cancer Institute; NIH Clinical Center; National Institutes of Health","keywords":"Clinical Oncology; Internal medicine; Oncology; Medicine; Cancer","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001728788,0.00003600154,0.0001879886,0.00008538419,0.000009835803,0.000003123111,0.00002274613,0.00009745132,0.0002988584],"category_scores_gemma":[0.0002996231,0.00003025008,0.00003956769,0.0001906132,0.00007225616,0.0000435092,0.000005623258,0.0002134647,0.0001684538],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00014694,"about_ca_system_score_gemma":0.001072047,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003838913,"about_ca_topic_score_gemma":0.0003878985,"domain_scores_codex":[0.9989985,0.0001728624,0.0004492797,0.0001217857,0.0001791343,0.00007840811],"domain_scores_gemma":[0.9992801,0.0003125084,0.00003612934,0.00009122334,0.0002369247,0.00004313935],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001471538,0.0003893924,0.03655686,0.0001066305,0.00002849466,0.000007364754,0.01366831,0.00008336497,0.0005492608,0.001775588,0.01121788,0.9354697],"study_design_scores_gemma":[0.0006801003,0.005586064,0.4301989,0.001501187,0.0005888637,0.0000505785,0.02534678,0.3491817,0.04211337,0.03956284,0.1048736,0.0003160839],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9815709,0.0008751158,0.00007900089,0.00673067,0.001115174,0.0002977347,8.11407e-7,0.00001423465,0.009316374],"genre_scores_gemma":[0.9988258,0.00005495475,0.0001182337,0.000527098,0.0003142745,0.0000226199,0.00001103344,0.00000329274,0.0001226473],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9351536,"threshold_uncertainty_score":0.327229,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7465636406134145,"score_gpt":0.698507249034476,"score_spread":0.04805639157893848,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}