{"id":"W4393090969","doi":"10.1158/1538-7445.am2024-1296","title":"Abstract 1296: Evaluating the accuracy and reproducibility of ChatGPT models in answering lung cancer patient queries","year":2024,"lang":"en","type":"article","venue":"Cancer Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; Southlake Regional Health Center","funders":"","keywords":"Reproducibility; Lung cancer; Medicine; Cancer; Question answering; Computer science; Medical physics; Oncology; Internal medicine; Information retrieval; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04008476,0.001138372,0.0009484667,0.001338559,0.0006033432,0.002323696,0.001671895,0.002039553,0.001791583],"category_scores_gemma":[0.2013925,0.0004670593,0.001226373,0.0008608741,0.001027355,0.002270567,0.002602344,0.001720056,0.001255299],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001039247,"about_ca_system_score_gemma":0.001313278,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004900673,"about_ca_topic_score_gemma":0.002813687,"domain_scores_codex":[0.9655627,0.02529267,0.002771827,0.002912391,0.002956802,0.0005035818],"domain_scores_gemma":[0.6694317,0.2927229,0.007908304,0.01407243,0.01355749,0.002307278],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.02059625,0.004887315,0.4640599,0.003891654,0.002254248,0.001120883,0.02030164,0.1247983,0.01814862,0.001932632,0.01441189,0.3235967],"study_design_scores_gemma":[0.000756813,0.008058062,0.1569165,0.0004869652,0.0009519015,0.0009766131,0.004040037,0.7999393,0.01827816,0.00325343,0.005955763,0.0003863963],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9717919,0.0002654529,0.01974662,0.0004289772,0.0001282925,0.0006915681,0.001576084,0.002429514,0.002941612],"genre_scores_gemma":[0.9778183,0.00007199878,0.01849296,0.0001811123,0.0000485944,0.0003830495,0.002116855,0.0001675801,0.0007195207],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9599152,"threshold_uncertainty_score":0.211991,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5203333576261823,"score_gpt":0.6176051776411716,"score_spread":0.09727182001498935,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}