{"id":"W4409491863","doi":"10.2196/63677","title":"Assessing the Quality and Reliability of ChatGPT’s Responses to Radiotherapy-Related Patient Queries: Comparative Study With GPT-3.5 and GPT-4","year":2025,"lang":"en","type":"article","venue":"JMIR Cancer","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Readability; Quality Score; Likert scale; Reliability (semiconductor); Misinformation; Quality (philosophy); Computer science; Medicine; Comprehension; Medical physics; Metric (unit); Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004877963,0.00009560477,0.0002735038,0.00006008948,0.0001428849,0.00002633283,0.00003452385,0.0000435831,0.000029842],"category_scores_gemma":[0.0001264623,0.000057741,0.00001628708,0.0003032757,0.0001965053,0.00008096407,0.0000192613,0.0001355463,3.465764e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001314574,"about_ca_system_score_gemma":0.0003494548,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004427007,"about_ca_topic_score_gemma":0.0006741528,"domain_scores_codex":[0.9988635,0.0002615699,0.0003724191,0.0002320332,0.0001533118,0.0001172168],"domain_scores_gemma":[0.9988788,0.0004434661,0.0001135938,0.0002448611,0.0002585333,0.00006074507],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001886605,0.0003204014,0.8525375,0.0001579887,0.00006908866,8.824398e-7,0.1106734,0.00003592929,0.0006001295,0.00009910392,0.000372903,0.03324604],"study_design_scores_gemma":[0.0001543531,0.001090249,0.918453,0.0003011863,0.00003876272,0.000001447738,0.0750518,0.00009534807,0.002530288,0.0001467931,0.002059924,0.00007679685],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9879057,0.0009853297,0.00007814554,0.009064586,0.0001339889,0.001639674,0.000004476768,0.00001601646,0.0001721166],"genre_scores_gemma":[0.9984676,0.00009021534,0.0001505713,0.0007647866,0.00002633074,0.0002914703,0.000001021258,0.000005255003,0.0002027828],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06591551,"threshold_uncertainty_score":0.669234,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1761122054590457,"score_gpt":0.54766975135184,"score_spread":0.3715575458927943,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}