{"id":"W4412072331","doi":"10.2196/70176","title":"Reducing Hallucinations and Trade-Offs in Responses in Generative AI Chatbots for Cancer Information: Development and Evaluation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Cancer","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Psychology; Data science; Computer science; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02513499,0.001373491,0.001283146,0.001481181,0.0005093327,0.001320668,0.001758293,0.001016945,0.00214491],"category_scores_gemma":[0.1128691,0.0005894565,0.00198412,0.0007906691,0.001013699,0.002089504,0.001700579,0.000952243,0.0004369516],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001518083,"about_ca_system_score_gemma":0.001716607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002249628,"about_ca_topic_score_gemma":0.001961046,"domain_scores_codex":[0.9782456,0.01596871,0.001593039,0.001150494,0.002629735,0.000412484],"domain_scores_gemma":[0.8483712,0.1296178,0.006601188,0.003904672,0.009006969,0.002498102],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.03321486,0.02296572,0.09486842,0.01401417,0.002574168,0.0004581025,0.009135944,0.007533989,0.005855301,0.000746094,0.003985919,0.8046474],"study_design_scores_gemma":[0.03089034,0.2630329,0.3746707,0.004067872,0.01841887,0.002395588,0.009603557,0.2512891,0.02715019,0.002922329,0.0145573,0.001001214],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9808611,0.001219091,0.01029328,0.000166094,0.00007181821,0.004956685,0.0002982832,0.0006658275,0.001467768],"genre_scores_gemma":[0.9540091,0.001144197,0.03431918,0.0002140253,0.00006756619,0.00839102,0.0009569786,0.00007468403,0.000823277],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02513499,"threshold_uncertainty_score":0.1329281,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1862239589163959,"score_gpt":0.5187098469781782,"score_spread":0.3324858880617823,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}