{"id":"W4413423557","doi":"10.1016/b978-0-443-30046-2.00012-0","title":"Identifying large language model hallucinations in health communication","year":2025,"lang":"en","type":"book-chapter","venue":"Elsevier eBooks","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Psychology; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001266688,0.001014214,0.0006993948,0.000955051,0.0003485151,0.001742781,0.0006314793,0.001091623,0.006412179],"category_scores_gemma":[0.00471819,0.000383535,0.0008954573,0.001117647,0.0003563112,0.001563215,0.001009802,0.001520514,0.003608376],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003932109,"about_ca_system_score_gemma":0.0004121521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002792235,"about_ca_topic_score_gemma":0.003238374,"domain_scores_codex":[0.9995781,0.0002003554,0.00002233354,0.00008539738,0.00008213087,0.00003177171],"domain_scores_gemma":[0.9971806,0.002454874,0.0000784131,0.0001312693,0.0001033639,0.00005135861],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000276695,0.0001896626,0.005371344,0.0004195423,0.000210234,0.0004896953,0.0006607064,0.07444534,0.0157372,0.01474039,0.03429322,0.853166],"study_design_scores_gemma":[0.00001677616,0.0001098172,0.005205263,0.0001074167,0.00007343356,0.0005521369,0.0004068163,0.9293423,0.00554105,0.04547426,0.01312347,0.00004723621],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03790528,0.005367311,0.9410103,0.001997118,0.0003672662,0.00007386474,0.001562173,0.002653126,0.009063478],"genre_scores_gemma":[0.5946839,0.008130644,0.342937,0.00082169,0.001133837,0.0002897855,0.007541334,0.0009178082,0.04354394],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006412179,"threshold_uncertainty_score":0.02145088,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0335838217936583,"score_gpt":0.3080661655594898,"score_spread":0.2744823437658315,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}