{"id":"W4414750656","doi":"10.1038/s41746-025-01943-1","title":"A longitudinal analysis of declining medical safety messaging in generative AI models","year":2025,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Disclaimer; Generative grammar; The Internet; Generative model; Function (biology); Patient safety; MEDLINE","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0124092,0.0002975233,0.0003054319,0.000935868,0.0005553743,0.001767068,0.0008157285,0.001092032,0.005685238],"category_scores_gemma":[0.0956392,0.0003401079,0.0004315206,0.0008750023,0.0006560517,0.002383029,0.001177975,0.002316428,0.001422291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001451874,"about_ca_system_score_gemma":0.0006033249,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01175455,"about_ca_topic_score_gemma":0.009442042,"domain_scores_codex":[0.9972064,0.001635126,0.0001331745,0.000479027,0.0003292207,0.000217088],"domain_scores_gemma":[0.9088178,0.06589807,0.007857304,0.008239915,0.006667447,0.002519475],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0007649255,0.0006562361,0.8966602,0.0001036076,0.000247941,0.0003075985,0.002602698,0.04011804,0.001020903,0.006850033,0.01024352,0.04042431],"study_design_scores_gemma":[0.000100978,0.001048242,0.468253,0.0001513873,0.0002357427,0.0007208614,0.003246578,0.4935964,0.002261416,0.01611297,0.01411799,0.0001543858],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9888131,0.000515621,0.004275505,0.001827175,0.00003948157,0.00003445441,0.002102409,0.000139584,0.002252584],"genre_scores_gemma":[0.9958906,0.0001029586,0.0007559501,0.0001270973,0.000019353,0.00002840827,0.002210536,0.00002976514,0.0008352631],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9875908,"threshold_uncertainty_score":0.06562686,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0645174626387021,"score_gpt":0.4407533161531101,"score_spread":0.376235853514408,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}