{"id":"W6979346650","doi":"","title":"Red Teaming Large Language Models for Healthcare","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Traditional and Medicinal Uses of Annonaceae","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto","keywords":"Health care; Identification (biology); Process (computing); Language model; Computational model; Action (physics); Patient care","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01420163,0.001052186,0.0006455505,0.0006724544,0.001126184,0.003396099,0.002550798,0.001613887,0.01036384],"category_scores_gemma":[0.03967897,0.0009100465,0.002432843,0.0003645802,0.001700576,0.005375199,0.007114968,0.00455793,0.004712964],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001908909,"about_ca_system_score_gemma":0.004596782,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003465689,"about_ca_topic_score_gemma":0.006279571,"domain_scores_codex":[0.9890114,0.007251276,0.0005571973,0.001279995,0.001492304,0.0004078423],"domain_scores_gemma":[0.9766324,0.01325709,0.0008336661,0.005234817,0.00303914,0.001002945],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001646881,0.0007894029,0.009051857,0.001417387,0.0004836486,0.001291134,0.007130447,0.1642297,0.02404652,0.1743725,0.09603766,0.5195029],"study_design_scores_gemma":[0.0002230107,0.0004935832,0.0007088677,0.0003860326,0.000186345,0.0005066158,0.001150418,0.5821761,0.01819333,0.1879367,0.2078704,0.0001687195],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02175094,0.0005039988,0.9380307,0.008102278,0.0006524799,0.0007509965,0.0008104661,0.02080118,0.008596908],"genre_scores_gemma":[0.208318,0.000422369,0.7715805,0.004374182,0.0001928352,0.0009578048,0.002331001,0.003351936,0.00847132],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01420163,"threshold_uncertainty_score":0.07510626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02268061364616695,"score_gpt":0.2980134473756859,"score_spread":0.2753328337295189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}