{"id":"W6979346650","doi":"","title":"Red Teaming Large Language Models for Healthcare","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Traditional and Medicinal Uses of Annonaceae","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto","keywords":"Health care; Identification (biology); Process (computing); Language model; Computational model; Action (physics); Patient care","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00009215595,0.00008863473,0.00009788485,0.00002973667,0.00009263115,0.000004494265,0.0001075256,0.00008592947,0.000005656793],"category_scores_gemma":[0.00004763738,0.00008177439,0.00007518318,0.0000513379,0.00002371019,0.000003821465,0.00003937574,0.00006133959,0.000004466626],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000114063,"about_ca_system_score_gemma":0.00007594805,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002447402,"about_ca_topic_score_gemma":0.00004527436,"domain_scores_codex":[0.9993815,0.00001380548,0.0001176433,0.00022156,0.00005713199,0.0002083686],"domain_scores_gemma":[0.9996606,0.00001409038,0.00002854882,0.0001763806,0.00006387722,0.00005654649],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007597586,0.000442873,0.09550141,0.0006417063,0.0003194711,0.00002547868,0.0005774333,0.0001479156,0.793407,0.0185056,0.08137302,0.008298371],"study_design_scores_gemma":[0.005083521,0.001087883,0.1361361,0.000353999,0.0001127745,0.00002178735,0.002277701,0.0007284234,0.2305266,0.004520709,0.6184217,0.0007288154],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9862342,0.002022122,0.005879801,0.003406597,0.0002072502,0.0001942423,0.00008855695,0.00002199233,0.001945273],"genre_scores_gemma":[0.991523,0.00009845918,0.0004745823,0.003230404,0.0002796737,0.00004190835,0.0003671526,0.00001068309,0.003974138],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5628803,"threshold_uncertainty_score":0.3334663,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02268061364616695,"score_gpt":0.2980134473756859,"score_spread":0.2753328337295189,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}