{"id":"W4408167715","doi":"10.1038/s41746-025-01542-0","title":"Red teaming ChatGPT in medicine to yield real-world insights on model behavior","year":2025,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Yield (engineering); Psychology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002922602,0.0002704358,0.0006495912,0.001288787,0.00009055625,0.00001553123,0.0001665162,0.0001489111,0.0001447623],"category_scores_gemma":[0.002480175,0.0002033998,0.00004855856,0.001432968,0.000173986,0.0001695024,0.00004613921,0.0004445993,0.00006963281],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004017787,"about_ca_system_score_gemma":0.00027547,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002002651,"about_ca_topic_score_gemma":0.001044068,"domain_scores_codex":[0.9975911,0.00002070023,0.0009334209,0.0005247761,0.0005017039,0.0004283114],"domain_scores_gemma":[0.9982505,0.0005206928,0.00009718059,0.0005497178,0.0002171866,0.0003647639],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001780457,0.001543387,0.09659509,0.0005762759,0.00007242131,0.0003148003,0.0214058,0.0004177192,0.01040615,0.02027582,0.08436977,0.7622423],"study_design_scores_gemma":[0.004831049,0.0209857,0.5608184,0.1155302,0.001196304,0.0001186002,0.04675532,0.02677296,0.04700325,0.05515384,0.1175144,0.003319941],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7145814,0.0001895963,0.0005646332,0.05466503,0.0009980473,0.001243376,0.000003763532,0.0001292658,0.2276249],"genre_scores_gemma":[0.9774374,0.0000967758,0.0001021749,0.007604624,0.000703136,0.0001651468,0.00005367503,0.00002785197,0.01380925],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7589223,"threshold_uncertainty_score":0.8294404,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1804216933400206,"score_gpt":0.4531145109831594,"score_spread":0.2726928176431388,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}