{"id":"W4408167715","doi":"10.1038/s41746-025-01542-0","title":"Red teaming ChatGPT in medicine to yield real-world insights on model behavior","year":2025,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Yield (engineering); Psychology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0431629,0.001676922,0.0007850493,0.002431951,0.001810861,0.003278974,0.00221894,0.00241643,0.008557173],"category_scores_gemma":[0.2966875,0.0006587197,0.000763652,0.001042323,0.001928484,0.004488617,0.006596114,0.002654351,0.003558561],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002217306,"about_ca_system_score_gemma":0.002970383,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001341306,"about_ca_topic_score_gemma":0.002781928,"domain_scores_codex":[0.9601337,0.03003011,0.002030199,0.003093865,0.00384363,0.0008684409],"domain_scores_gemma":[0.6399692,0.2738109,0.01279567,0.04302786,0.02606212,0.004334251],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004146651,0.001764421,0.1071267,0.005376791,0.0003735447,0.002674074,0.1341053,0.04053603,0.0424574,0.01467857,0.1006782,0.5460823],"study_design_scores_gemma":[0.0008018718,0.004923653,0.07195275,0.003940888,0.0004393995,0.003341579,0.06124064,0.3924057,0.08305462,0.1025296,0.2742073,0.001161921],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5216552,0.0007120771,0.4004873,0.00738471,0.001351926,0.003175298,0.005106885,0.03985141,0.02027523],"genre_scores_gemma":[0.7496678,0.0002097578,0.2339165,0.002066671,0.0002434059,0.00230261,0.003261214,0.003058819,0.005273102],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0431629,"threshold_uncertainty_score":0.2282699,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1804216933400206,"score_gpt":0.4531145109831594,"score_spread":0.2726928176431388,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}