{"id":"W4409283601","doi":"10.1038/s41586-025-08866-7","title":"Towards conversational diagnostic artificial intelligence","year":2025,"lang":"en","type":"article","venue":"Nature","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":246,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"JSS Academy of Higher Education and Research; DeepMind","keywords":"Milestone; Empathy; Objective structured clinical examination; Excellence; Diagnostic accuracy; Scale (ratio); Medical education; Primary care; Medical history; Artificial intelligence; Patient care; Psychology; Computer science; Medicine; Family medicine; Nursing; Radiology; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01582715,0.0008438706,0.0006592376,0.001046358,0.0008528592,0.005925648,0.002327698,0.002370889,0.005149216],"category_scores_gemma":[0.03265716,0.0005794629,0.000795169,0.0004144015,0.004443169,0.0053277,0.007548884,0.003646172,0.001347794],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001381211,"about_ca_system_score_gemma":0.001212884,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000591303,"about_ca_topic_score_gemma":0.0003886253,"domain_scores_codex":[0.9814292,0.01512409,0.0004514582,0.001109395,0.001554685,0.0003312301],"domain_scores_gemma":[0.9770478,0.01786606,0.0007370667,0.002141958,0.001390783,0.0008162912],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001226401,0.001416852,0.007948156,0.002324556,0.0003388457,0.0008681485,0.02661816,0.04192697,0.05591603,0.3136566,0.01972796,0.5280314],"study_design_scores_gemma":[0.0001787785,0.0006650639,0.002099664,0.0005538302,0.00008787613,0.0005546482,0.00449101,0.4153486,0.01686836,0.4750206,0.08399576,0.000135683],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"commentary","genre_scores_codex":[0.05971012,0.002996028,0.8955089,0.01315308,0.0003957984,0.0005090306,0.0002436191,0.002506435,0.02497696],"genre_scores_gemma":[0.5029024,0.0009004718,0.4883492,0.00244086,0.0002495176,0.0004934051,0.0004514361,0.000228444,0.003984242],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.01582715,"threshold_uncertainty_score":0.08370298,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07912451454278564,"score_gpt":0.4313034769996503,"score_spread":0.3521789624568646,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}