{"id":"W4384561684","doi":"10.1038/s41591-023-02437-x","title":"Enhancing the reliability and accuracy of AI-enabled diagnosis via complementarity-driven deferral to clinicians","year":2023,"lang":"en","type":"article","venue":"Nature Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":110,"is_retracted":false,"has_abstract":false,"ca_institutions":"Google (Canada)","funders":"","keywords":"Deferral; Complementarity (molecular biology); Reliability (semiconductor); Medicine; Reliability engineering; Computer science; Intensive care medicine; Economics; Biology; Engineering; Genetics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001423747,0.0001360513,0.000395717,0.0001636019,0.0001771852,0.000006141677,0.0001333478,0.0001883473,0.0002478916],"category_scores_gemma":[0.005499333,0.00008464945,0.00005341938,0.0007895027,0.0002260704,0.00005765574,0.00006990018,0.0008528095,0.00003382915],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008107278,"about_ca_system_score_gemma":0.0001487154,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00301634,"about_ca_topic_score_gemma":0.002254451,"domain_scores_codex":[0.9982237,0.0001063038,0.0006782865,0.0003038115,0.0003903789,0.0002975443],"domain_scores_gemma":[0.9969385,0.001942568,0.000126835,0.0004252201,0.0003753992,0.0001915287],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005750071,0.0002689212,0.7582279,0.0009253977,0.0001093525,0.00002234774,0.02609219,0.00006258162,0.01258882,0.0009706449,0.09267597,0.1074809],"study_design_scores_gemma":[0.0009231474,0.00397225,0.684074,0.002592275,0.0006821193,0.00005103085,0.02892327,0.002204415,0.1267074,0.0149822,0.1343965,0.0004913711],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8228471,0.0003677103,0.0002921692,0.1746175,0.000851634,0.0008020387,0.000006771023,0.00005440652,0.0001606836],"genre_scores_gemma":[0.9853284,0.0003037512,0.0002888778,0.01301123,0.0008188325,0.0000762346,0.00005864396,0.00001480141,0.00009925269],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1624813,"threshold_uncertainty_score":0.6583613,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1265204214225965,"score_gpt":0.4847186370965837,"score_spread":0.3581982156739871,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}