{"id":"W3205551384","doi":"10.1038/s41591-021-01517-0","title":"A quality assessment tool for artificial intelligence-centered diagnostic test accuracy studies: QUADAS-AI","year":2021,"lang":"en","type":"letter","venue":"Nature Medicine","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":242,"is_retracted":false,"has_abstract":false,"ca_institutions":"Ottawa Hospital; Hospital for Sick Children; University of Ottawa","funders":"National Institute of Diabetes and Digestive and Kidney Diseases; NIHR Oxford Biomedical Research Centre; NIHR Imperial Biomedical Research Centre; Massachusetts General Hospital; UK Research and Innovation; Assistance publique-Hôpitaux de Paris; Institut National de la Santé et de la Recherche Médicale; Singapore Eye Research Institute; University of Ottawa; Imperial College London; Linköpings Universitet; Wellcome Trust; University College London; Cancer Research UK; Oxford University Hospitals NHS Foundation Trust; Amsterdam University Medical Centers; National Institutes of Health; University Hospitals Birmingham NHS Foundation Trust; Nuclear Power Institute of China; Universiteit van Amsterdam; Université de Paris; Ottawa Hospital Research Institute; University of Oxford; University of Leeds; National Institute for Health and Care Excellence; Massachusetts Institute of Technology; University of Pennsylvania; National Institute for Health and Care Research","keywords":"Test (biology); Diagnostic accuracy; Diagnostic test; Artificial intelligence; Computer science; Quality (philosophy); Quality assessment; Medical physics; Machine learning; External quality assessment; Medicine; Pathology; Biology; Internal medicine; Pediatrics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1431283,0.0006895731,0.003111313,0.003525721,0.00182662,0.004745736,0.002511932,0.00857059,0.003936284],"category_scores_gemma":[0.4491502,0.0007716498,0.002493064,0.004666708,0.002462954,0.003434506,0.004291064,0.01149002,0.002316769],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00435238,"about_ca_system_score_gemma":0.009551831,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004893442,"about_ca_topic_score_gemma":0.009082839,"domain_scores_codex":[0.8566421,0.09483488,0.02837501,0.002594614,0.01603187,0.001521529],"domain_scores_gemma":[0.3795264,0.4765529,0.02869061,0.01495997,0.09142112,0.008848988],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001370911,0.000171224,0.04984705,0.001005949,0.000832439,0.0005654604,0.0009291286,0.001038075,0.000517427,0.01661545,0.6822008,0.244906],"study_design_scores_gemma":[0.004859292,0.001877511,0.09622201,0.008700063,0.002073256,0.00467838,0.002354971,0.03881063,0.002749332,0.1033014,0.7333102,0.001062969],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"methods","genre_scores_codex":[0.00969048,0.006537922,0.05894409,0.89668,0.0102097,0.001414965,0.004499749,0.001335671,0.0106874],"genre_scores_gemma":[0.1622997,0.004285864,0.3555372,0.4397772,0.01850894,0.007961388,0.004445665,0.0008652374,0.006318686],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8568717,"threshold_uncertainty_score":0.7569437,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3335800058022347,"score_gpt":0.5481464259654348,"score_spread":0.2145664201632001,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}