{"id":"W3173393692","doi":"10.1136/bmjopen-2020-047709","title":"Developing a reporting guideline for artificial intelligence-centred diagnostic test accuracy studies: the STARD-AI protocol","year":2021,"lang":"en","type":"article","venue":"BMJ Open","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":312,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa; Ottawa Hospital","funders":"NIHR Oxford Biomedical Research Centre; National Institute for Health and Care Research; UK Research and Innovation; NIHR Imperial Biomedical Research Centre; Cancer Research UK; Nuclear Power Institute of China","keywords":"Checklist; Medicine; Protocol (science); Test (biology); Diagnostic accuracy; Delphi method; Medical education; Transparency (behavior); Guideline; Medical physics; Artificial intelligence; Alternative medicine; Psychology; Computer science; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5540551,0.003880049,0.007563308,0.02084018,0.005208431,0.01646108,0.01354231,0.01729488,0.01697482],"category_scores_gemma":[0.6965496,0.006406503,0.0138426,0.01610545,0.007742356,0.01241666,0.01293036,0.02123169,0.01822257],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01456847,"about_ca_system_score_gemma":0.07234307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003546568,"about_ca_topic_score_gemma":0.003120471,"domain_scores_codex":[0.2695116,0.4499375,0.2377001,0.008213752,0.03030729,0.004329751],"domain_scores_gemma":[0.2151116,0.4272275,0.07790754,0.05755751,0.2163158,0.005880047],"domain_codex":"methods","domain_gemma":"reporting","domain_candidate":"reporting","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002004974,0.0006468617,0.003972056,0.1294913,0.001232602,0.0007900593,0.02275713,0.005210375,0.002992543,0.0681408,0.5052869,0.2574744],"study_design_scores_gemma":[0.003504666,0.001044203,0.006216104,0.1808961,0.0008375199,0.001122712,0.006932599,0.005624371,0.004375924,0.04401292,0.7446774,0.0007556025],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"protocol","genre_gemma":"protocol","genre_scores_codex":[0.00193407,0.004709698,0.2534505,0.02571569,0.006958611,0.6765484,0.01685393,0.002774931,0.01105412],"genre_scores_gemma":[0.003269813,0.00216794,0.2304199,0.004422111,0.0005357471,0.7522799,0.005132637,0.0002899365,0.001481876],"genre_candidate":"protocol","genre_consensus":"protocol","teacher_disagreement_score":0.4459449,"threshold_uncertainty_score":0.5499295,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7499802218592867,"score_gpt":0.646529031348602,"score_spread":0.1034511905106847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}