{"id":"W3207796226","doi":"10.1007/s10140-021-01954-x","title":"Limited generalizability of deep learning algorithm for pediatric pneumonia classification on external data","year":2021,"lang":"en","type":"article","venue":"Emergency Radiology","topic":"COVID-19 diagnosis using AI","field":"Medicine","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Generalizability theory; Medicine; Artificial intelligence; Test set; Convolutional neural network; Radiography; Receiver operating characteristic; Pneumonia; Deep learning; Data set; Set (abstract data type); Test (biology); Machine learning; Pattern recognition (psychology); Radiology; Computer science; Statistics; Internal medicine; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01384062,0.001454485,0.0007839294,0.001133047,0.0003982615,0.001794936,0.001448376,0.001253583,0.001818803],"category_scores_gemma":[0.04625225,0.0004202494,0.0009297733,0.0008398676,0.0008618548,0.001933261,0.001818049,0.00185847,0.00106946],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001136466,"about_ca_system_score_gemma":0.001215621,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007727812,"about_ca_topic_score_gemma":0.005680117,"domain_scores_codex":[0.9935694,0.002288048,0.0006770325,0.001815266,0.00135889,0.000291373],"domain_scores_gemma":[0.9800355,0.01133767,0.001097314,0.003749903,0.003466135,0.000313579],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001665023,0.0005265338,0.2069536,0.0007223705,0.0008873645,0.0005244841,0.0003413867,0.2198597,0.02137478,0.001444054,0.006489915,0.5392108],"study_design_scores_gemma":[0.0001302762,0.0009126943,0.06281435,0.000200973,0.0002118362,0.0004897686,0.0002552072,0.9043426,0.0236292,0.003263955,0.003683947,0.00006519501],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7948809,0.003613217,0.186137,0.001792195,0.0003381041,0.0005151979,0.001861701,0.003024025,0.007837755],"genre_scores_gemma":[0.978883,0.0003922348,0.01769103,0.0003286739,0.00005316366,0.0001363052,0.00162837,0.0001125364,0.0007747423],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01384062,"threshold_uncertainty_score":0.07319707,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08301160984302676,"score_gpt":0.3652829491124194,"score_spread":0.2822713392693926,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}