{"id":"W3036632634","doi":"10.2196/18089","title":"Assessment of the Robustness of Convolutional Neural Networks in Labeling Noise by Using Chest X-Ray Images From Multiple Centers","year":2020,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"COVID-19 diagnosis using AI","field":"Medicine","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Korea Health Industry Development Institute","keywords":"Convolutional neural network; Receiver operating characteristic; Artificial intelligence; Deep learning; Computer science; Robustness (evolution); Test set; Pattern recognition (psychology); Medicine; Binary classification; Computer-aided diagnosis; Random forest; Noise (video); Medical imaging; Machine learning; Radiology; Support vector machine; Image (mathematics)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007485316,0.001564479,0.000695411,0.00153259,0.0005228746,0.0009853796,0.001079172,0.001552812,0.0005256034],"category_scores_gemma":[0.02251811,0.000337102,0.001169803,0.0007553531,0.000790767,0.001040899,0.001137062,0.0009925549,0.0003033026],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00119045,"about_ca_system_score_gemma":0.0007494953,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009340677,"about_ca_topic_score_gemma":0.006250639,"domain_scores_codex":[0.9965628,0.0009978849,0.0003905944,0.001036842,0.0007132032,0.0002987167],"domain_scores_gemma":[0.989179,0.005664483,0.001527914,0.001610728,0.001691498,0.0003263177],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","study_design_scores_codex":[0.004529382,0.0009443974,0.2978878,0.0007789506,0.002120808,0.0005721014,0.0004179658,0.4776756,0.02212944,0.0007714356,0.005001311,0.1871708],"study_design_scores_gemma":[0.00008070612,0.00102271,0.0668318,0.0001640972,0.000444921,0.0003638296,0.0001788792,0.8966095,0.03179941,0.0008774819,0.001565918,0.00006072933],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9747692,0.001194757,0.02049523,0.0003502131,0.000180263,0.00009727,0.0009445465,0.0007174987,0.001251124],"genre_scores_gemma":[0.9853299,0.0002348957,0.01070561,0.000140339,0.00005231779,0.0000466457,0.003055863,0.00005213813,0.000382357],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009340677,"threshold_uncertainty_score":0.0395866,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03116503130725056,"score_gpt":0.3248385462882414,"score_spread":0.2936735149809909,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}