{"id":"W2954903132","doi":"10.1109/icse.2019.00107","title":"CRADLE: Cross-Backend Validation to Detect and Localize Bugs in Deep Learning Libraries","year":2019,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":191,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; MNIST database; Implementation; Deep learning; Artificial intelligence; Software bug; Software; Reliability (semiconductor); Machine learning; Anomaly detection; Key (lock); Software engineering; Programming language; Operating system","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01093274,0.003369329,0.001110214,0.00283665,0.0007240705,0.001392847,0.004515148,0.002304416,0.002367147],"category_scores_gemma":[0.03848501,0.0009717626,0.00134196,0.001102404,0.002073682,0.002772414,0.003123455,0.003045958,0.001274853],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001726898,"about_ca_system_score_gemma":0.003516469,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01035919,"about_ca_topic_score_gemma":0.0151584,"domain_scores_codex":[0.99409,0.002261246,0.0004751243,0.001543657,0.001199398,0.0004305871],"domain_scores_gemma":[0.9759797,0.01408896,0.001572273,0.004993697,0.002777449,0.0005879396],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001103867,0.0007789735,0.03827513,0.0006850408,0.0009051965,0.0004257143,0.0002173521,0.6536398,0.01172243,0.003942431,0.03238506,0.2559189],"study_design_scores_gemma":[0.00006694255,0.0001428315,0.000840531,0.00003383153,0.00003522159,0.00004635313,0.00002303967,0.9912566,0.004306037,0.002319861,0.0009136865,0.00001503692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3315875,0.00442188,0.5375835,0.001303136,0.0005909204,0.0005423759,0.003085834,0.1166804,0.004204518],"genre_scores_gemma":[0.8213282,0.0002604121,0.1644996,0.00115407,0.00007515705,0.0003552035,0.007522583,0.002526735,0.0022781],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01093274,"threshold_uncertainty_score":0.05781859,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007898350116497276,"score_gpt":0.2634367416265574,"score_spread":0.2555383915100602,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}