{"id":"W4384345647","doi":"10.1109/icse48619.2023.00152","title":"Aries: Efficient Testing of Deep Neural Networks via Labeling-Free Accuracy Estimation","year":2023,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Japan Society for the Promotion of Science; Department of Forestry and Natural Resources, Purdue University; Japan Science and Technology Corporation; Japan Society for the Promotion of Science London; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Labeled data; Artificial neural network; Machine learning; Deep neural networks; Deep learning; Test data; Key (lock); De facto; Set (abstract data type); Data mining; Transformation (genetics); Data set; Test set","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008795697,0.002731757,0.001457191,0.00233315,0.000711311,0.001659539,0.005425922,0.002287057,0.00214031],"category_scores_gemma":[0.04338574,0.0008684265,0.001520698,0.001199694,0.001350684,0.003608999,0.003068219,0.003128881,0.001372017],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001420415,"about_ca_system_score_gemma":0.001902889,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005512141,"about_ca_topic_score_gemma":0.007523214,"domain_scores_codex":[0.9913317,0.002939445,0.0007977778,0.002035425,0.002429195,0.000466466],"domain_scores_gemma":[0.972744,0.01523806,0.002516255,0.005682104,0.003194977,0.0006246801],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001906652,0.0007199994,0.03244467,0.0006833904,0.0008439872,0.0003840445,0.0002620266,0.4979006,0.01742673,0.007980457,0.01370085,0.4257466],"study_design_scores_gemma":[0.00005675666,0.0001780811,0.001388708,0.00002551552,0.00003129771,0.00007940101,0.00003409486,0.986882,0.007124433,0.003236722,0.0009383445,0.00002464763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1596221,0.001828877,0.7973962,0.0005439571,0.0003502268,0.0003351226,0.001995839,0.03451805,0.003409609],"genre_scores_gemma":[0.6436622,0.0003275767,0.3463024,0.0004534366,0.0001179233,0.0004974964,0.00514668,0.001240294,0.002251901],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008795697,"threshold_uncertainty_score":0.04651666,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02082330005433738,"score_gpt":0.2748311056085461,"score_spread":0.2540078055542088,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}