{"id":"W4401032320","doi":"10.1007/978-3-031-66064-1_3","title":"Pierce: A Testing Tool for Neural Network Verification Solvers","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; University of Waterloo","funders":"","keywords":"Computer science; Artificial neural network; Programming language; Software engineering; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004050108,0.002736285,0.001116885,0.002303568,0.0006818108,0.002186758,0.005116879,0.001966425,0.06755883],"category_scores_gemma":[0.02255155,0.00181777,0.002144968,0.0011769,0.001664507,0.005795258,0.004597386,0.003700192,0.01487504],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008313789,"about_ca_system_score_gemma":0.00164142,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00143708,"about_ca_topic_score_gemma":0.002168405,"domain_scores_codex":[0.9960443,0.001183241,0.0003684538,0.0005909805,0.001515184,0.0002978543],"domain_scores_gemma":[0.9872562,0.008760191,0.0004589291,0.002294053,0.001070015,0.0001606828],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009430264,0.0003807264,0.002489688,0.001610617,0.0003368672,0.0008972936,0.0002802115,0.08901452,0.01804866,0.1422149,0.2050229,0.5387604],"study_design_scores_gemma":[0.0003586599,0.0002374804,0.0004775989,0.0002729135,0.00009899556,0.0007485077,0.00008709374,0.693271,0.0598221,0.1577566,0.08676375,0.0001052418],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.002835228,0.0001716861,0.895996,0.0003143006,0.000136506,0.0001428512,0.001228943,0.09087916,0.008295359],"genre_scores_gemma":[0.1774666,0.0005225469,0.7508534,0.001063202,0.0001841224,0.001023386,0.007608392,0.04254099,0.0187373],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.06755883,"threshold_uncertainty_score":0.2260068,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02549068661471008,"score_gpt":0.2695432457109795,"score_spread":0.2440525590962694,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}