{"id":"W4401635890","doi":"10.1145/3688838","title":"History-Driven Fuzzing for Deep Learning Libraries","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Fuzz testing; Computer science; Heuristic; Artificial intelligence; Set (abstract data type); Machine learning; Natural language processing; Programming language; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00307524,0.001558583,0.0008434035,0.002372415,0.0006095274,0.001321784,0.002726801,0.001168504,0.002169154],"category_scores_gemma":[0.01692525,0.0009518524,0.001801157,0.0007352441,0.00192463,0.003028335,0.002092718,0.001917382,0.00031941],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003163549,"about_ca_system_score_gemma":0.003097777,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01015311,"about_ca_topic_score_gemma":0.01579886,"domain_scores_codex":[0.997214,0.0006404426,0.000252869,0.0007310903,0.0008390958,0.0003224429],"domain_scores_gemma":[0.9900324,0.006543951,0.000949138,0.001343813,0.0009298315,0.0002007882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004964342,0.0002446533,0.02951989,0.0004987547,0.0002455995,0.0004972302,0.0003298028,0.6994122,0.01286484,0.01367593,0.003121489,0.2390932],"study_design_scores_gemma":[0.00002202356,0.00005275797,0.0007615606,0.00003843674,0.00003046039,0.00005140977,0.00002026347,0.9775757,0.007284441,0.01348482,0.0006636434,0.00001444591],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2661752,0.001482722,0.7071781,0.001156446,0.00009203577,0.000371016,0.001327132,0.01902752,0.0031899],"genre_scores_gemma":[0.8413182,0.0002252835,0.1549188,0.0004515397,0.00002337388,0.000237959,0.001233994,0.0003528538,0.001237992],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01015311,"threshold_uncertainty_score":0.02295327,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06773932218144975,"score_gpt":0.2985530311061435,"score_spread":0.2308137089246937,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}