{"id":"W4409478867","doi":"10.1145/3729533","title":"Evaluating API-Level Deep Learning Fuzzers: A Comprehensive Benchmarking Study","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Computer science; Benchmarking; Deep learning; Artificial intelligence; Software engineering; Machine learning; Data science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006426266,0.0001723561,0.0002447674,0.0003395655,0.0003826034,0.00005775238,0.0003673899,0.0001015562,0.000008515324],"category_scores_gemma":[0.0003853727,0.0001827684,0.00006806474,0.0005683123,0.00002705402,0.0001150751,0.00004602585,0.0004457548,0.000002851862],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000454295,"about_ca_system_score_gemma":0.00002917125,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003542308,"about_ca_topic_score_gemma":0.000005192685,"domain_scores_codex":[0.9986883,0.0002703237,0.0002458399,0.0004566435,0.0001104563,0.0002283771],"domain_scores_gemma":[0.9973029,0.00200261,0.00005521025,0.0004888744,0.00009287034,0.00005747941],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009402515,0.00007822061,0.0002105447,0.000029109,0.00008987127,0.000002615959,0.0008101516,0.06342825,0.001276885,0.001407285,0.00000461475,0.9326531],"study_design_scores_gemma":[0.003530667,0.004307518,0.05576034,0.0003572016,0.000469598,0.0002380605,0.003410676,0.8894279,0.01605471,0.009477514,0.01501116,0.001954646],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03114826,0.000161471,0.9672189,0.0001212007,0.0002841033,0.0002850941,0.000001163384,0.000765939,0.00001389277],"genre_scores_gemma":[0.2740613,0.00003601756,0.725509,0.00008960723,0.00001759948,0.0001682797,7.750582e-7,0.00000937207,0.0001080552],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9306984,"threshold_uncertainty_score":0.7453077,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1624685076286059,"score_gpt":0.392204384378163,"score_spread":0.2297358767495571,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}