{"id":"W4362606928","doi":"10.1007/s10664-023-10291-1","title":"Bugs in machine learning-based systems: a faultload benchmark","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"ca_institutions":"York University; Polytechnique Montréal","funders":"","keywords":"Benchmark (surveying); Computer science; Software portability; Debugging; Software bug; Software quality; Usability; Software engineering; Software; Relevance (law); Benchmarking; Machine learning; Software development; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004833115,0.0008351082,0.0005435743,0.00157792,0.0006123739,0.0006558035,0.001456835,0.001849112,0.002227243],"category_scores_gemma":[0.05051913,0.0003247797,0.0005521344,0.001169502,0.00174326,0.001919174,0.001405402,0.001350043,0.0003221775],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001153667,"about_ca_system_score_gemma":0.00101297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002640595,"about_ca_topic_score_gemma":0.002833005,"domain_scores_codex":[0.9960338,0.00159428,0.0002751331,0.000542413,0.001244862,0.0003095715],"domain_scores_gemma":[0.9267501,0.05921593,0.002349732,0.007210484,0.003636859,0.000836858],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002324553,0.002315596,0.02525999,0.0008620161,0.0002939112,0.0006188678,0.000343471,0.8297448,0.006553147,0.0239475,0.02411843,0.08361784],"study_design_scores_gemma":[0.0002861449,0.0007395287,0.004720035,0.00003606307,0.00004902907,0.0001909443,0.00008493661,0.9715788,0.005104201,0.01591736,0.001274072,0.0000188568],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9597038,0.001127356,0.03017073,0.001276735,0.0002029593,0.00008055145,0.001064521,0.002103769,0.004269484],"genre_scores_gemma":[0.9881141,0.0001234414,0.009742257,0.00008884154,0.00003910391,0.00004009381,0.0008856943,0.000170712,0.000795804],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004833115,"threshold_uncertainty_score":0.02556026,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01719232494440034,"score_gpt":0.2629710838120742,"score_spread":0.2457787588676739,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}