{"id":"W4404199042","doi":"10.1007/s10664-024-10579-w","title":"Towards enhancing the reproducibility of deep learning bugs: an empirical study","year":2024,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal; Dalhousie University","funders":"","keywords":"Reproducibility; Empirical research; Computer science; Data science; Artificial intelligence; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02801437,0.0007686686,0.0006922095,0.001668398,0.0009113282,0.001835115,0.002547333,0.002165787,0.002668585],"category_scores_gemma":[0.3509727,0.0005191975,0.000932158,0.001252972,0.003048871,0.004317667,0.002505649,0.003479533,0.0005502381],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001179659,"about_ca_system_score_gemma":0.001735462,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001399817,"about_ca_topic_score_gemma":0.001408489,"domain_scores_codex":[0.9721835,0.01341174,0.001975531,0.004361202,0.007370084,0.0006979575],"domain_scores_gemma":[0.4054099,0.4433994,0.03109423,0.09931839,0.0188909,0.00188724],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.003330105,0.0036839,0.3234826,0.001712085,0.001206833,0.001089352,0.003280581,0.2255359,0.0193563,0.04354087,0.01230748,0.3614739],"study_design_scores_gemma":[0.0004235656,0.002751231,0.06616565,0.000491078,0.0005331383,0.002004683,0.001021031,0.7969645,0.02488175,0.09724044,0.007359338,0.0001636171],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8724037,0.001311158,0.1181343,0.00141649,0.0001254078,0.0001738377,0.0005143989,0.001773521,0.004147117],"genre_scores_gemma":[0.9853261,0.00009823441,0.01345038,0.0001134205,0.00003084016,0.00003700562,0.0002413371,0.0001760422,0.0005265636],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9719856,"threshold_uncertainty_score":0.1481559,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02200891701765622,"score_gpt":0.3207673699558249,"score_spread":0.2987584529381687,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}