{"id":"W4282830605","doi":"10.1109/icse-companion55297.2022.9793757","title":"DiffWatch: Watch Out for the Evolving Differential Testing in Deep Learning Libraries","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/ACM 44th International Conference on Software Engineering: Companion Proceedings (ICSE-Companion)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Python (programming language); Computer science; Differential (mechanical device); Deep learning; Artificial intelligence; Operating system; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007657402,0.002065289,0.0006658867,0.003365738,0.0007296712,0.002124683,0.003738379,0.001555551,0.01016371],"category_scores_gemma":[0.0425839,0.001603735,0.000809815,0.001048466,0.001720912,0.007138996,0.005338734,0.003392112,0.003829634],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001324362,"about_ca_system_score_gemma":0.001762314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003434832,"about_ca_topic_score_gemma":0.005208088,"domain_scores_codex":[0.9945447,0.001502125,0.0004880501,0.001074367,0.001822595,0.0005680268],"domain_scores_gemma":[0.9701855,0.01615217,0.002216541,0.007583562,0.002637145,0.001225074],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002009161,0.0005386356,0.06485463,0.001113505,0.0001998527,0.002195382,0.003279122,0.009508835,0.02233503,0.01069064,0.2593462,0.623929],"study_design_scores_gemma":[0.0008781738,0.001589537,0.05643957,0.001346633,0.0002296031,0.00325752,0.001261347,0.387534,0.1578259,0.05577077,0.3329603,0.0009066405],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"empirical","genre_scores_codex":[0.06965683,0.00115197,0.2332982,0.003174818,0.0007047448,0.0005209165,0.004866465,0.6761665,0.01045955],"genre_scores_gemma":[0.6606765,0.0006951561,0.2400252,0.004127253,0.0002478852,0.001000689,0.009988563,0.06688268,0.0163561],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01016371,"threshold_uncertainty_score":0.04049671,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05555531626052307,"score_gpt":0.2678024017501799,"score_spread":0.2122470854896569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}