{"id":"W4402661245","doi":"10.1109/tse.2024.3461657","title":"D<sup>3</sup>: Differential Testing of Distributed Deep Learning With Model Generation","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Division of Computing and Communication Foundations","keywords":"Computer science; Artificial intelligence; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003678836,0.0009066171,0.0005298175,0.0008295536,0.0005591001,0.001741857,0.004244563,0.001472288,0.01793471],"category_scores_gemma":[0.01703981,0.0004286156,0.00114065,0.0005652162,0.00192387,0.002348737,0.002263523,0.002118479,0.003983344],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001431849,"about_ca_system_score_gemma":0.001729574,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003184719,"about_ca_topic_score_gemma":0.004231963,"domain_scores_codex":[0.9953928,0.001075683,0.0005095119,0.0009012477,0.001731271,0.0003894561],"domain_scores_gemma":[0.9892,0.004475313,0.0004731213,0.003719741,0.001870096,0.0002617463],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001697915,0.0006816193,0.01321409,0.0004889234,0.0001482066,0.001433531,0.0003225387,0.1011796,0.059908,0.1382799,0.09610344,0.5865424],"study_design_scores_gemma":[0.0002020449,0.0003830834,0.00217632,0.00005570136,0.00004006967,0.0006669865,0.00007909438,0.7584573,0.1357754,0.06862751,0.03347526,0.00006122161],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02394977,0.00008880115,0.9395513,0.000917003,0.0003235708,0.0002224302,0.0009126747,0.01466219,0.01937217],"genre_scores_gemma":[0.5317449,0.0001127331,0.4438465,0.001981874,0.0001620088,0.000534597,0.004870977,0.002991416,0.01375497],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01793471,"threshold_uncertainty_score":0.05999756,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0148244596035767,"score_gpt":0.2191913172321724,"score_spread":0.2043668576285957,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}