{"id":"W3196239222","doi":"10.1145/3468264.3468591","title":"A comprehensive study of deep learning compiler bugs","year":2021,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Compiler; Computer science; Programming language; Boosting (machine learning); Deep learning; Context (archaeology); Code (set theory); Code generation; Parallel computing; Artificial intelligence; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.003163037,0.0009864264,0.0009047389,0.001823818,0.0006628768,0.00155269,0.001748697,0.001501509,0.001691998],"category_scores_gemma":[0.02756425,0.0009137109,0.0007732537,0.001420534,0.002671411,0.004236799,0.001535836,0.003144266,0.0002856702],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00195161,"about_ca_system_score_gemma":0.002321689,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004740118,"about_ca_topic_score_gemma":0.003711327,"domain_scores_codex":[0.9977367,0.0004666333,0.0001396764,0.0003975914,0.001021998,0.0002373493],"domain_scores_gemma":[0.9842939,0.01028648,0.001690864,0.001361885,0.00207438,0.0002924992],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","study_design_scores_codex":[0.0002006549,0.0003042237,0.02158268,0.001679014,0.000271313,0.0006008615,0.0006300489,0.334504,0.005439697,0.28861,0.01619087,0.3299867],"study_design_scores_gemma":[0.00002551133,0.0001234547,0.003662958,0.0003944419,0.00007936738,0.0004250145,0.0001112877,0.7019027,0.005616136,0.2769485,0.01064884,0.00006191315],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2516542,0.03492869,0.6749967,0.01395905,0.0005903266,0.0001080604,0.0005866203,0.003076228,0.02010017],"genre_scores_gemma":[0.8845194,0.008845049,0.09753366,0.001386161,0.0004888463,0.00009539775,0.0004958673,0.0007039179,0.005931674],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.996837,"threshold_uncertainty_score":0.01672792,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03173090442745961,"score_gpt":0.283983725789503,"score_spread":0.2522528213620434,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}