{"id":"W4319594569","doi":"10.1145/3583566","title":"COMET: Coverage-guided Model Generation For Deep Learning Library Testing","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Comet; Computer science; Layer (electronics); Set (abstract data type); Test set; Artificial intelligence; Machine learning; Algorithm; Data mining; Programming language; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001818279,0.002113462,0.0007328211,0.001608194,0.0004862146,0.001281762,0.003289008,0.001629108,0.004352611],"category_scores_gemma":[0.01305786,0.0009571011,0.002095909,0.0007716082,0.001112711,0.002439669,0.001971077,0.001933411,0.001059767],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001970455,"about_ca_system_score_gemma":0.003316126,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008751678,"about_ca_topic_score_gemma":0.01367688,"domain_scores_codex":[0.9978654,0.0006125343,0.0001646149,0.0003964342,0.0007428766,0.0002181488],"domain_scores_gemma":[0.9935083,0.004061937,0.0003819581,0.001090847,0.00079067,0.000166239],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005086184,0.0004096909,0.01552883,0.0007710829,0.0002601144,0.0007387292,0.0002712176,0.6845154,0.02082738,0.01206446,0.01845272,0.2456517],"study_design_scores_gemma":[0.00004323537,0.00007853045,0.0002713669,0.00002263625,0.00002362289,0.00008396732,0.00002182616,0.9837958,0.008532496,0.005106953,0.002007734,0.00001193219],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1257488,0.001223309,0.8079842,0.0009854222,0.0001450124,0.0003668038,0.002635455,0.05451574,0.006395147],"genre_scores_gemma":[0.6080656,0.0003630286,0.3774452,0.0008249438,0.000037304,0.0005813542,0.007109942,0.003129092,0.00244359],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008751678,"threshold_uncertainty_score":0.01740152,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.210945814946218,"score_gpt":0.3372706176748377,"score_spread":0.1263248027286197,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}