{"id":"W4310673963","doi":"10.1016/j.infsof.2022.107129","title":"A probabilistic framework for mutation testing in deep neural networks","year":2022,"lang":"en","type":"article","venue":"Information and Software Technology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Testability; Context (archaeology); Machine learning; Probabilistic logic; Mutation; Set (abstract data type); Artificial intelligence; Test suite; Artificial neural network; Suite; Data mining; Model-based testing; Reliability engineering; Test case; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006901001,0.001205528,0.001950504,0.002513807,0.0008372524,0.001989259,0.005048874,0.002692709,0.004513327],"category_scores_gemma":[0.02835049,0.001350885,0.001845405,0.001757536,0.002907121,0.004431457,0.003290173,0.003064001,0.0004295779],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002903314,"about_ca_system_score_gemma":0.00309239,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01154577,"about_ca_topic_score_gemma":0.01629814,"domain_scores_codex":[0.9956766,0.001677788,0.0002505497,0.0006490391,0.001269858,0.0004761856],"domain_scores_gemma":[0.9818723,0.01383151,0.0009904505,0.001149833,0.001764968,0.0003909021],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009238568,0.00005313623,0.0007586619,0.00009121014,0.00006338503,0.00009305923,0.00006287444,0.8462417,0.0007588119,0.1078822,0.001187847,0.04271475],"study_design_scores_gemma":[0.000004580184,0.0000058676,0.00004257461,0.000008895208,0.000005631757,0.000009564093,0.000002490205,0.9663716,0.0001681304,0.03324368,0.0001323919,0.000004632482],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004137769,0.0001733243,0.9943461,0.0001793855,0.00001967676,0.00002385283,0.00006151788,0.0004760759,0.0005822986],"genre_scores_gemma":[0.5691989,0.000492399,0.4235414,0.0003351642,0.000183218,0.0003588292,0.0004299934,0.0006159001,0.004844273],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01154577,"threshold_uncertainty_score":0.0364964,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01667141722285493,"score_gpt":0.2536663021430027,"score_spread":0.2369948849201478,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}