{"id":"W4224214784","doi":"10.1145/3511598","title":"An Empirical Study on Data Distribution-Aware Test Selection for Deep Learning Enhancement","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Retraining; Computer science; Artificial intelligence; Machine learning; Selection (genetic algorithm); Test data; Artificial neural network; Software deployment; Metric (unit); Data mining; Data set; Test set; Process (computing); Model selection; Deep learning; Distribution (mathematics); Set (abstract data type); Empirical research; Statistics; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01633086,0.00145385,0.0006818135,0.00162212,0.0006124081,0.001143372,0.001842155,0.001071738,0.0007652937],"category_scores_gemma":[0.0871673,0.000326845,0.0004918426,0.00145219,0.0012088,0.003204224,0.001571432,0.001860745,0.0003500251],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001401287,"about_ca_system_score_gemma":0.001102433,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002191701,"about_ca_topic_score_gemma":0.003098221,"domain_scores_codex":[0.9881107,0.006427385,0.0009183824,0.001730278,0.002405663,0.0004076071],"domain_scores_gemma":[0.9035783,0.06847522,0.00522935,0.01185768,0.009297406,0.001561963],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003134692,0.003055,0.2464016,0.0009582599,0.0004753898,0.0006083663,0.0008204237,0.245084,0.02277862,0.004151051,0.01311086,0.4594217],"study_design_scores_gemma":[0.0002517436,0.002795634,0.04606795,0.0001333102,0.0002246915,0.0007425127,0.0005388322,0.9053727,0.03486914,0.003471004,0.005460335,0.00007214529],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9261658,0.002355196,0.06555314,0.000692505,0.000133839,0.0003195179,0.0006597347,0.001450996,0.002669272],"genre_scores_gemma":[0.9669216,0.000224935,0.03062416,0.0001750571,0.00003732488,0.0001340591,0.001174174,0.0001014449,0.0006072318],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01633086,"threshold_uncertainty_score":0.08636683,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1239226798585278,"score_gpt":0.3887620461359517,"score_spread":0.2648393662774238,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}