{"id":"W4414404854","doi":"10.1109/tse.2025.3612253","title":"MetaSel: A Test Selection Approach for Fine-Tuned DNN Models","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Huawei Technologies (Canada); University of Ottawa","funders":"Alliance de recherche numérique du Canada; Mitacs; Huawei Technologies","keywords":"Covariate; Software deployment; Model selection; Selection (genetic algorithm); Context (archaeology); Subspace topology; Statistical hypothesis testing; Test data; Probability distribution","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005409073,0.003694072,0.001785994,0.002766972,0.0006798228,0.001682355,0.005135519,0.002027119,0.003803097],"category_scores_gemma":[0.02058259,0.001116232,0.001646732,0.001053173,0.0009281341,0.003077069,0.003139745,0.003194689,0.001973659],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00146965,"about_ca_system_score_gemma":0.00231306,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006039591,"about_ca_topic_score_gemma":0.01420435,"domain_scores_codex":[0.9969812,0.001019557,0.0002531388,0.0009037023,0.0006214065,0.000220896],"domain_scores_gemma":[0.9905934,0.005397995,0.0006125016,0.001513878,0.001565166,0.0003171171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009997992,0.0005604075,0.02429804,0.0005042787,0.0007189038,0.0006738885,0.0002538028,0.3509657,0.02146559,0.003412532,0.01628289,0.5798642],"study_design_scores_gemma":[0.00008312088,0.0002426702,0.001281036,0.00004265992,0.00008688369,0.0001483016,0.00006121718,0.9818594,0.008856926,0.005247846,0.002056417,0.00003348927],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1011846,0.002462985,0.8488248,0.0005733983,0.0002494447,0.0005115588,0.001628246,0.04082659,0.003738366],"genre_scores_gemma":[0.5566549,0.0004284972,0.4251301,0.001576354,0.0001693721,0.0007668921,0.007639664,0.003337156,0.004297034],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006039591,"threshold_uncertainty_score":0.0286063,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01927186097653321,"score_gpt":0.2414419262966003,"score_spread":0.2221700653200671,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}