{"id":"W4415002193","doi":"10.1002/smr.70057","title":"Evaluation and Improvement of Test Selection for Large Language Models","year":2025,"lang":"en","type":"article","venue":"Journal of Software Evolution and Process","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"National Key Research and Development Program of China; Université du Luxembourg","keywords":"Selection (genetic algorithm); Test (biology); Margin (machine learning); Process (computing); Harm; Deep learning; Empirical research; Ground truth","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01071992,0.001996123,0.00131441,0.002241459,0.0005444797,0.001186004,0.00305389,0.001874736,0.001815461],"category_scores_gemma":[0.04633984,0.0005353609,0.001075748,0.00105975,0.0009744908,0.001969392,0.001832189,0.00257533,0.0008810385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001220565,"about_ca_system_score_gemma":0.001729711,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003621168,"about_ca_topic_score_gemma":0.004627749,"domain_scores_codex":[0.9908624,0.004914472,0.0006518821,0.001479915,0.001660431,0.0004309189],"domain_scores_gemma":[0.9509532,0.0377036,0.001882314,0.003079158,0.004800829,0.001581048],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003701656,0.00162204,0.05437334,0.0005223356,0.000699196,0.0005305255,0.0002060363,0.3760423,0.02064386,0.002717192,0.01688714,0.5220543],"study_design_scores_gemma":[0.00008963785,0.0003493817,0.001581894,0.00001353756,0.00002963047,0.0000523601,0.00002883332,0.9903829,0.006244252,0.0008379133,0.0003762049,0.00001354916],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7105112,0.004129075,0.2590421,0.001610882,0.0005100897,0.0004467353,0.001300538,0.01960591,0.002843415],"genre_scores_gemma":[0.9386343,0.0001251695,0.05689755,0.0004016583,0.00007722643,0.0001595362,0.002396249,0.0003730918,0.0009351607],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01071992,"threshold_uncertainty_score":0.05669308,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01554874228645765,"score_gpt":0.3038847352079227,"score_spread":0.2883359929214651,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}