{"id":"W4411058752","doi":"10.1145/3727582.3728681","title":"Leveraging LLM Enhanced Commit Messages to Improve Machine Learning Based Test Case Prioritization","year":2025,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"IBM (Canada); Ontario Tech University","funders":"","keywords":"Commit; Prioritization; Computer science; Test (biology); Machine learning; Artificial intelligence; Process management; Engineering; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005142023,0.001248171,0.0007279505,0.002132284,0.0004130668,0.001560832,0.002066198,0.0009075653,0.002207521],"category_scores_gemma":[0.0528041,0.0004021184,0.0005218877,0.0008628984,0.0005949391,0.002265897,0.001235651,0.00225107,0.0009957587],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00101908,"about_ca_system_score_gemma":0.002497589,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003187608,"about_ca_topic_score_gemma":0.005151335,"domain_scores_codex":[0.9951783,0.001892838,0.0004514724,0.0007302989,0.001498675,0.0002482379],"domain_scores_gemma":[0.9577842,0.02432811,0.004812248,0.006046887,0.006097494,0.000931168],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001471906,0.00158077,0.02947608,0.0005449015,0.0000974261,0.0007057564,0.001056627,0.09133901,0.082561,0.004103919,0.008106893,0.7789557],"study_design_scores_gemma":[0.0001239556,0.000598931,0.00432403,0.0000569677,0.0000527419,0.0002551815,0.0001873095,0.9191347,0.06577509,0.004527331,0.004909202,0.00005455204],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2789814,0.0003933075,0.6654438,0.001406077,0.0002259489,0.0007834781,0.0007297114,0.04938939,0.002646803],"genre_scores_gemma":[0.6292804,0.00008181018,0.3647056,0.0004100549,0.00007595572,0.0002555412,0.001647776,0.001285897,0.00225685],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005142023,"threshold_uncertainty_score":0.0271939,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01080975645387326,"score_gpt":0.2664477481246184,"score_spread":0.2556379916707451,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}