{"id":"W4410536892","doi":"10.1109/icst62969.2025.10988995","title":"An Analysis of LLM Fine-Tuning and Few-Shot Learning for Flaky Test Detection and Classification","year":2025,"lang":"en","type":"article","venue":"","topic":"Metallurgy and Material Forming","field":"Engineering","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ontario Tech University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Shot (pellet); Computer science; Test (biology); One shot; Artificial intelligence; Machine learning; Materials science; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001419123,0.00005294661,0.0001201717,0.0001889713,0.00006199439,0.00002356857,0.00002066032,0.00004257982,0.00001630134],"category_scores_gemma":[0.00005921704,0.00005143761,0.00001956175,0.0001919479,0.00001116839,0.0001060597,0.000006795445,0.00003327268,2.301133e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000008575917,"about_ca_system_score_gemma":0.000002572867,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001837839,"about_ca_topic_score_gemma":0.0001666315,"domain_scores_codex":[0.9996904,0.000007735567,0.0001248949,0.00008847761,0.00002476285,0.0000637695],"domain_scores_gemma":[0.9998034,0.0000740074,0.00002063975,0.00006139943,0.00002165081,0.00001888949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009499159,0.000006895779,0.003136387,0.0001141042,0.0001191887,5.553369e-8,0.0001033056,0.01180189,0.9545867,0.001258021,0.000002010763,0.02886196],"study_design_scores_gemma":[0.0001166643,0.00004123021,0.03160997,0.00001129756,0.0002331772,2.891443e-7,0.0001477677,0.9074274,0.05966042,0.00006390993,0.0006284047,0.00005941791],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8254983,0.00003751923,0.173162,0.000006478064,0.00005605127,0.00005884644,0.000001023317,0.00006704764,0.001112654],"genre_scores_gemma":[0.9986691,0.0000232119,0.001030521,0.000003263869,0.00001065775,0.000009658125,0.0000132351,0.000005035765,0.0002352569],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8956256,"threshold_uncertainty_score":0.2097565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01702181613840942,"score_gpt":0.2534774304121387,"score_spread":0.2364556142737293,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}