{"id":"W7153576139","doi":"10.1145/3779657.3779658","title":"From Scenario to Code: Structured Prompting for LLM-Based Unit Test Generation","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université du Québec à Trois-Rivières; Innovation and Economic Development Trois Rivières","funders":"","keywords":"Unit testing; Test case; Code (set theory); Readability; Test (biology); Java; Software; Relevance (law); Code coverage","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001798042,0.001548311,0.0005909157,0.001677976,0.000340344,0.0009910676,0.001758214,0.001301254,0.005806943],"category_scores_gemma":[0.01578239,0.0004197157,0.0009011736,0.0007049177,0.0008127324,0.001567954,0.002269852,0.001355466,0.002188234],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006266237,"about_ca_system_score_gemma":0.001629497,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001937534,"about_ca_topic_score_gemma":0.003579209,"domain_scores_codex":[0.997612,0.001027194,0.0001663144,0.0005737466,0.000454989,0.0001657891],"domain_scores_gemma":[0.9923023,0.004699106,0.0005260489,0.001319891,0.0008526418,0.0002999852],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001434614,0.0008039073,0.02394207,0.001341157,0.0001011852,0.002308132,0.001792955,0.08504537,0.05717676,0.00964074,0.0509303,0.7654829],"study_design_scores_gemma":[0.0003562811,0.0005364696,0.004040433,0.0001949324,0.00008460049,0.0009924421,0.0006011561,0.8593212,0.06945694,0.03584596,0.02846649,0.0001031502],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08212508,0.0005839364,0.787745,0.000974575,0.000180019,0.0006948009,0.003813382,0.1205672,0.003316112],"genre_scores_gemma":[0.4622176,0.0001726273,0.5207739,0.0007302369,0.00006667853,0.0005802002,0.009472589,0.004107276,0.001878902],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005806943,"threshold_uncertainty_score":0.01942617,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05725744101598945,"score_gpt":0.327343913897419,"score_spread":0.2700864728814296,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}