{"id":"W4415221973","doi":"10.1109/ms.2025.3621128","title":"Using LLMs to Bridge the Gaps in QA Test Plans at Firefox","year":2025,"lang":"en","type":"article","venue":"IEEE Software","topic":"SAS software applications and methods","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ministère des Ressources naturelles et des Forêts (Québec); Government of Canada; University of Calgary","funders":"","keywords":"Bridge (graph theory); Test (biology); Quality assurance; Reliability (semiconductor); Test plan; Software quality; Process (computing)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006227601,0.0008274556,0.0002687797,0.00163579,0.0005233956,0.0009893879,0.001306515,0.0008231209,0.002251969],"category_scores_gemma":[0.02997684,0.0004840689,0.0005900734,0.000605679,0.0009987234,0.001226411,0.001371633,0.001099437,0.0004858044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001504717,"about_ca_system_score_gemma":0.001831023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008265564,"about_ca_topic_score_gemma":0.01382449,"domain_scores_codex":[0.9972018,0.001631175,0.0001761735,0.0003337518,0.0004837788,0.000173261],"domain_scores_gemma":[0.976225,0.01742978,0.001464831,0.002569506,0.001890261,0.0004205367],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001445542,0.001478725,0.04348951,0.0006646077,0.0001281724,0.0009276866,0.005081546,0.211177,0.05782131,0.01039019,0.00799849,0.6593971],"study_design_scores_gemma":[0.0002534664,0.00130119,0.008215176,0.0002341429,0.00009563986,0.0004000755,0.0006838222,0.9049304,0.0646679,0.006423036,0.01271487,0.00008029159],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3612001,0.0002341073,0.6083425,0.0006787615,0.00006071762,0.0007243958,0.0004766212,0.02394979,0.004333044],"genre_scores_gemma":[0.4018951,0.00005582153,0.5951054,0.0001770241,0.000008105731,0.0002794201,0.0005943563,0.0008145415,0.001070231],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008265564,"threshold_uncertainty_score":0.03293508,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0331678809201386,"score_gpt":0.3164767833720971,"score_spread":0.2833089024519584,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}