{"id":"W7077493384","doi":"10.5281/zenodo.16944474","title":"Using LLMs to Bridge the Gaps in QA Test Plans at Firefox","year":2025,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Test (biology); Test plan; Bridge (graph theory); Code (set theory); Plan (archaeology); Quality assurance; Test harness; Relevance (law); Test case","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003667848,0.001356603,0.0004043575,0.002135405,0.0005445391,0.002068277,0.001910166,0.001002572,0.02963733],"category_scores_gemma":[0.01268907,0.001090158,0.001119728,0.0008435172,0.001034708,0.002044603,0.002130373,0.001991961,0.0152798],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001785694,"about_ca_system_score_gemma":0.001731617,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005739306,"about_ca_topic_score_gemma":0.005652718,"domain_scores_codex":[0.9985995,0.0003733542,0.0001345775,0.0003199979,0.0004385556,0.0001339964],"domain_scores_gemma":[0.9935399,0.003454235,0.0004846235,0.001422555,0.0008460786,0.0002525623],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001336744,0.0006525634,0.005183745,0.001095951,0.0001112483,0.0008331815,0.001741231,0.02020839,0.03671546,0.02748945,0.1767851,0.727847],"study_design_scores_gemma":[0.0009190422,0.0006274039,0.006646076,0.00118911,0.0001019027,0.001137171,0.0003318474,0.3397503,0.1107257,0.03429539,0.5039614,0.0003146123],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.009635375,0.0002093018,0.5998018,0.0005642999,0.0001727812,0.0006810564,0.00293602,0.3705744,0.01542502],"genre_scores_gemma":[0.09103344,0.0003068272,0.7964389,0.0005979115,0.00008363355,0.0009486602,0.01266043,0.07848696,0.01944327],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.02963733,"threshold_uncertainty_score":0.09914678,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04307308353455938,"score_gpt":0.2549190010051355,"score_spread":0.2118459174705761,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}