{"id":"W4416235960","doi":"10.48550/arxiv.2511.09038","title":"Test Plan Generation for Live Testing of Cloud Services","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Test (biology); Plan (archaeology); Software deployment; Test plan; Test Management Approach; Production (economics); Cloud computing; Task (project management)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001932067,0.001104119,0.000370073,0.001297561,0.0004596659,0.0009823528,0.001262082,0.0007328902,0.00710737],"category_scores_gemma":[0.008700751,0.0003769842,0.0007397181,0.0005376486,0.0009627848,0.0008066415,0.001186278,0.0008973266,0.001039529],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009376559,"about_ca_system_score_gemma":0.001803644,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005265948,"about_ca_topic_score_gemma":0.005603903,"domain_scores_codex":[0.9981445,0.0008277777,0.00009064507,0.0002351136,0.0005333078,0.0001686832],"domain_scores_gemma":[0.9943117,0.003614191,0.0004067869,0.0007814327,0.000695189,0.0001907385],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007925269,0.0007172499,0.00640331,0.0006389082,0.0001050069,0.001729677,0.0007588909,0.3736595,0.04582101,0.02594113,0.01778069,0.5256522],"study_design_scores_gemma":[0.0001342371,0.0002570243,0.001095699,0.00007422501,0.00003669044,0.0003418055,0.0001431935,0.9396597,0.03410091,0.01497183,0.009151015,0.00003361745],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02843292,0.0001394975,0.9556991,0.0002278224,0.00004299288,0.0005603728,0.0006279252,0.01116459,0.003104635],"genre_scores_gemma":[0.3580803,0.0001404045,0.6346806,0.0001449034,0.00002495767,0.0006138281,0.003223513,0.001202956,0.00188857],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00710737,"threshold_uncertainty_score":0.02377653,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07671994857185874,"score_gpt":0.2854198562953189,"score_spread":0.2086999077234601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}