{"id":"W4386140507","doi":"10.1007/s10664-023-10363-2","title":"A comparison of reinforcement learning frameworks for software testing tasks","year":2023,"lang":"en","type":"article","venue":"Empirical Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Regression testing; Software performance testing; Software engineering; System integration testing; Software reliability testing; Software development; Adaptation (eye); Context (archaeology); Test strategy; Reinforcement learning; Software inspection; Manual testing; Software testing; Software construction; Machine learning; Software; Software quality; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01013823,0.0008326941,0.001251961,0.001459247,0.0005108964,0.001552112,0.00268524,0.001681937,0.002464345],"category_scores_gemma":[0.0381407,0.0004474191,0.0007257073,0.0008434043,0.001060918,0.002579225,0.001626477,0.002085396,0.0004923152],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002282094,"about_ca_system_score_gemma":0.002836999,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007389067,"about_ca_topic_score_gemma":0.005608095,"domain_scores_codex":[0.9950066,0.002737408,0.0002599138,0.0003649395,0.001265639,0.0003656197],"domain_scores_gemma":[0.9532519,0.03867261,0.001200451,0.002481481,0.003197604,0.001195987],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003247918,0.001953824,0.00643556,0.0005297229,0.0002028203,0.00006232991,0.0005686759,0.409329,0.002634111,0.04275133,0.002124801,0.5301599],"study_design_scores_gemma":[0.0002072941,0.0005641889,0.002207078,0.00005893735,0.00005135781,0.00003704463,0.00007937659,0.9826974,0.001014434,0.01184347,0.001206994,0.00003234466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2330351,0.003478516,0.7472488,0.0009183697,0.0001964668,0.0004726429,0.0001173063,0.003286756,0.01124606],"genre_scores_gemma":[0.8627403,0.0008000259,0.133407,0.0001246526,0.00004897991,0.00025319,0.0001381523,0.0002589108,0.002228786],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01013823,"threshold_uncertainty_score":0.0536167,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06973939181903052,"score_gpt":0.3503954270661822,"score_spread":0.2806560352471517,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}