{"id":"W4400484851","doi":"10.1145/3663529.3663855","title":"Leveraging Large Language Models for the Auto-remediation of Microservice Applications: An Experimental Study","year":2024,"lang":"en","type":"article","venue":"","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"IBM (Canada); York University","funders":"","keywords":"Computer science; Language model; Microservices; Natural language processing; Cloud computing; Operating system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002498088,0.001942402,0.0006710111,0.0006243397,0.0004916433,0.001049888,0.002171923,0.00139928,0.003059458],"category_scores_gemma":[0.01347605,0.0005181308,0.001047052,0.0004232805,0.0008281545,0.002390093,0.001273431,0.003524154,0.002004445],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009757455,"about_ca_system_score_gemma":0.001290624,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009373776,"about_ca_topic_score_gemma":0.01473122,"domain_scores_codex":[0.9978374,0.0009052869,0.0001407732,0.0007117995,0.0002526323,0.0001521775],"domain_scores_gemma":[0.9890099,0.007748001,0.0004171001,0.001660204,0.0008481547,0.000316573],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002677604,0.007028408,0.02958767,0.001876515,0.0005673047,0.001541693,0.001832341,0.3110191,0.04833648,0.004007314,0.04467434,0.5468513],"study_design_scores_gemma":[0.0002629638,0.0009722207,0.005226024,0.00007827632,0.0001343477,0.0003370178,0.0004582482,0.9504212,0.02954142,0.002939779,0.009532673,0.00009583869],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8507466,0.001923252,0.1045808,0.001193244,0.0006606798,0.0006584126,0.004343423,0.02969287,0.006200732],"genre_scores_gemma":[0.8895862,0.0003941301,0.09209371,0.0006952722,0.0001011692,0.0004020712,0.01142717,0.001145248,0.004155034],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009373776,"threshold_uncertainty_score":0.01863843,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02218837485246432,"score_gpt":0.3130548366701679,"score_spread":0.2908664618177036,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}