{"id":"W4404782438","doi":"10.18653/v1/2024.emnlp-main.551","title":"StablePrompt : Automatic Prompt Tuning using Reinforcement Learning for Large Language Model","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea","keywords":"Reinforcement learning; Computer science; Language model; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00298895,0.001331023,0.001293792,0.0005391596,0.0005150916,0.001197294,0.00288211,0.001651322,0.005338078],"category_scores_gemma":[0.01329134,0.0007082641,0.0007162577,0.0004942634,0.0008819927,0.002701835,0.002689847,0.003735511,0.002840786],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008157351,"about_ca_system_score_gemma":0.001913979,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002777902,"about_ca_topic_score_gemma":0.004173557,"domain_scores_codex":[0.998579,0.000621901,0.00007562113,0.000405171,0.0002113276,0.0001070671],"domain_scores_gemma":[0.9955975,0.003060887,0.0002099144,0.0005313588,0.0003629515,0.0002373018],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008969785,0.0006882786,0.002751978,0.0006174574,0.000123364,0.0003460324,0.0005233651,0.2969006,0.02079504,0.01138508,0.03407093,0.6309008],"study_design_scores_gemma":[0.00007649602,0.00004943605,0.00008535464,0.000009209348,0.000007331886,0.00002293049,0.00001638025,0.9902089,0.002241354,0.005721692,0.001550214,0.0000107065],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01344798,0.0005021167,0.9511554,0.000278772,0.0001833912,0.0001978521,0.0002469759,0.03269733,0.001290284],"genre_scores_gemma":[0.3949014,0.0003062478,0.5956582,0.000628388,0.0001676852,0.0007943736,0.001277125,0.002729636,0.00353688],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005338078,"threshold_uncertainty_score":0.01785761,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03792502520895687,"score_gpt":0.2995942561215851,"score_spread":0.2616692309126282,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}