{"id":"W4389519204","doi":"10.18653/v1/2023.emnlp-main.721","title":"Reward-Augmented Decoding: Efficient Controlled Text Generation With a Unidirectional Reward Model","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"","keywords":"Computer science; Language model; Decoding methods; Overhead (engineering); Cache; Artificial intelligence; Text generation; Algorithm; Parallel computing; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004667675,0.0001400654,0.0001987773,0.0001975298,0.0001965494,0.0001439765,0.0003291664,0.00004963562,0.0000185117],"category_scores_gemma":[0.00004426359,0.00010554,0.00006260906,0.00057443,0.00001798148,0.000156462,0.0001293159,0.00008963794,0.00008398651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008933429,"about_ca_system_score_gemma":0.000145045,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003188376,"about_ca_topic_score_gemma":0.00002746687,"domain_scores_codex":[0.9984655,0.00005228718,0.0002655968,0.0004745637,0.0004538946,0.000288128],"domain_scores_gemma":[0.9992067,0.00006463264,0.0000706468,0.0004252038,0.0001417702,0.00009103341],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000344814,0.00003335442,0.00004502976,0.000003187099,0.00002731146,0.000007399474,0.0002464795,0.9554853,0.001919213,0.03745497,0.001778558,0.002964651],"study_design_scores_gemma":[0.001885901,0.00004096554,0.00002847288,0.000009942135,0.000007361126,0.0000122894,0.00001652369,0.9963142,0.0009466159,0.0003789659,0.0002149019,0.0001438805],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09761047,0.00001725651,0.8948272,0.002007479,0.0002312448,0.0002954712,8.368738e-7,0.0006813079,0.004328718],"genre_scores_gemma":[0.8862886,0.000006952711,0.1078524,0.0004696248,0.0001015476,0.00008120383,0.000007374628,0.00001243675,0.005179946],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7886781,"threshold_uncertainty_score":0.4303797,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04398754862890349,"score_gpt":0.256004301114469,"score_spread":0.2120167524855655,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}