{"id":"W4386566626","doi":"10.18653/v1/2023.eacl-main.88","title":"Policy-based Reinforcement Learning for Generalisation in Interactive Text-based Environments","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"University of Cape Town; National Research Foundation","keywords":"Converse; Reinforcement learning; Computer science; Benchmark (surveying); Natural language; Artificial intelligence; Variety (cybernetics); Simple (philosophy); Question answering; Value (mathematics); Natural language understanding; Baseline (sea); Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002596852,0.00008242553,0.0000797091,0.000289098,0.00005827127,0.00004995214,0.0002449682,0.00003241175,0.00001430376],"category_scores_gemma":[0.00007429154,0.00008122833,0.00003887903,0.0002716283,0.000008361641,0.0002060234,0.00007225676,0.00007177817,0.00004858206],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001792785,"about_ca_system_score_gemma":0.00007623275,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001904778,"about_ca_topic_score_gemma":0.0000183742,"domain_scores_codex":[0.99912,0.00004007179,0.0001888732,0.0002581683,0.0001686773,0.0002242319],"domain_scores_gemma":[0.9995555,0.0001191534,0.00005838126,0.0002202915,0.0000098422,0.00003685815],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006630215,0.00001017152,0.0005827118,0.000005655371,0.000002510688,0.000001019642,0.000261783,0.9802301,0.00197649,0.00650942,0.00005271682,0.01036082],"study_design_scores_gemma":[0.0006063465,0.00004850335,0.001023843,0.00001214021,8.716994e-7,8.251588e-8,0.0000269766,0.9877575,0.008470902,0.000574876,0.001387997,0.00008993429],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02387895,0.00000102309,0.9729236,0.002186439,0.00008393173,0.0002409527,2.533805e-7,0.00011293,0.0005718653],"genre_scores_gemma":[0.9637889,8.132502e-7,0.03375834,0.0009288163,0.00004229943,0.000095616,0.00002115856,0.000007189607,0.001356878],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9399099,"threshold_uncertainty_score":0.3312395,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.035474172160724,"score_gpt":0.2884375952942587,"score_spread":0.2529634231335347,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}