{"id":"W2997475283","doi":"10.1609/aaai.v34i04.5857","title":"Algorithmic Improvements for Deep Reinforcement Learning Applied to Interactive Fiction","year":2020,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Institute for Advanced Research; McGill University","funders":"Compute Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Observability; Computer science; Artificial intelligence; Action (physics); Function (biology); Domain (mathematical analysis); Baseline (sea); Point (geometry); Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001054107,0.0001086764,0.0001071362,0.0000514686,0.0001162916,0.0001284824,0.0004435775,0.00002982247,0.00007036611],"category_scores_gemma":[0.00009461965,0.0001039963,0.0000456626,0.0002150525,0.000009644299,0.0002968807,0.000258396,0.0001010897,0.0002583436],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007358993,"about_ca_system_score_gemma":0.00001936553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002376364,"about_ca_topic_score_gemma":0.000004642092,"domain_scores_codex":[0.9989813,0.00001079973,0.0002377696,0.0003641908,0.0001731001,0.0002328279],"domain_scores_gemma":[0.9994783,0.00007577726,0.00007166795,0.0001581558,0.00008809041,0.0001280231],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008465329,0.00003070248,0.0000282973,0.00001573639,0.00004199951,0.000001187914,0.01174324,0.2053891,0.02628605,0.0250719,0.002498426,0.7288087],"study_design_scores_gemma":[0.00007652474,0.0003499844,0.000009707343,0.00000353421,0.000002696944,3.009245e-7,0.0007564297,0.8601572,0.1289367,0.0004605565,0.009122569,0.0001238202],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001034937,0.000002241068,0.9905301,0.001267091,0.00024925,0.000697041,1.785895e-7,0.0002159847,0.006003221],"genre_scores_gemma":[0.908294,8.466498e-7,0.08810812,0.002837044,0.0001284438,0.0001835135,0.000002850615,0.000009060772,0.0004361582],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.907259,"threshold_uncertainty_score":0.4240844,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02986483739447762,"score_gpt":0.290029295663619,"score_spread":0.2601644582691414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}