{"id":"W2947766523","doi":"10.24963/ijcai.2019/439","title":"Advantage Amplification in Slowly Evolving Latent-State Environments","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Stylized fact; Abstraction; Computer science; Reinforcement learning; Key (lock); Artificial intelligence; Action (physics); Machine learning; State (computer science); Task (project management); Algorithm; Computer security; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000407669,0.0002917736,0.0002871453,0.0002621625,0.00003913123,0.0003272506,0.001700952,0.0001816281,0.00007377383],"category_scores_gemma":[0.00003980503,0.0002985779,0.00008888202,0.0001431847,0.00002672859,0.0004520322,0.002410696,0.0006903197,0.001124213],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003172976,"about_ca_system_score_gemma":0.0000784058,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005319411,"about_ca_topic_score_gemma":0.000003170719,"domain_scores_codex":[0.9976674,0.00007328279,0.0005335902,0.0008103639,0.0005086185,0.0004067857],"domain_scores_gemma":[0.9978997,0.0000824367,0.0003459801,0.001578612,0.00002187593,0.00007141882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001696326,0.0000223937,0.01511968,0.00005015877,0.00001360715,0.000006538353,0.0003092028,0.979381,0.0003032653,0.001554609,0.0001070116,0.003130805],"study_design_scores_gemma":[0.0002408802,0.00002394675,0.03049537,0.0001045933,0.000004202604,0.000001223401,0.000009398121,0.9665113,0.000229354,0.001101275,0.0009314353,0.0003470379],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006901513,0.0000684354,0.9853594,0.0001416556,0.0009146639,0.0005777526,0.000001570185,0.0001414852,0.005893503],"genre_scores_gemma":[0.8237867,0.0002845376,0.1593408,0.000244358,0.00003435751,0.00004418908,0.00006921106,0.00003872338,0.01615715],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8260186,"threshold_uncertainty_score":0.9999467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01805212001434894,"score_gpt":0.2516236503439548,"score_spread":0.2335715303296058,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}