{"id":"W4283801867","doi":"10.1609/aaai.v36i7.20764","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","year":2022,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Observable; Reinforcement learning; Computer science; Latent variable; Block (permutation group theory); Artificial intelligence; Artificial neural network; Markov process; Variable (mathematics); Machine learning; Markov chain; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.001096746,0.0002553705,0.0002599261,0.000154136,0.001625147,0.0003213598,0.001314606,0.00005456744,0.0003500913],"category_scores_gemma":[0.0001646781,0.0002796948,0.0001900241,0.0003987436,0.00003204516,0.0005682815,0.00155705,0.0006790732,0.00004361264],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002505935,"about_ca_system_score_gemma":0.0002798032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003382116,"about_ca_topic_score_gemma":0.000002249793,"domain_scores_codex":[0.9970753,0.0001601133,0.0005393227,0.0005933674,0.000850083,0.0007818771],"domain_scores_gemma":[0.9987401,0.0001519322,0.0002836102,0.0005353689,0.0001365545,0.0001523974],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002451242,0.0000197023,0.0002608621,0.00001829295,0.00003462914,0.000004825072,0.000659742,0.9569856,0.0009270739,0.0388077,0.001405556,0.0008514887],"study_design_scores_gemma":[0.0007455645,0.0005553709,0.000007790494,0.000005560281,0.00001747841,0.000008034292,0.0001388118,0.9401147,0.0007476675,0.0004044443,0.05691121,0.0003433063],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001571101,0.00002276194,0.9873211,0.0005898941,0.0004278288,0.0005419915,3.730529e-7,0.0006819971,0.008842939],"genre_scores_gemma":[0.8202723,0.000009616918,0.08326696,0.0005985442,0.00008038855,0.0003367442,0.00004315214,0.00004105478,0.09535127],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9040542,"threshold_uncertainty_score":0.9999655,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0468047699607824,"score_gpt":0.2675118803325101,"score_spread":0.2207071103717277,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}