{"id":"W4225848424","doi":"10.1609/aaai.v36i7.20764","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","year":2022,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Reinforcement learning; Observable; Computer science; Latent variable; Block (permutation group theory); Artificial intelligence; Artificial neural network; Markov process; Machine learning; Markov decision process; Variable (mathematics); Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001096223,0.0007516913,0.001128079,0.0003099284,0.000309595,0.0005918344,0.001353326,0.0008977058,0.002692821],"category_scores_gemma":[0.00276718,0.0005176463,0.0005522965,0.0003347301,0.0007429704,0.001214661,0.001054363,0.001627589,0.0003849148],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000936331,"about_ca_system_score_gemma":0.00139827,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005605086,"about_ca_topic_score_gemma":0.006285844,"domain_scores_codex":[0.9995104,0.0001724711,0.00002170628,0.0001112224,0.0001279963,0.00005622754],"domain_scores_gemma":[0.9991456,0.0004764002,0.00009709202,0.00008657957,0.0001372997,0.00005700054],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005360716,0.00004001097,0.0003551331,0.00004888521,0.00003255621,0.00003907986,0.0000416469,0.9455801,0.001424288,0.02197119,0.0007156216,0.02969788],"study_design_scores_gemma":[0.000002711454,0.000008786444,0.00001436889,8.638907e-7,0.000001307222,0.000002326161,7.221093e-7,0.9971042,0.0001027541,0.002652421,0.0001083858,0.00000118832],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006876817,0.00009073036,0.992056,0.00008720047,0.00002658982,0.00002008168,0.00002648944,0.000213786,0.0006023544],"genre_scores_gemma":[0.7916082,0.000246586,0.2028359,0.0001781427,0.00007014209,0.0002829644,0.0002359648,0.0001165967,0.004425486],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005605086,"threshold_uncertainty_score":0.01114488,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1480917507999508,"score_gpt":0.3238548945512752,"score_spread":0.1757631437513243,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}