{"id":"W4225848424","doi":"10.1609/aaai.v36i7.20764","title":"Blockwise Sequential Model Learning for Partially Observable Reinforcement Learning","year":2022,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"National Research Foundation of Korea; National Research Foundation","keywords":"Reinforcement learning; Observable; Computer science; Latent variable; Block (permutation group theory); Artificial intelligence; Artificial neural network; Markov process; Machine learning; Markov decision process; Variable (mathematics); Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001777161,0.0006804159,0.0007676685,0.0002861875,0.001043775,0.0009608407,0.005623844,0.0003253492,0.0001941187],"category_scores_gemma":[0.001350312,0.0006282656,0.0005499944,0.0005592981,0.0002576107,0.0004607374,0.005920912,0.002682238,0.00003408584],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003366556,"about_ca_system_score_gemma":0.0007748823,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006285763,"about_ca_topic_score_gemma":0.000004104274,"domain_scores_codex":[0.9946226,0.00007624093,0.001515924,0.001315786,0.001532488,0.0009369603],"domain_scores_gemma":[0.9955468,0.0002193057,0.002037942,0.0007615866,0.001258249,0.0001761064],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000883356,0.00005054368,0.00007877832,0.0002209438,0.00005779117,4.818026e-7,0.00155921,0.7264161,0.003399348,0.2651346,0.0001339078,0.002859968],"study_design_scores_gemma":[0.00006434446,0.0004892159,0.000005176941,0.0003643083,0.0000683875,0.00000165316,0.0003852052,0.9061012,0.04771074,0.04339894,0.0008292898,0.0005815598],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.00864207,0.00002926423,0.9765597,0.00184015,0.001410495,0.001855258,0.000004469375,0.0003702261,0.009288367],"genre_scores_gemma":[0.972746,0.0001206862,0.01777588,0.0001892134,0.0001640066,0.000491689,0.00001780731,0.00006889545,0.008425872],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9641039,"threshold_uncertainty_score":0.9997562,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1480917507999508,"score_gpt":0.3238548945512752,"score_spread":0.1757631437513243,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}