{"id":"W2108005621","doi":"10.1609/aaai.v26i1.8260","title":"Sample Bounded Distributed Reinforcement Learning for Decentralized POMDPs","year":2021,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"","keywords":"Reinforcement learning; Computer science; Partially observable Markov decision process; Benchmark (surveying); Markov decision process; Bounded function; Mathematical optimization; Sample (material); Sample complexity; Computation; Set (abstract data type); Artificial intelligence; Markov process; Markov chain; Machine learning; Mathematics; Markov model; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003504028,0.001041823,0.001601152,0.0004400761,0.000631093,0.001197093,0.001421988,0.001090189,0.002013648],"category_scores_gemma":[0.01682935,0.00067605,0.0006549965,0.0004459966,0.001925186,0.001734358,0.001922209,0.002674219,0.0001790964],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002421401,"about_ca_system_score_gemma":0.002088703,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004270456,"about_ca_topic_score_gemma":0.004178923,"domain_scores_codex":[0.9982637,0.0007433659,0.00008467233,0.0003040332,0.0004199192,0.0001842188],"domain_scores_gemma":[0.9868623,0.01058138,0.0009039173,0.0006321298,0.0006269439,0.0003932921],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005351675,0.00002768932,0.0002546709,0.00003408616,0.00001316165,0.00002453935,0.00002481285,0.9836088,0.0002469883,0.01146532,0.0001624207,0.004084087],"study_design_scores_gemma":[0.000009310274,0.00001031818,0.00002425539,0.00000195611,0.000001334302,0.000002232621,0.000002690131,0.9915904,0.00008960245,0.008208821,0.00005762274,0.000001462334],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0277174,0.0001394559,0.9699755,0.0003068139,0.00002265447,0.00005730741,0.00005218536,0.0002604321,0.001468381],"genre_scores_gemma":[0.9143212,0.0001324044,0.08374402,0.0000908092,0.0000290304,0.0002616442,0.0001311081,0.00007368232,0.001216212],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004270456,"threshold_uncertainty_score":0.01853126,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07689906790923026,"score_gpt":0.3054622507888089,"score_spread":0.2285631828795786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}