{"id":"W2890185089","doi":"","title":"Monte-Carlo Tree Search for Constrained POMDPs","year":2018,"lang":"en","type":"article","venue":"Neural Information Processing Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":31,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Monte Carlo tree search; Computer science; Partially observable Markov decision process; Mathematical optimization; Monte Carlo method; Markov decision process; Tree (set theory); Action selection; Scale (ratio); Markov process; Machine learning; Markov model; Mathematics; Markov chain","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00174501,0.00125677,0.001678974,0.0007232064,0.0007013428,0.001021748,0.001424554,0.001620763,0.00572188],"category_scores_gemma":[0.007960003,0.0008343482,0.0009789519,0.0008469151,0.001343438,0.00141482,0.001555802,0.002314292,0.0005033601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00150839,"about_ca_system_score_gemma":0.002645431,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008495186,"about_ca_topic_score_gemma":0.009017726,"domain_scores_codex":[0.9992051,0.0003306995,0.0000423905,0.000139407,0.0001802721,0.0001022625],"domain_scores_gemma":[0.9951891,0.004044799,0.0002238537,0.0001255036,0.0002572992,0.0001594541],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003403025,0.00002047434,0.0002291497,0.00005034192,0.00001562202,0.00003181241,0.0000215543,0.9783145,0.0001450426,0.01336199,0.0004878938,0.007287666],"study_design_scores_gemma":[0.00001009637,0.000006904401,0.0000170352,0.000004704245,0.00000244979,0.000003704145,0.000002811233,0.9937974,0.00004720106,0.005903375,0.0002025783,0.000001709543],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01460313,0.0003612688,0.9796382,0.0002644672,0.00004790552,0.000106274,0.0001254365,0.0003739192,0.00447935],"genre_scores_gemma":[0.602487,0.0004192382,0.3921625,0.000294768,0.00006989703,0.0007206461,0.0004797956,0.0002515625,0.003114547],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008495186,"threshold_uncertainty_score":0.01914155,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03579990415296933,"score_gpt":0.284999641127341,"score_spread":0.2491997369743717,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}