{"id":"W3026736587","doi":"10.48550/arxiv.2005.11335","title":"Single-Agent Optimization Through Policy Iteration Using Monte-Carlo Tree Search","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Monte Carlo tree search; Computer science; Reinforcement learning; Normalization (sociology); Set (abstract data type); Search algorithm; Tree (set theory); Iterative deepening depth-first search; Monte Carlo method; Beam search; Mathematical optimization; Artificial intelligence; Algorithm; Best-first search; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00158151,0.001024653,0.001593801,0.0007959044,0.0005129488,0.0009662612,0.001468156,0.00173811,0.004024198],"category_scores_gemma":[0.006389765,0.0006454847,0.000672923,0.0008708448,0.001291438,0.001242047,0.001214965,0.001708058,0.0007421385],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001334653,"about_ca_system_score_gemma":0.002519302,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007248133,"about_ca_topic_score_gemma":0.009143168,"domain_scores_codex":[0.9994243,0.0002353143,0.00002862497,0.00009594444,0.0001429837,0.00007283228],"domain_scores_gemma":[0.996283,0.002917303,0.0001908487,0.0001977275,0.0002414181,0.0001697131],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004795605,0.00004320136,0.0004293146,0.00003561925,0.00002870124,0.00003406764,0.00002813489,0.9666432,0.0002706182,0.01257544,0.001065252,0.0187985],"study_design_scores_gemma":[0.000007520121,0.000005802239,0.00001320132,0.000002584689,0.000001696034,0.000003428124,0.000001884915,0.9972939,0.00005372582,0.002488419,0.0001266003,0.000001171607],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02175307,0.0002948465,0.9703816,0.000341703,0.00008676702,0.0000849619,0.00005770247,0.0008755519,0.006123861],"genre_scores_gemma":[0.6103547,0.0001933529,0.3832515,0.0003527694,0.00007896269,0.0004606699,0.000198887,0.000300529,0.004808733],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007248133,"threshold_uncertainty_score":0.01441187,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2789313618876238,"score_gpt":0.2631439524162559,"score_spread":0.01578740947136786,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}