{"id":"W2140035747","doi":"","title":"Action selection in Bayesian reinforcement learning","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Action selection; Machine learning; Computer science; Artificial intelligence; Bayesian probability; Markov decision process; Selection (genetic algorithm); Variable-order Bayesian network; Action (physics); Perspective (graphical); Bayesian inference; Markov process; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003499861,0.001085256,0.001663774,0.0006593346,0.0004818816,0.001100214,0.00184263,0.001601669,0.003803117],"category_scores_gemma":[0.01370897,0.0007920624,0.0006592702,0.0007322779,0.00204245,0.0024571,0.001334907,0.002440835,0.0005571507],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00182975,"about_ca_system_score_gemma":0.001414746,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005285316,"about_ca_topic_score_gemma":0.005499345,"domain_scores_codex":[0.9976881,0.001246464,0.00008431232,0.0003154149,0.0004876239,0.0001781498],"domain_scores_gemma":[0.9931463,0.005483787,0.0003844046,0.0002548603,0.0004906025,0.0002400158],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001322972,0.0001079213,0.0009709139,0.0001188592,0.00007441759,0.0000650334,0.0001376243,0.8443677,0.0006127929,0.1006146,0.00133004,0.05146778],"study_design_scores_gemma":[0.00003044319,0.00002927355,0.00007668073,0.00001035609,0.000008077681,0.000008640101,0.000006230649,0.9555749,0.0001574453,0.04368626,0.0004040673,0.000007621437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01391846,0.0003488882,0.981222,0.0005163572,0.00004716348,0.00006121951,0.00004319202,0.0001917352,0.003650999],"genre_scores_gemma":[0.795929,0.0005859812,0.1958569,0.0004552705,0.000149634,0.0004646126,0.0001523754,0.00009769171,0.006308498],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005285316,"threshold_uncertainty_score":0.01850927,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01405629486968636,"score_gpt":0.2499175292317624,"score_spread":0.235861234362076,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}