{"id":"W2140035747","doi":"","title":"Action selection in Bayesian reinforcement learning","year":2006,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Action selection; Machine learning; Computer science; Artificial intelligence; Bayesian probability; Markov decision process; Selection (genetic algorithm); Variable-order Bayesian network; Action (physics); Perspective (graphical); Bayesian inference; Markov process; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002358048,0.00009547723,0.00008238672,0.0002004603,0.00009976092,0.000131758,0.0002240174,0.00005246604,0.00007640277],"category_scores_gemma":[0.00002179347,0.00009600167,0.00002886212,0.0005086361,0.000009612158,0.0005628166,0.00007192036,0.0002113999,0.00008444098],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001667339,"about_ca_system_score_gemma":0.00003028468,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004436522,"about_ca_topic_score_gemma":0.00009680985,"domain_scores_codex":[0.9989795,0.00004803714,0.0002492704,0.0002153156,0.0002505094,0.0002573506],"domain_scores_gemma":[0.9997022,0.00003113079,0.0000836799,0.0001222711,0.00003346501,0.00002725098],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001708719,0.000007359141,0.008639298,0.000004048924,0.000001652992,0.000001429947,0.00004153633,0.9586403,0.0008943626,0.02930001,0.0002594612,0.00220888],"study_design_scores_gemma":[0.0002161757,0.00007686849,0.009835611,0.000008283077,0.000001072106,0.000005050223,0.00001698907,0.9831927,0.002154995,0.0002945019,0.004080295,0.0001174584],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002661383,0.000003486743,0.9464512,0.0001792144,0.0001170346,0.0001015382,3.447793e-9,0.000274414,0.05021176],"genre_scores_gemma":[0.9660203,0.00000407678,0.0196898,0.00006859368,0.00005676534,0.000008202712,0.000002575488,0.00000637866,0.01414331],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9633589,"threshold_uncertainty_score":0.3914835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01405629486968636,"score_gpt":0.2499175292317624,"score_spread":0.235861234362076,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}