{"id":"W2071814471","doi":"10.1145/1102351.1102472","title":"Bayesian sparse sampling for on-line reward optimization","year":2005,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":122,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Machine learning; Bayesian probability; Sampling (signal processing); Action selection; Artificial intelligence; Bayesian optimization; Thompson sampling; Selection (genetic algorithm); Bayesian inference; Mathematical optimization; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002255764,0.0009223177,0.001644395,0.0007098412,0.0004746097,0.0009450305,0.00146517,0.00131731,0.003496309],"category_scores_gemma":[0.01244097,0.0007488164,0.0005442426,0.0008439115,0.001223319,0.001603749,0.001386503,0.002144269,0.0007544998],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00135506,"about_ca_system_score_gemma":0.001582255,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004599147,"about_ca_topic_score_gemma":0.005815434,"domain_scores_codex":[0.9984425,0.0007559828,0.00004584587,0.0001393258,0.000466181,0.0001502371],"domain_scores_gemma":[0.995276,0.003629533,0.000259422,0.0002724325,0.0004076521,0.0001548478],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001066579,0.00007068943,0.0003730438,0.00006610405,0.00002535379,0.00003549924,0.00004085288,0.9208956,0.0007691351,0.0339901,0.001533236,0.0420938],"study_design_scores_gemma":[0.000008057229,0.000008654721,0.00001993896,0.000003585806,0.000001767189,0.000003524868,0.000001376963,0.9900764,0.0001279549,0.009598591,0.0001484074,0.00000188478],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003945312,0.0001295823,0.9946301,0.0001174024,0.00001576587,0.00002922191,0.00002683472,0.00018899,0.000916827],"genre_scores_gemma":[0.5552104,0.0003945354,0.4399188,0.000301768,0.0001119003,0.0004578192,0.0002412844,0.0001829981,0.003180459],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004599147,"threshold_uncertainty_score":0.01192981,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06390024574608347,"score_gpt":0.3090228942151267,"score_spread":0.2451226484690433,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}