{"id":"W2100785108","doi":"10.7551/mitpress/7503.003.0062","title":"Bayesian Policy Gradient Algorithms","year":2007,"lang":"en","type":"book-chapter","venue":"The MIT Press eBooks","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Bayesian probability; Computer science; Algorithm; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025596,0.002113917,0.002276008,0.001374376,0.0008698372,0.002661515,0.002969465,0.002458009,0.0183328],"category_scores_gemma":[0.009630341,0.001191791,0.000903022,0.001731591,0.001354469,0.003183526,0.002393002,0.002924052,0.006969027],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00194434,"about_ca_system_score_gemma":0.002691027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005256637,"about_ca_topic_score_gemma":0.005057265,"domain_scores_codex":[0.9982662,0.0006290472,0.00008855371,0.0002891527,0.0005672771,0.0001598086],"domain_scores_gemma":[0.9980646,0.001126506,0.0001567287,0.00017882,0.000395037,0.00007839024],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001186011,0.000142952,0.0006347791,0.0004005961,0.0001203385,0.00005697396,0.00009748002,0.330004,0.0006483807,0.2537549,0.02476422,0.3892568],"study_design_scores_gemma":[0.00006127701,0.00003144977,0.0001672062,0.0001052338,0.00002689312,0.00004672017,0.00001834561,0.7847727,0.0007908118,0.1849424,0.02900892,0.00002805399],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0007606031,0.001473881,0.9870446,0.0002997174,0.0001227049,0.0001027112,0.0001257112,0.0008005696,0.009269485],"genre_scores_gemma":[0.111871,0.004556104,0.8492439,0.0006432585,0.000364257,0.0009476403,0.0008407934,0.0008031041,0.03072992],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0183328,"threshold_uncertainty_score":0.06132936,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05302789778127596,"score_gpt":0.2785326680918616,"score_spread":0.2255047703105856,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}