{"id":"W2726989136","doi":"10.1007/978-3-319-63004-5_6","title":"A Deterministic Actor-Critic Approach to Stochastic Reinforcements","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Game Theory and Applications","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Variance (accounting); Function (biology); Computer science; Reinforcement; Artificial intelligence; Temporal difference learning; Mathematical optimization; Mathematics; Psychology; Economics; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001265874,0.001034337,0.001174681,0.0004414231,0.0004450355,0.00142721,0.002129126,0.001846647,0.0044681],"category_scores_gemma":[0.003956914,0.0007819899,0.0008380733,0.0006113435,0.00125549,0.001129237,0.001490168,0.002415954,0.0006821872],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001496423,"about_ca_system_score_gemma":0.001217852,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004322309,"about_ca_topic_score_gemma":0.004695648,"domain_scores_codex":[0.9992504,0.0003364363,0.0000377531,0.0001229479,0.0001930706,0.00005943703],"domain_scores_gemma":[0.9986199,0.0009472444,0.00008204025,0.00007861403,0.0002025881,0.00006954304],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002312625,0.00001929063,0.0001088075,0.00005153261,0.00003910868,0.00004630495,0.00003444437,0.83667,0.0006019613,0.1448489,0.001537257,0.01601928],"study_design_scores_gemma":[0.000004113599,0.000006460124,0.00001776151,0.000005466514,0.000005176628,0.00000717012,0.000001859505,0.9656881,0.00008874319,0.03363317,0.0005380317,0.000003924531],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001864389,0.0003862577,0.9913006,0.0002572901,0.0001005423,0.00001328144,0.0000227557,0.00009665471,0.00595822],"genre_scores_gemma":[0.6418667,0.001694958,0.3211388,0.000350703,0.0005004059,0.0002229416,0.000149754,0.0002801067,0.0337956],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0044681,"threshold_uncertainty_score":0.01494735,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1002996725334714,"score_gpt":0.3690955365600705,"score_spread":0.2687958640265991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}