{"id":"W2726989136","doi":"10.1007/978-3-319-63004-5_6","title":"A Deterministic Actor-Critic Approach to Stochastic Reinforcements","year":2017,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Game Theory and Applications","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Variance (accounting); Function (biology); Computer science; Reinforcement; Artificial intelligence; Temporal difference learning; Mathematical optimization; Mathematics; Psychology; Economics; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002431236,0.0004488762,0.000608184,0.0009462024,0.0006816395,0.0012674,0.005597729,0.0002268051,0.0001110185],"category_scores_gemma":[0.00243583,0.0003583283,0.0001433782,0.0003369752,0.001197712,0.0003292176,0.001247369,0.0005548595,0.0008155481],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001545004,"about_ca_system_score_gemma":0.0003994204,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000006976353,"about_ca_topic_score_gemma":0.00001527149,"domain_scores_codex":[0.9949405,0.00004032233,0.0007539953,0.001652643,0.002002735,0.0006098652],"domain_scores_gemma":[0.9944639,0.001576195,0.0004071807,0.00282538,0.0003799321,0.0003474359],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00003036517,0.00005965388,0.00001233401,0.0000231451,0.0000124608,0.00002390738,0.001614374,0.1190704,0.0001800128,0.06672387,0.0001205477,0.812129],"study_design_scores_gemma":[0.0002369433,0.0001679191,0.0001530959,0.0002556958,0.00002111767,0.00006135381,0.000001361792,0.1608779,0.0001092844,0.8331711,0.004208234,0.0007360302],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0002147957,0.0000297471,0.9557107,0.0002273222,0.000792505,0.0006491864,0.00002044355,0.00004540493,0.04230988],"genre_scores_gemma":[0.9538777,0.00000119352,0.0376747,0.001055128,0.0003906163,0.00004268268,0.000005373554,0.00002955303,0.006923021],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9536629,"threshold_uncertainty_score":0.9999624,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1002996725334714,"score_gpt":0.3690955365600705,"score_spread":0.2687958640265991,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}