{"id":"W3208296466","doi":"10.1609/aaai.v36i8.20891","title":"Convergence and Optimality of Policy Gradient Methods in Weakly Smooth Settings","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Convergence (economics); Gradient method; Computer science; Mathematical optimization; Applied mathematics; Work (physics); Rate of convergence; Mathematics; Economics; Physics; Telecommunications","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007945873,0.001314179,0.001343145,0.001533327,0.0008053213,0.002287074,0.0011249,0.001842011,0.002666135],"category_scores_gemma":[0.04089117,0.0006841876,0.001089918,0.0006457492,0.003867765,0.002685139,0.003990862,0.003629299,0.0006911429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001678552,"about_ca_system_score_gemma":0.002318605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002359537,"about_ca_topic_score_gemma":0.001065901,"domain_scores_codex":[0.9973193,0.001587656,0.0001046686,0.0002944033,0.000482801,0.0002112402],"domain_scores_gemma":[0.9853217,0.01193056,0.0005757585,0.0005202619,0.001177726,0.0004739506],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001791012,0.00008575827,0.001309208,0.0002381949,0.00005403577,0.0001237182,0.0003420664,0.477485,0.001994754,0.4900622,0.001602983,0.02652308],"study_design_scores_gemma":[0.0000180125,0.00003477392,0.0001336424,0.00003952095,0.000005722154,0.00001636267,0.00002437124,0.9075592,0.0004521772,0.09120892,0.0004981876,0.000009193886],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02433385,0.0006975434,0.9670954,0.000797983,0.0000519944,0.000095849,0.00004987907,0.0001909886,0.006686443],"genre_scores_gemma":[0.7449962,0.001657315,0.2433509,0.0003918556,0.0001513554,0.0005825658,0.0001745735,0.0004070279,0.008288264],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007945873,"threshold_uncertainty_score":0.04202235,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2328124462140117,"score_gpt":0.4911329585570102,"score_spread":0.2583205123429986,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}