{"id":"W2050992779","doi":"10.1103/physreve.67.026706","title":"Convergence of reinforcement learning algorithms and acceleration of learning","year":2003,"lang":"en","type":"article","venue":"Physical review. E, Statistical physics, plasmas, fluids, and related interdisciplinary topics","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":30,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Acceleration; Reinforcement learning; Convergence (economics); Rate of convergence; Computer science; Conjecture; Algorithm; Relation (database); Popularity; Reinforcement; Mathematical optimization; Mathematics; Applied mathematics; Artificial intelligence; Physics; Discrete mathematics; Quantum mechanics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003220737,0.001382142,0.001409572,0.001006778,0.0005829519,0.00123836,0.001449022,0.001741129,0.004074197],"category_scores_gemma":[0.01752316,0.0005652005,0.0006980032,0.0008012342,0.002098398,0.001731659,0.001859028,0.002646718,0.0008208487],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001292231,"about_ca_system_score_gemma":0.001330301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003371981,"about_ca_topic_score_gemma":0.00166183,"domain_scores_codex":[0.998767,0.0005139566,0.00006672971,0.0002064023,0.000300362,0.0001456351],"domain_scores_gemma":[0.9935752,0.004744235,0.0004723921,0.0003548769,0.0006819444,0.0001714213],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001159439,0.00008286572,0.0008880911,0.0001954369,0.00006717682,0.00004206786,0.00009887942,0.8414043,0.001000485,0.08912071,0.001374742,0.06560934],"study_design_scores_gemma":[0.00002452707,0.00005333015,0.00009729103,0.00002727644,0.000007276404,0.00001617682,0.000007562822,0.9597124,0.0003502815,0.03886237,0.0008339635,0.000007449136],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01373433,0.001520275,0.9766406,0.0004946393,0.0001195031,0.00008456391,0.00002621122,0.0003846652,0.006995318],"genre_scores_gemma":[0.7347751,0.002629044,0.2522425,0.0003804588,0.0003244339,0.0006739852,0.0001284778,0.0002333888,0.008612641],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004074197,"threshold_uncertainty_score":0.0170331,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01570648697182246,"score_gpt":0.3051240425009342,"score_spread":0.2894175555291117,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}