{"id":"W2970483889","doi":"10.48550/arxiv.1911.04448","title":"Real-Time Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Open Philanthropy Project","keywords":"Reinforcement learning; Markov decision process; Computer science; Action selection; Computation; Artificial intelligence; State (computer science); Action (physics); Markov process; Mathematical optimization; Machine learning; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001169656,0.0007917386,0.000713021,0.0002919217,0.0003151613,0.001246099,0.001228696,0.0009188627,0.006233101],"category_scores_gemma":[0.004235407,0.0002635176,0.0004496548,0.0002856157,0.000775684,0.001081258,0.0009153464,0.00140573,0.001185749],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000953122,"about_ca_system_score_gemma":0.001260358,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003052456,"about_ca_topic_score_gemma":0.003227834,"domain_scores_codex":[0.999124,0.0002821317,0.00004633056,0.0002189199,0.0002460315,0.00008250719],"domain_scores_gemma":[0.9986499,0.0006982684,0.0001295069,0.0001530338,0.0002851034,0.00008422973],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002625767,0.0001500806,0.00113917,0.0002947661,0.00009159948,0.0001995469,0.0001385596,0.7322657,0.005956901,0.0690992,0.005552731,0.1848492],"study_design_scores_gemma":[0.00002697593,0.00005654654,0.0001461117,0.00001460157,0.00001143518,0.00004236267,0.00001376678,0.9757965,0.001369457,0.01704372,0.005468039,0.00001046764],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01618581,0.001246168,0.9666417,0.0005851805,0.000253014,0.00008170822,0.0000778394,0.001258711,0.01366998],"genre_scores_gemma":[0.7911433,0.001145819,0.1847803,0.0003527589,0.000136248,0.0002140159,0.0002317785,0.0001735156,0.02182224],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006233101,"threshold_uncertainty_score":0.02085179,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.008090322237890085,"score_gpt":0.2214901116315436,"score_spread":0.2133997893936535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}