{"id":"W3165994454","doi":"","title":"Reinforcement Learning as One Big Sequence Modeling Problem","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Sequence (biology); Transformer; Artificial intelligence; Markov chain; Sequence learning; Markov decision process; Machine learning; Markov process; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002299928,0.0008299897,0.001241628,0.000279694,0.000362857,0.001139674,0.00155036,0.001388531,0.004102944],"category_scores_gemma":[0.006841409,0.0005032245,0.0007637411,0.0003779288,0.001673912,0.002965019,0.001559793,0.002969851,0.0004997905],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001263689,"about_ca_system_score_gemma":0.001363849,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00288336,"about_ca_topic_score_gemma":0.00302796,"domain_scores_codex":[0.9987289,0.0005425141,0.00005319613,0.0003633618,0.0002234453,0.00008851478],"domain_scores_gemma":[0.9968112,0.002314609,0.000189489,0.0003492876,0.0001874969,0.0001478063],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001729014,0.00007609125,0.0008311134,0.0001219358,0.00005079848,0.0001137569,0.0001745689,0.8103381,0.00173465,0.1332453,0.001891736,0.051249],"study_design_scores_gemma":[0.0000148408,0.00002766828,0.0000713896,0.000005644073,0.000006675048,0.0000160804,0.00001039144,0.9383397,0.000405228,0.06049542,0.0006004963,0.000006471604],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01631244,0.0001543657,0.9795769,0.0008129469,0.00003765708,0.00004498661,0.00008825785,0.0005114219,0.002460997],"genre_scores_gemma":[0.8085867,0.0002998113,0.1840909,0.0004755641,0.00007291329,0.0002266297,0.0002743049,0.0001835881,0.005789621],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004102944,"threshold_uncertainty_score":0.0137257,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1422045581090734,"score_gpt":0.2102997366343601,"score_spread":0.06809517852528674,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}