{"id":"W4287251902","doi":"10.48550/arxiv.2103.15793","title":"LASER: Learning a Latent Action Space for Efficient Reinforcement\\n Learning","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reinforcement learning; Action (physics); Computer science; Artificial intelligence; Space (punctuation); Task (project management); Machine learning; Engineering; Physics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","research_integrity"],"consensus_categories":["metaepi_narrow"],"category_scores_codex":[0.001587554,0.001276561,0.001192321,0.0008065649,0.002001075,0.001352499,0.002496229,0.0009403523,0.0003177998],"category_scores_gemma":[0.0006464418,0.001697606,0.00120236,0.001959615,0.0002988299,0.0009341436,0.004552827,0.003610651,0.0003067632],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001700446,"about_ca_system_score_gemma":0.000898058,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000206292,"about_ca_topic_score_gemma":0.00001904135,"domain_scores_codex":[0.9921796,0.0008715851,0.001064409,0.003415758,0.0006106955,0.001857982],"domain_scores_gemma":[0.993365,0.0007523189,0.002050384,0.001945218,0.001239953,0.0006471661],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001774789,0.0001414558,0.002090252,0.0005325471,0.0005214415,0.000261785,0.001789073,0.9826034,0.000219199,0.009843749,0.00006643976,0.00175319],"study_design_scores_gemma":[0.002346488,0.0007933825,0.0002998148,0.0007675873,0.0004904427,0.00001945926,0.001829691,0.9843642,0.001010882,0.00006912642,0.006402782,0.001606132],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1225928,0.00007534787,0.8702118,0.0001446878,0.002578402,0.001525076,0.000001811934,0.0005576061,0.002312436],"genre_scores_gemma":[0.9366442,0.001107849,0.003966744,0.00007886093,0.000271308,0.00001112969,0.0001736736,0.0001202463,0.05762597],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8662452,"threshold_uncertainty_score":0.9999986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0890968772339231,"score_gpt":0.2143847517416132,"score_spread":0.1252878745076901,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}