{"id":"W2951553872","doi":"","title":"SOLAR: Deep Structured Representations for Model-Based Reinforcement Learning","year":2018,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Robotics; Range (aeronautics); Simple (philosophy); Linear-quadratic regulator; Offline learning; Quadratic equation; Image (mathematics); Machine learning; Control (management); Robot; Online learning; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008328533,0.0009263203,0.000787622,0.0003985454,0.0002747382,0.0009391803,0.001506701,0.001198087,0.005506192],"category_scores_gemma":[0.003890238,0.0005341017,0.0006359236,0.0004592341,0.0006919852,0.001498821,0.001448567,0.002492253,0.001193912],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008322253,"about_ca_system_score_gemma":0.0009986935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003021113,"about_ca_topic_score_gemma":0.003948487,"domain_scores_codex":[0.9996849,0.0001003373,0.00001696584,0.00006632185,0.0001015004,0.00003001002],"domain_scores_gemma":[0.9992263,0.0003868874,0.00007645486,0.0001515374,0.0001112723,0.00004745484],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008492717,0.00007317487,0.0003192085,0.00009561157,0.00004671598,0.00005239472,0.00004095331,0.856506,0.002224689,0.04022343,0.006793636,0.09353925],"study_design_scores_gemma":[0.000006145215,0.000009616067,0.00001487536,0.000003971886,0.000001928383,0.000003882083,0.000001349758,0.9880539,0.000300126,0.01102312,0.0005791,0.00000199476],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003530218,0.0001443093,0.9926267,0.0001740383,0.00005765904,0.00002935144,0.0001523687,0.001651436,0.001633822],"genre_scores_gemma":[0.5543815,0.0004309752,0.4352029,0.0004287085,0.0001118198,0.0004907981,0.001281187,0.0006783358,0.00699383],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005506192,"threshold_uncertainty_score":0.01842004,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0705683000237207,"score_gpt":0.2240496447580726,"score_spread":0.1534813447343519,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}