{"id":"W2889732123","doi":"10.1609/aaai.v33i01.33013582","title":"Combined Reinforcement Learning via Abstract Representations","year":2019,"lang":"en","type":"preprint","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Modularity (biology); Bridging (networking); Generalization; Artificial intelligence; Representation (politics); Encoding (memory); Machine learning; Theoretical computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001115697,0.0008902925,0.001142879,0.0004843684,0.0002693833,0.001152195,0.001521001,0.001016181,0.002411722],"category_scores_gemma":[0.004874594,0.0004328392,0.0006690721,0.0005017414,0.001299671,0.002603273,0.00244879,0.002205943,0.0004356679],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009402296,"about_ca_system_score_gemma":0.0007981521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001791335,"about_ca_topic_score_gemma":0.001945373,"domain_scores_codex":[0.999182,0.0003108845,0.00004450377,0.0001867127,0.0002017931,0.00007402126],"domain_scores_gemma":[0.9984164,0.0008165807,0.0001619119,0.0003382856,0.0001694401,0.00009740033],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001131592,0.00006647401,0.0004400682,0.00009434177,0.00006832677,0.0000690148,0.00008418022,0.8272984,0.002515244,0.08750541,0.001023141,0.08072221],"study_design_scores_gemma":[0.00001141768,0.0000291853,0.0000409477,0.000005527249,0.000006683923,0.000008489867,0.000003897196,0.9398499,0.0004114611,0.05919209,0.0004343658,0.000006006574],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.009697022,0.0001405865,0.988126,0.000188648,0.00002886287,0.00002206839,0.00004433558,0.0004085918,0.001343947],"genre_scores_gemma":[0.8262839,0.0002484146,0.1698511,0.0001580972,0.00005787252,0.0001841737,0.0001939767,0.00009997062,0.002922389],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002411722,"threshold_uncertainty_score":0.008068025,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08144392931513404,"score_gpt":0.3154658059877066,"score_spread":0.2340218766725725,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}