{"id":"W2997289589","doi":"10.1609/aaai.v34i04.5955","title":"Count-Based Exploration with the Successor Representation","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Alberta Innovates; University of Alberta; Alberta Machine Intelligence Institute; Compute Canada","keywords":"Successor cardinal; Representation (politics); Norm (philosophy); Computer science; Sample complexity; Generalization; Reinforcement learning; Similarity (geometry); Artificial intelligence; State (computer science); Algorithm; Theoretical computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001682806,0.0007093452,0.001201087,0.0007489717,0.0004694233,0.001207078,0.002174494,0.001219488,0.004734872],"category_scores_gemma":[0.008043453,0.0003787967,0.0006237776,0.000742825,0.001403205,0.003673082,0.002593866,0.00191675,0.0006386793],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001004792,"about_ca_system_score_gemma":0.001669419,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001206231,"about_ca_topic_score_gemma":0.001733762,"domain_scores_codex":[0.99907,0.0003261522,0.00006086248,0.000208141,0.0002375726,0.00009725537],"domain_scores_gemma":[0.9969981,0.001787649,0.000285678,0.0004745996,0.0002592752,0.0001946769],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003936095,0.0001718341,0.001918441,0.0001995781,0.00006370949,0.0001232637,0.0002134152,0.5496461,0.003868591,0.2249373,0.003327467,0.2151367],"study_design_scores_gemma":[0.00002407543,0.00006771441,0.00007536897,0.0000127907,0.000007218848,0.00002768724,0.000008651916,0.9410114,0.0007906344,0.05722805,0.000736565,0.000009859263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02740588,0.0002049848,0.9676494,0.0003616572,0.00005335372,0.00006878617,0.0000974993,0.0008739844,0.003284447],"genre_scores_gemma":[0.6898327,0.0001538185,0.3038738,0.0002174671,0.00005781549,0.0003127986,0.0002552164,0.0001898054,0.005106587],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004734872,"threshold_uncertainty_score":0.01583976,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05162394606866699,"score_gpt":0.263603992355036,"score_spread":0.211980046286369,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}