{"id":"W4318620677","doi":"10.48550/arxiv.2301.11490","title":"Neural Episodic Control with State Abstraction","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"JST-Mirai Program; Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canadian Institute for Advanced Research","keywords":"Abstraction; Computer science; Leverage (statistics); Episodic memory; Reinforcement learning; Artificial intelligence; Sample (material); State (computer science); Inefficiency; Control (management); Machine learning; Scalability; Programming language; Psychology; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003284183,0.00029917,0.0002794604,0.0003390154,0.0001504635,0.0002032294,0.00134474,0.0001764234,0.00001193101],"category_scores_gemma":[0.00002876849,0.0003145824,0.0001221545,0.0005314637,0.0001032954,0.0004493061,0.0007450345,0.0008760485,0.0003256976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002065315,"about_ca_system_score_gemma":0.0001413639,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002010811,"about_ca_topic_score_gemma":0.00004855948,"domain_scores_codex":[0.9981953,0.00008921212,0.0002174467,0.0008732615,0.0001777067,0.0004470801],"domain_scores_gemma":[0.9980365,0.0001965092,0.0003960058,0.001045882,0.0001675726,0.0001575032],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004322843,0.0000135645,0.00429111,0.00003917121,0.00008807249,0.0005706627,0.00008375565,0.9890248,0.000008603662,0.00545831,0.000103621,0.0002751509],"study_design_scores_gemma":[0.0006574069,0.0001101787,0.009960201,0.00005364519,0.00004864408,0.000006052297,0.00002622329,0.9867234,0.00002705731,0.001845007,0.0001862735,0.0003559169],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07359044,0.00000560528,0.9235982,0.0001338593,0.0008125949,0.0003073686,0.000004620424,0.0006741811,0.0008731251],"genre_scores_gemma":[0.9923491,0.00003648604,0.00106447,0.00009548456,0.00005283294,7.738053e-7,0.00001345352,0.0000290021,0.006358364],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9225338,"threshold_uncertainty_score":0.9999306,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07064646825082561,"score_gpt":0.1866594901952144,"score_spread":0.1160130219443888,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}