{"id":"W3169292790","doi":"10.48550/arxiv.2106.06854","title":"A Deep Reinforcement Learning Approach to Marginalized Importance\\n Sampling with the Successor Representation","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Successor cardinal; Reinforcement learning; Representation (politics); Computer science; Sampling (signal processing); Artificial intelligence; Bridge (graph theory); Variety (cybernetics); Machine learning; Simple random sample; Reinforcement; State (computer science); Mathematics; Algorithm; Sociology; Psychology; Political science; Social psychology; Detector; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","scholarly_communication","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001360222,0.001048268,0.001014971,0.0005006532,0.001551533,0.00176278,0.004147687,0.0004134691,0.0002125413],"category_scores_gemma":[0.0002659425,0.0009667496,0.0004835717,0.003663776,0.0003601373,0.001271758,0.00456692,0.002326558,0.00009515963],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000935884,"about_ca_system_score_gemma":0.0006831262,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003955124,"about_ca_topic_score_gemma":0.00006806635,"domain_scores_codex":[0.9927858,0.0008714744,0.0009028434,0.003262043,0.0008140013,0.001363903],"domain_scores_gemma":[0.9929694,0.0004736109,0.001636135,0.003368606,0.001008673,0.000543555],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003110582,0.00008440832,0.008741038,0.0002033785,0.0005106336,0.0002151048,0.005558733,0.9661682,0.00003805595,0.01783633,0.00004327904,0.0002898121],"study_design_scores_gemma":[0.001462343,0.0003139583,0.001930293,0.000326026,0.0003972617,0.00003565455,0.005363574,0.9877515,0.00006839941,0.00009661892,0.001071871,0.001182482],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05412823,0.00007886752,0.9315495,0.0003772594,0.0004462854,0.001916959,5.198721e-7,0.0002467262,0.01125567],"genre_scores_gemma":[0.9665712,0.0003374072,0.02197758,0.0003808456,0.0001593764,0.00002206658,0.0001339553,0.00008395544,0.01033358],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.912443,"threshold_uncertainty_score":0.9999751,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09879524359877373,"score_gpt":0.2253305824909199,"score_spread":0.1265353388921462,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}