{"id":"W3172765438","doi":"10.48550/arxiv.2012.13658","title":"Locally Persistent Exploration in Continuous Control Tasks with Sparse\\n Rewards","year":2020,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"","keywords":"Reinforcement learning; Computer science; State space; Trajectory; State (computer science); Action (physics); Task (project management); Space (punctuation); Artificial intelligence; Algorithm; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009648275,0.0005484416,0.0007121487,0.0003638972,0.0004140895,0.0009489657,0.0009271897,0.0007967884,0.00161247],"category_scores_gemma":[0.006220373,0.0004709108,0.0003508672,0.0003431377,0.001284783,0.001800318,0.001624212,0.001260688,0.0001347731],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006943386,"about_ca_system_score_gemma":0.0008173784,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003819765,"about_ca_topic_score_gemma":0.003643751,"domain_scores_codex":[0.9996459,0.0001277612,0.00001972599,0.00008795733,0.00006919222,0.00004935145],"domain_scores_gemma":[0.9967991,0.00236489,0.0002992649,0.00023552,0.00009424442,0.000207047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002337882,0.00008352513,0.001478009,0.00008998748,0.00002506086,0.0001424522,0.0002257778,0.9461716,0.003624499,0.01655747,0.0003729134,0.03099492],"study_design_scores_gemma":[0.00001352082,0.00003232459,0.0002039784,0.000004021446,0.000002127115,0.00001039796,0.00001430196,0.990049,0.0003561008,0.009196129,0.0001140807,0.000004057024],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2469739,0.000367106,0.7498816,0.0003058221,0.00002541731,0.00005075331,0.00005451161,0.000541333,0.00179959],"genre_scores_gemma":[0.9371868,0.0001286087,0.0610227,0.0000420552,0.00001766415,0.0001064832,0.00005700436,0.00004981121,0.001388832],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003819765,"threshold_uncertainty_score":0.007595062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09598491008680228,"score_gpt":0.207414975584561,"score_spread":0.1114300654977588,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}