{"id":"W2995372087","doi":"","title":"Learning the Arrow of Time for Problems in Reinforcement Learning","year":2020,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Arrow; Arrow of time; Computer science; Reinforcement learning; Markov decision process; Reachability; Artificial intelligence; Machine learning; Class (philosophy); Function (biology); Markov process; Selection (genetic algorithm); Process (computing); Theoretical computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002553545,0.0008242182,0.0008684381,0.0006171106,0.0006274078,0.001780859,0.001101288,0.001621792,0.003550589],"category_scores_gemma":[0.01387671,0.0004120594,0.0008505912,0.0006235651,0.002704767,0.005086039,0.001807498,0.00458183,0.0003341021],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001867161,"about_ca_system_score_gemma":0.001176526,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002410463,"about_ca_topic_score_gemma":0.002255719,"domain_scores_codex":[0.9990543,0.0004534036,0.00005860139,0.0002395171,0.0001253738,0.00006882441],"domain_scores_gemma":[0.9949659,0.003940517,0.000415997,0.0002753604,0.0001965755,0.0002058043],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001799698,0.00009638528,0.001496673,0.0002318705,0.00006269703,0.00007176805,0.0002536391,0.4776212,0.001802988,0.461256,0.00142641,0.05550043],"study_design_scores_gemma":[0.00001481261,0.00003956532,0.0001389681,0.00002447399,0.000006924173,0.00001245537,0.00002338092,0.6537325,0.0003882343,0.3446155,0.0009920171,0.00001122537],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0172671,0.0005251392,0.9798452,0.0008020037,0.0000380623,0.00002315368,0.00004201003,0.0001154161,0.001341876],"genre_scores_gemma":[0.6324654,0.001295026,0.3617703,0.000275282,0.0001841842,0.0002069498,0.0001983078,0.0001239662,0.003480573],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003550589,"threshold_uncertainty_score":0.01354724,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05955057235187068,"score_gpt":0.3172682931974696,"score_spread":0.2577177208455989,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}