{"id":"W2966537673","doi":"10.24963/ijcai.2019/840","title":"LTL and Beyond: Formal Languages for Reward Function Specification in Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":152,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Social Innovation; Vector Institute; University of Toronto","funders":"Comisión Nacional de Investigación Científica y Tecnológica; Natural Sciences and Engineering Research Council of Canada; Microsoft Research","keywords":"Reinforcement learning; Computer science; Function (biology); Artificial intelligence; Automaton; Representation (politics); Temporal difference learning; Reinforcement; Machine learning; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005083673,0.001395117,0.0007623931,0.0008307347,0.0006858223,0.003375202,0.002041877,0.001979582,0.007150466],"category_scores_gemma":[0.02006145,0.001006435,0.001986192,0.0009767839,0.004660803,0.005352544,0.002210238,0.006037684,0.002186079],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001835213,"about_ca_system_score_gemma":0.003086314,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003616436,"about_ca_topic_score_gemma":0.004541392,"domain_scores_codex":[0.9954404,0.002414245,0.0004885991,0.0005338441,0.0008618232,0.0002611509],"domain_scores_gemma":[0.9892531,0.007684153,0.0006675111,0.00149458,0.0007055899,0.0001950978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001039257,0.00006737574,0.0004680937,0.0002367265,0.00002459195,0.0001611037,0.0004924806,0.0811724,0.002164507,0.870478,0.003584672,0.04104614],"study_design_scores_gemma":[0.0000570925,0.00004368497,0.00006422536,0.0001389684,0.00002090474,0.00008083292,0.00006585196,0.3428856,0.002662052,0.6382343,0.0157141,0.00003250549],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0009239567,0.00011456,0.996861,0.0003354216,0.00003354926,0.00003943208,0.0001137088,0.0006158713,0.0009625165],"genre_scores_gemma":[0.1348026,0.0005784435,0.8584221,0.0009340654,0.0001318681,0.000854929,0.0006257404,0.0008868663,0.002763517],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007150466,"threshold_uncertainty_score":0.02688533,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01755740490172809,"score_gpt":0.2801044758634209,"score_spread":0.2625470709616928,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}