{"id":"W2115083386","doi":"10.1007/978-3-642-04174-7_43","title":"Learning the Difference between Partially Observable Dynamical Systems","year":2009,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université Laval; McGill University","funders":"","keywords":"Observable; Computer science; Markov decision process; Divergence (linguistics); Reinforcement learning; Dynamical systems theory; Key (lock); Temporal difference learning; Markov process; Markov chain; Artificial intelligence; Process (computing); Mathematical optimization; Machine learning; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001507779,0.0004567291,0.0008455677,0.0003645122,0.0001934681,0.0008867414,0.001203622,0.0008923366,0.00284866],"category_scores_gemma":[0.00962594,0.0004553697,0.000398682,0.0002691849,0.00113091,0.00268838,0.001929607,0.002246424,0.0002730841],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005924996,"about_ca_system_score_gemma":0.0004411036,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006930659,"about_ca_topic_score_gemma":0.0007601373,"domain_scores_codex":[0.9993085,0.0002287291,0.00004133665,0.0002007184,0.0001703503,0.00005037764],"domain_scores_gemma":[0.9952094,0.003888085,0.0002211517,0.0003037338,0.0002213275,0.0001562262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004292158,0.000134629,0.002547915,0.0003154908,0.0001019816,0.00008709601,0.0001964143,0.5602937,0.005422303,0.196829,0.001640606,0.2320017],"study_design_scores_gemma":[0.00001974956,0.00007824954,0.0002427747,0.00001143509,0.000006387787,0.00001714223,0.00000879917,0.9214355,0.0006715333,0.07705445,0.0004446993,0.000009319168],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07526908,0.0004029663,0.9199266,0.0004275863,0.0001368758,0.0000326097,0.00008973909,0.0003546079,0.003359816],"genre_scores_gemma":[0.9082778,0.0002433511,0.08763633,0.0001157452,0.00007163843,0.00008164062,0.0002452045,0.00005598122,0.003272342],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00284866,"threshold_uncertainty_score":0.00952971,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02598432659031535,"score_gpt":0.244464198501162,"score_spread":0.2184798719108467,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}