{"id":"W4389218123","doi":"10.48550/arxiv.2311.17855","title":"Maximum Entropy Model Correction in Reinforcement Learning","year":2023,"lang":"en","type":"preprint","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Government of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Principle of maximum entropy; Computer science; Convergence (economics); Bellman equation; Entropy (arrow of time); Algorithm; Mathematical optimization; Function (biology); Applied mathematics; Mathematics; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002009905,0.0007826586,0.001109684,0.0005010835,0.0005208349,0.001016834,0.001569542,0.001116217,0.001582402],"category_scores_gemma":[0.01033984,0.0005214434,0.000574732,0.0004764797,0.002037841,0.001950701,0.001441029,0.001756541,0.0002603405],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001525466,"about_ca_system_score_gemma":0.00165307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003972773,"about_ca_topic_score_gemma":0.002855808,"domain_scores_codex":[0.99874,0.000575661,0.00003229792,0.0001723723,0.0003764585,0.0001031975],"domain_scores_gemma":[0.9958577,0.002990779,0.000347722,0.0003375423,0.0003474489,0.0001188233],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003930961,0.00002102445,0.0003475347,0.00003945274,0.00002041989,0.00004720119,0.00005185534,0.9274867,0.0005727522,0.05697771,0.0003302125,0.01406585],"study_design_scores_gemma":[0.000003525278,0.00001260147,0.00002492337,0.000003599466,0.000001929565,0.000006961683,0.000002059028,0.984341,0.0002253009,0.0152192,0.0001557903,0.000003091695],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006459998,0.00009126473,0.9921814,0.0001326908,0.00001945195,0.00001384957,0.000009632314,0.0001240958,0.0009676352],"genre_scores_gemma":[0.8087522,0.0002029164,0.1875972,0.0001433172,0.00005421974,0.0001558592,0.00005072159,0.0001151103,0.002928525],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003972773,"threshold_uncertainty_score":0.01106805,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02159192054808646,"score_gpt":0.2504683555440546,"score_spread":0.2288764349959681,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}