{"id":"W2947183726","doi":"10.1287/moor.2021.1177","title":"On Linear Programming for Constrained and Unconstrained Average-Cost Markov Decision Processes with Countable Action Spaces and Strictly Unbounded Costs","year":2021,"lang":"en","type":"article","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Alberta Innovates; Alberta Innovates - Technology Futures; DeepMind; Alberta Machine Intelligence Institute","keywords":"Mathematics; Countable set; Markov decision process; Duality (order theory); Mathematical optimization; State space; Action (physics); Markov kernel; Markov chain; Discrete mathematics; Applied mathematics; Markov process; Markov model; Variable-order Markov model; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004399921,0.002037339,0.001967076,0.001182413,0.0008134798,0.002815936,0.002007467,0.002406248,0.00479155],"category_scores_gemma":[0.01350131,0.0008966663,0.001622415,0.001345061,0.003670384,0.003977745,0.003339821,0.003637986,0.0003510677],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00369554,"about_ca_system_score_gemma":0.00326492,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004662331,"about_ca_topic_score_gemma":0.003617778,"domain_scores_codex":[0.9974255,0.001356858,0.000101984,0.0004823503,0.0003457097,0.0002874258],"domain_scores_gemma":[0.986129,0.01197313,0.0007867591,0.0002109053,0.0004650464,0.0004351402],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00006662939,0.0001005398,0.0004293277,0.0002312351,0.0000622504,0.0001741293,0.0001560272,0.4530931,0.0006297394,0.5362975,0.0008516857,0.007907815],"study_design_scores_gemma":[0.00001656956,0.00004191765,0.0001230726,0.00003543213,0.00001224138,0.00002225956,0.00003082995,0.7698802,0.0001366438,0.2291522,0.0005339796,0.00001468399],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02982806,0.001190415,0.9562587,0.001497091,0.00005885374,0.00007091321,0.0001700965,0.00009345968,0.01083247],"genre_scores_gemma":[0.7909776,0.002291509,0.1892852,0.0007670325,0.0003853709,0.0009511905,0.0005161091,0.0002197631,0.01460623],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00479155,"threshold_uncertainty_score":0.02681309,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07719914338480655,"score_gpt":0.3803864094396801,"score_spread":0.3031872660548735,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}