{"id":"W2136937392","doi":"10.1287/moor.1050.0148","title":"On the Empirical State-Action Frequencies in Markov Decision Processes Under General Policies","year":2005,"lang":"en","type":"article","venue":"Mathematics of Operations Research","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"National Science Foundation","keywords":"Polytope; Mathematics; Markov decision process; Markov chain; State (computer science); Element (criminal law); Action (physics); Limit (mathematics); Empirical research; Finite state; Markov process; Mathematical optimization; Mathematical economics; Combinatorics; Statistics; Algorithm; Mathematical analysis; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006519512,0.0008115121,0.001385516,0.001574354,0.001019292,0.002532159,0.001687082,0.001435592,0.003627048],"category_scores_gemma":[0.03610321,0.0008273764,0.0007925252,0.001188459,0.004793552,0.005824296,0.002714182,0.002637249,0.0003513564],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002483077,"about_ca_system_score_gemma":0.001310067,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001870346,"about_ca_topic_score_gemma":0.0009423223,"domain_scores_codex":[0.9962279,0.001602604,0.0001920541,0.0006974065,0.0008329007,0.0004472666],"domain_scores_gemma":[0.9466075,0.04479291,0.003884357,0.001877104,0.001604581,0.001233593],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001635847,0.00007220241,0.001572581,0.0001117577,0.00004387851,0.0001289329,0.0002769449,0.3607748,0.002026488,0.6243321,0.0003400574,0.01015672],"study_design_scores_gemma":[0.00002902365,0.00008151016,0.0008360812,0.00004627588,0.00001346819,0.00005799465,0.00007294708,0.6539472,0.0006073636,0.3437505,0.0005225363,0.00003501765],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2336994,0.0006927629,0.7578002,0.0008126369,0.0000465773,0.00007434916,0.000192752,0.0001825535,0.006498866],"genre_scores_gemma":[0.9342227,0.0008678451,0.06139459,0.0001173793,0.0001250734,0.000269089,0.0002017432,0.0001105585,0.002691039],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006519512,"threshold_uncertainty_score":0.0344789,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3170072528469867,"score_gpt":0.4601731656936224,"score_spread":0.1431659128466358,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}