{"id":"W4301342375","doi":"10.48550/arxiv.1503.02244","title":"Asymptotic Optimality of Finite Approximations to Markov Decision\\n Processes with Borel Spaces","year":2015,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Markov decision process; Mathematics; State space; Convergence (economics); Applied mathematics; Action (physics); Mathematical optimization; Markov chain; Space (punctuation); Class (philosophy); Markov process; Q-learning; Finite state; Rate of convergence; Average cost; Reinforcement learning; Computer science; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007662143,0.001172942,0.001856931,0.001528057,0.000820918,0.002643143,0.002186453,0.001751259,0.001974996],"category_scores_gemma":[0.0519775,0.001006457,0.001309258,0.0009417963,0.004099776,0.003659005,0.002574377,0.004076618,0.0003599579],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004216149,"about_ca_system_score_gemma":0.002775344,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005674932,"about_ca_topic_score_gemma":0.003637457,"domain_scores_codex":[0.9963369,0.001825737,0.0001676214,0.0005395353,0.0007891085,0.0003410516],"domain_scores_gemma":[0.9627417,0.03318574,0.001298236,0.00128697,0.001001596,0.000485771],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001101702,0.00005585543,0.0007659945,0.00009592878,0.00005043015,0.00005213762,0.0001039913,0.8343242,0.0004631123,0.1515998,0.0004307493,0.01194769],"study_design_scores_gemma":[0.000004758792,0.00001330356,0.0000497147,0.00001413042,0.000002777836,0.000005637484,0.000007400297,0.9551384,0.0001639236,0.04448669,0.0001093086,0.00000390315],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0456089,0.000737763,0.9493845,0.0005640002,0.00005755287,0.00004756762,0.00008021224,0.0002392139,0.003280291],"genre_scores_gemma":[0.8268139,0.001106192,0.1678881,0.0002200354,0.0001181729,0.0002927108,0.0003934919,0.0001587288,0.003008533],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007662143,"threshold_uncertainty_score":0.0405218,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05481596787929066,"score_gpt":0.2097878608282766,"score_spread":0.1549718929489859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}