{"id":"W2950048158","doi":"10.48550/arxiv.1404.3328","title":"Myopic Bounds for Optimal Policy of POMDPs: An extension of Lovejoy's structural results","year":2014,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada; Canada Research Chairs","keywords":"Extension (predicate logic); Bounded function; Markov decision process; Mathematical optimization; Relaxation (psychology); Mathematical economics; Upper and lower bounds; Mathematics; Computer science; Markov process; Economics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003848972,0.0003015669,0.0005037995,0.0004538435,0.0000980647,0.00006512203,0.002098022,0.0003029961,0.000003737694],"category_scores_gemma":[0.000198823,0.0003324581,0.0002475173,0.000409773,0.0001691199,0.0003824671,0.001521461,0.000328855,0.000004096266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001622556,"about_ca_system_score_gemma":0.0003770095,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002344338,"about_ca_topic_score_gemma":0.000005747887,"domain_scores_codex":[0.9979711,0.0001289498,0.0004710662,0.000894443,0.0001691405,0.0003652741],"domain_scores_gemma":[0.9964243,0.0001763455,0.0009261806,0.001847558,0.0004821627,0.0001435026],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001005688,0.0000242904,0.0002859398,0.0001656329,0.000057585,0.000006358854,0.0003877022,0.9055719,0.000148039,0.09298968,0.00005171393,0.0002105409],"study_design_scores_gemma":[0.001033972,0.0005238313,0.002739372,0.0001130883,0.00005699545,0.000002355851,0.00003757379,0.9890705,0.0006031425,0.005314229,0.0001982923,0.0003066742],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3359544,0.000007572669,0.6627658,0.00004351963,0.000323701,0.0002746948,0.00001756135,0.00008521347,0.0005275435],"genre_scores_gemma":[0.9667715,0.00002452373,0.0322866,0.00003440529,0.000117378,3.540682e-7,0.00004458694,0.00001912618,0.0007015005],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6308172,"threshold_uncertainty_score":0.9999127,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07299322227582464,"score_gpt":0.2316462662448447,"score_spread":0.15865304396902,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}