{"id":"W2809151193","doi":"10.48550/arxiv.1808.09127","title":"High-confidence error estimates for learned value functions","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bellman equation; Benchmark (surveying); Computer science; Value (mathematics); Convergence (economics); Stability (learning theory); Function (biology); Mathematical optimization; Upper and lower bounds; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01505754,0.001986473,0.002640625,0.001745194,0.0009637493,0.003954647,0.004771421,0.00344626,0.004496049],"category_scores_gemma":[0.2000636,0.001681725,0.001270269,0.001181929,0.003969869,0.008407442,0.004742395,0.007989336,0.001396788],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003117444,"about_ca_system_score_gemma":0.002706942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003129006,"about_ca_topic_score_gemma":0.002782106,"domain_scores_codex":[0.989794,0.003511368,0.0007824513,0.002035278,0.003172221,0.0007047137],"domain_scores_gemma":[0.8307855,0.1379354,0.006805725,0.01293963,0.0101333,0.00140047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000542653,0.0002478015,0.005541068,0.0004077223,0.0001617918,0.0001644885,0.0003189248,0.8073922,0.00414513,0.07032061,0.00246937,0.1082882],"study_design_scores_gemma":[0.00002474091,0.00005594833,0.0005270262,0.00007020498,0.00000956362,0.00004595803,0.00002328591,0.9626518,0.003240956,0.03290323,0.0004257366,0.00002152948],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00970447,0.0002282708,0.9882342,0.0003151994,0.0000350771,0.00003947505,0.00005926597,0.0005390593,0.000845013],"genre_scores_gemma":[0.5151451,0.0004229403,0.4798037,0.0003372456,0.0001384061,0.0003747154,0.0006296842,0.000906935,0.002241275],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01505754,"threshold_uncertainty_score":0.07963282,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08861862077521193,"score_gpt":0.2193237207379867,"score_spread":0.1307050999627748,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}