{"id":"W2809151193","doi":"10.48550/arxiv.1808.09127","title":"High-confidence error estimates for learned value functions","year":2018,"lang":"en","type":"article","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Bellman equation; Benchmark (surveying); Computer science; Value (mathematics); Convergence (economics); Stability (learning theory); Function (biology); Mathematical optimization; Upper and lower bounds; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001976149,0.0001558694,0.0001492045,0.0001414871,0.0004232228,0.0001106402,0.0009696576,0.00007832472,0.00006715247],"category_scores_gemma":[0.0001634837,0.0001732005,0.00008894983,0.0005928826,0.0001933691,0.0006848249,0.000253284,0.0001198753,0.0005424361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008118354,"about_ca_system_score_gemma":0.00008407635,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006994038,"about_ca_topic_score_gemma":0.00001071107,"domain_scores_codex":[0.9988394,0.0000376366,0.0001371102,0.0005470573,0.00007843778,0.0003603488],"domain_scores_gemma":[0.9986116,0.0002365372,0.0001211433,0.0006877731,0.0002254396,0.0001175418],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001520749,0.00001353965,0.0006148407,0.000006237613,0.00002450916,0.000007755261,0.00007628676,0.5893306,0.0001581884,0.408956,0.0006206082,0.0001762382],"study_design_scores_gemma":[0.0004811403,0.0003300054,0.0007232252,0.00001966366,0.00003727078,0.000003371997,0.00005571287,0.9838074,0.0009903177,0.01129493,0.002034998,0.0002219374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02214424,0.000006249737,0.9748216,0.0002288142,0.0006589738,0.0002010139,0.000001808097,0.0003222753,0.001615041],"genre_scores_gemma":[0.9649066,0.00000423477,0.02714739,0.0001603256,0.0000984476,0.000001098657,0.000004566767,0.0000121866,0.007665152],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9476742,"threshold_uncertainty_score":0.7062913,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08861862077521193,"score_gpt":0.2193237207379867,"score_spread":0.1307050999627748,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}