{"id":"W4308071056","doi":"10.48550/arxiv.2110.01954","title":"Continuous-Time Fitted Value Iteration for Robust Policies","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"European Commission; Canadian Institute for Advanced Research; Nvidia","keywords":"Bellman equation; Discretization; Reinforcement learning; Leverage (statistics); Mathematical optimization; Optimal control; Computer science; Dynamic programming; Robustness (evolution); Mathematics; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002439415,0.001301559,0.001393668,0.0006288458,0.0004176453,0.001374519,0.001062575,0.001575791,0.003757239],"category_scores_gemma":[0.010888,0.0007276162,0.0009042826,0.0005099897,0.00198006,0.001273039,0.001670445,0.002836786,0.0007083305],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001668913,"about_ca_system_score_gemma":0.002119203,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00364777,"about_ca_topic_score_gemma":0.002427807,"domain_scores_codex":[0.9989122,0.0004321945,0.00006144542,0.0002118354,0.000242509,0.000139792],"domain_scores_gemma":[0.9957391,0.003210999,0.0002909991,0.0002098532,0.000401655,0.0001472516],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007382223,0.00003085338,0.0003694459,0.00006261795,0.00002935331,0.00004934797,0.0000692503,0.9463266,0.0007342311,0.03226065,0.0006266992,0.01936714],"study_design_scores_gemma":[0.000006042562,0.00001274092,0.00001592439,0.000006935348,0.000001773221,0.000005844066,0.000003718031,0.9920101,0.000219211,0.007486924,0.0002275147,0.000003214848],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.007931715,0.0002661989,0.9891431,0.0001733998,0.00003683805,0.00003592873,0.00002277327,0.0002591991,0.00213091],"genre_scores_gemma":[0.7042421,0.0003548255,0.2885604,0.0002477835,0.00005607487,0.0003296977,0.000184301,0.0003207769,0.005704075],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003757239,"threshold_uncertainty_score":0.01290101,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05323358205397503,"score_gpt":0.2011440867758975,"score_spread":0.1479105047219225,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}