{"id":"W1541317966","doi":"10.1609/aaai.v24i1.7740","title":"Robust Policy Computation in Reward-Uncertain MDPs Using Nondominated Policies","year":2010,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":47,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Regret; Markov decision process; Minimax; Computer science; Mathematical optimization; Set (abstract data type); Computation; Parallels; Exploit; Mathematics; Markov process; Algorithm; Machine learning; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004740361,0.001287033,0.001847336,0.0008416537,0.0006756426,0.002160882,0.001534513,0.001471088,0.003252332],"category_scores_gemma":[0.02550971,0.001083523,0.001379734,0.0009105396,0.002081177,0.003884342,0.002948446,0.003078257,0.0003093014],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001925969,"about_ca_system_score_gemma":0.002689141,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003008072,"about_ca_topic_score_gemma":0.003260211,"domain_scores_codex":[0.998098,0.0007330634,0.0001485521,0.0004614026,0.0003641699,0.0001947201],"domain_scores_gemma":[0.9862354,0.01096285,0.0008195315,0.001208526,0.000464807,0.0003087411],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001091446,0.00003081679,0.0003628725,0.00006013036,0.00002297266,0.00004055092,0.00005473573,0.9570596,0.0005735706,0.02857684,0.0002095468,0.01289929],"study_design_scores_gemma":[0.0000156583,0.00002456796,0.00004190389,0.00001157199,0.000004081055,0.000007108903,0.000009541075,0.973992,0.0006280937,0.02512322,0.0001373205,0.00000494349],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02932183,0.0001055076,0.9688779,0.0001889052,0.00002029654,0.00005795185,0.00007805253,0.0003437759,0.001005803],"genre_scores_gemma":[0.5715406,0.000221616,0.4264326,0.0001175901,0.00002667682,0.0002756169,0.000265026,0.0001700849,0.0009501011],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004740361,"threshold_uncertainty_score":0.02506971,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1230832600318342,"score_gpt":0.3346686475142595,"score_spread":0.2115853874824253,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}