{"id":"W2785529341","doi":"10.1609/aaai.v32i1.12115","title":"Learning Robust Options","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Robustness (evolution); Reinforcement learning; Computer science; Artificial intelligence; Robust control; Mathematical optimization; Machine learning; Convergence (economics); Artificial neural network; Mathematics; Nonlinear system","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002155049,0.001021764,0.00115058,0.0005872558,0.0004022727,0.001158304,0.001537024,0.001341764,0.003204647],"category_scores_gemma":[0.01056295,0.0006750489,0.0008078803,0.0003634828,0.001789597,0.002664124,0.002097994,0.002191251,0.0005565417],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001173608,"about_ca_system_score_gemma":0.001460946,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001651561,"about_ca_topic_score_gemma":0.001916323,"domain_scores_codex":[0.9988667,0.0003555329,0.00005841703,0.0003290536,0.0002691805,0.0001210636],"domain_scores_gemma":[0.9967515,0.001892319,0.0004409532,0.0004249909,0.0003095412,0.000180664],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000109401,0.00004515125,0.001032171,0.00007340592,0.00006201628,0.00009414196,0.00008016323,0.9063559,0.002133737,0.04552291,0.0009557165,0.0435354],"study_design_scores_gemma":[0.000007442959,0.00002478548,0.00005211638,0.000008487976,0.000004177261,0.00001273915,0.000005834067,0.9729764,0.0005946856,0.02600799,0.0002998496,0.000005415803],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02569969,0.0001367177,0.9707161,0.0002160609,0.00002895418,0.00004255531,0.00007905604,0.0006302125,0.002450798],"genre_scores_gemma":[0.8287028,0.0001440643,0.1674075,0.0002280099,0.00003041503,0.0001705861,0.0002404042,0.0002324655,0.002843745],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003204647,"threshold_uncertainty_score":0.01139712,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09360344398562903,"score_gpt":0.2975536324951889,"score_spread":0.2039501885095599,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}