{"id":"W2754517384","doi":"10.1609/aaai.v32i1.11694","title":"Deep Reinforcement Learning That Matters","year":2018,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1504,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Open Philanthropy Project; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Standardization; Benchmark (surveying); Artificial intelligence; Field (mathematics); Variance (accounting); Deep learning; Machine learning; Data science; Risk analysis (engineering); Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01686241,0.0006897174,0.001237684,0.0004689479,0.0009112101,0.0047461,0.001606064,0.002484272,0.008552005],"category_scores_gemma":[0.09086421,0.0003431441,0.0003862609,0.0006189758,0.004028229,0.008155254,0.001766863,0.00595044,0.002446368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002013156,"about_ca_system_score_gemma":0.002038659,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002344082,"about_ca_topic_score_gemma":0.001879315,"domain_scores_codex":[0.9907906,0.004212761,0.0004522181,0.002325859,0.00190102,0.0003175655],"domain_scores_gemma":[0.9601553,0.02693311,0.002241239,0.00459459,0.004792576,0.001283257],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006528513,0.0003101072,0.01374803,0.001308145,0.000562492,0.0001698814,0.000865069,0.02696334,0.003734228,0.4226038,0.103449,0.425633],"study_design_scores_gemma":[0.0001455595,0.0004001205,0.008300369,0.001195333,0.0001573027,0.0002673063,0.0007166684,0.06270193,0.00626424,0.7822539,0.1374741,0.0001232102],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09335494,0.05890464,0.3875997,0.3470389,0.01483829,0.0001843496,0.002536151,0.002732888,0.09281014],"genre_scores_gemma":[0.8573246,0.0157044,0.06008426,0.03940038,0.004706854,0.0002564235,0.001061658,0.001838903,0.01962258],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01686241,"threshold_uncertainty_score":0.08917803,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07088540185984533,"score_gpt":0.288866159836522,"score_spread":0.2179807579766767,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}