{"id":"W2905224739","doi":"10.1609/aaai.v33i01.33014504","title":"A Comparative Analysis of Expected and Distributional Reinforcement Learning","year":2019,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":42,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Convergence (economics); Computer science; Mathematical optimization; Linear approximation; Mathematics; Econometrics; Artificial intelligence; Nonlinear system; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01444753,0.0006574413,0.00126623,0.001056437,0.0005262428,0.001883445,0.002281277,0.001520079,0.00347391],"category_scores_gemma":[0.08589091,0.0003914116,0.0007355409,0.0007650662,0.002676139,0.005074362,0.00253669,0.002335483,0.0003462694],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001910309,"about_ca_system_score_gemma":0.001491506,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001361981,"about_ca_topic_score_gemma":0.001365129,"domain_scores_codex":[0.9894434,0.006190744,0.000370361,0.0009737304,0.002567978,0.0004537298],"domain_scores_gemma":[0.9177887,0.07003044,0.00295087,0.004334801,0.003864434,0.001030785],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005763393,0.0003036895,0.005526537,0.000426674,0.0002085162,0.00009424277,0.0002766651,0.6504765,0.001327032,0.24082,0.001562163,0.09840161],"study_design_scores_gemma":[0.00002972126,0.0002361844,0.0007999751,0.00003990556,0.0000153067,0.00004057613,0.0000447237,0.9378833,0.0004967547,0.05980064,0.0005973191,0.00001566181],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09737974,0.002339596,0.887587,0.001648492,0.00008660953,0.00008967878,0.00007684207,0.0007308122,0.01006136],"genre_scores_gemma":[0.9278194,0.0005806717,0.06919786,0.0002579053,0.00007588551,0.0001163322,0.0001485068,0.0001428087,0.001660578],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01444753,"threshold_uncertainty_score":0.07640672,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05897849748042289,"score_gpt":0.3011855220912152,"score_spread":0.2422070246107923,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}