{"id":"W3181012797","doi":"10.1609/aaai.v36i6.20660","title":"Learning Expected Emphatic Traces for Deep RL","year":2022,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Weighting; Scalability; Convergence (economics); Stability (learning theory); Artificial intelligence; Machine learning; Sampling (signal processing); Sample (material); Artificial neural network; Baseline (sea); Key (lock); Variance (accounting); Algorithm; Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001997605,0.001037158,0.0009216742,0.0006188543,0.0003845272,0.001056403,0.001756619,0.0009589416,0.00251977],"category_scores_gemma":[0.01416253,0.0007893209,0.0004450436,0.0004399752,0.001204317,0.002203086,0.001921928,0.002326892,0.0004432451],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001159165,"about_ca_system_score_gemma":0.001451528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003365702,"about_ca_topic_score_gemma":0.004749532,"domain_scores_codex":[0.9994128,0.0001940903,0.00004656831,0.0001303737,0.0001449185,0.00007126926],"domain_scores_gemma":[0.9959472,0.002669186,0.0003020169,0.0004109671,0.0004664506,0.000204161],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001651554,0.00006257475,0.0009740539,0.00006988085,0.00003311984,0.00006179462,0.000100757,0.9214735,0.001754316,0.01815574,0.0007895121,0.0563597],"study_design_scores_gemma":[0.000005146296,0.00001098384,0.00002775124,0.000003075471,0.000001426263,0.000003605238,0.000003230649,0.9934834,0.0002996942,0.006070542,0.00008890963,0.000002268391],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01742267,0.00008106986,0.9807557,0.0001552704,0.00002313646,0.00003076936,0.00004562111,0.0006974584,0.0007882821],"genre_scores_gemma":[0.8052354,0.0001270857,0.1902327,0.00018003,0.00004110677,0.0002207219,0.0002492535,0.0002416313,0.003472049],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003365702,"threshold_uncertainty_score":0.01056445,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0655002341992619,"score_gpt":0.2902482295044936,"score_spread":0.2247479953052317,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}