{"id":"W4380684831","doi":"10.1109/tmlcn.2023.3285543","title":"Reinforcement Learning With Non-Cumulative Objective","year":2023,"lang":"en","type":"article","venue":"IEEE Transactions on Machine Learning in Communications and Networking","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Reinforcement learning; Computer science; Markov decision process; Bottleneck; Function (biology); Mathematical optimization; Process (computing); Artificial intelligence; Bellman equation; Convergence (economics); Markov process; Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003103451,0.001222154,0.001269731,0.000380003,0.0003594151,0.001228027,0.001318745,0.001187026,0.001724313],"category_scores_gemma":[0.008401847,0.0003503065,0.0004957058,0.0004889398,0.001752023,0.002225818,0.001415673,0.002220168,0.0002980849],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001579196,"about_ca_system_score_gemma":0.001838698,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003224723,"about_ca_topic_score_gemma":0.002927766,"domain_scores_codex":[0.998513,0.0006259601,0.00007730013,0.0003248052,0.0002843297,0.000174518],"domain_scores_gemma":[0.9958239,0.002874179,0.0003598041,0.0003209042,0.0004369952,0.000184236],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001037814,0.00008447616,0.0007205085,0.0000973415,0.00005286479,0.00006081283,0.00008408064,0.8155006,0.00121149,0.1383811,0.001292673,0.04241036],"study_design_scores_gemma":[0.00001679249,0.0000354258,0.00006017794,0.000006697523,0.000006844518,0.000008186427,0.000004423986,0.9711585,0.000365818,0.02798343,0.0003468266,0.00000679496],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01558343,0.0002067775,0.9815561,0.0002857944,0.00004511358,0.00004774833,0.0000249508,0.0001758763,0.00207419],"genre_scores_gemma":[0.8054505,0.0003720946,0.1876036,0.0003522816,0.00009738117,0.0002141837,0.00009866326,0.00008580861,0.005725511],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003224723,"threshold_uncertainty_score":0.01641279,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02896642669377206,"score_gpt":0.2792376464679182,"score_spread":0.2502712197741461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}