{"id":"W4402353118","doi":"10.1109/cisce62493.2024.10653102","title":"Exploration of the Possibility for a Wider Range of the Discount Factor for Reinforcement Learning","year":2024,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Reinforcement learning; Reinforcement; Range (aeronautics); Computer science; Factor (programming language); Artificial intelligence; Machine learning; Psychology; Engineering; Social psychology; Aerospace engineering; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000453589,0.00009916408,0.0001268334,0.00003203621,0.0001326429,0.00009490874,0.00064456,0.00003772477,0.00001667932],"category_scores_gemma":[0.0003037477,0.00004985292,0.0002089473,0.0002367476,0.00006446859,0.0005448999,0.0002345979,0.0000994302,0.000001359238],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005997523,"about_ca_system_score_gemma":0.0001120658,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002227857,"about_ca_topic_score_gemma":0.000009564505,"domain_scores_codex":[0.9988776,0.0000485964,0.0003772958,0.0002046303,0.0003319107,0.0001598949],"domain_scores_gemma":[0.9987816,0.0003495599,0.0001715889,0.0005358171,0.0001437937,0.00001768342],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001758576,0.00001315952,0.001233188,0.0003793097,0.00004707401,3.300748e-8,0.005148062,0.8957238,0.001378533,0.09237392,0.0003635892,0.0033218],"study_design_scores_gemma":[0.000237919,0.0001507702,0.001514009,0.00009735997,0.00001681035,2.653226e-7,0.0001544771,0.9820498,0.010696,0.001464875,0.003538997,0.00007875197],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005961839,0.00003015994,0.9903901,0.001321901,0.0006145237,0.001173724,0.000003169336,0.00004785728,0.0004566973],"genre_scores_gemma":[0.9878234,0.000003394959,0.006594393,0.00006956423,0.00003091036,0.00007718394,0.000001339632,0.000008598074,0.005391198],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9837958,"threshold_uncertainty_score":0.2032943,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0494761660167853,"score_gpt":0.2979828257252523,"score_spread":0.248506659708467,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}