{"id":"W4415688799","doi":"10.1007/978-3-032-04339-9_15","title":"Enhancing Off-Policy Method SAC with KAN for Continuous Reinforcement Learning","year":2025,"lang":"en","type":"book-chapter","venue":"Communications in computer and information science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Regina","funders":"","keywords":"Reinforcement learning; Embedding; Architecture; Reinforcement; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001974788,0.0003217481,0.0004182314,0.001646253,0.0008701669,0.0009549123,0.003429612,0.0001432433,0.00000386481],"category_scores_gemma":[0.0002276522,0.0003066056,0.00006701078,0.0006934633,0.0004835515,0.004485288,0.002479683,0.0006310462,0.00001465819],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003296437,"about_ca_system_score_gemma":0.001012032,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002154527,"about_ca_topic_score_gemma":0.000008748911,"domain_scores_codex":[0.9976344,0.00004950115,0.0009716516,0.0003775605,0.0005638861,0.0004030497],"domain_scores_gemma":[0.9957124,0.0007092893,0.000690995,0.002022332,0.0007533414,0.0001116143],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004510352,0.00000373731,0.00001270635,0.00006122538,0.00001342328,1.546017e-7,0.001380566,0.216181,0.000002398661,0.6259617,0.0001062066,0.1562724],"study_design_scores_gemma":[0.0003984313,0.0001701078,0.00003339674,0.0004124573,0.000009894019,0.000009753071,0.0000286847,0.7669005,0.00003111582,0.000995073,0.2307234,0.0002871684],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.000001193368,0.000111991,0.8438277,0.0007628181,0.00015001,0.0006882619,0.000001794651,0.000120403,0.1543358],"genre_scores_gemma":[0.004356877,0.001347113,0.9723024,0.001503859,0.00005841083,0.0001002575,0.00007869597,0.0000148874,0.02023748],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6249666,"threshold_uncertainty_score":0.9999386,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02445174135414655,"score_gpt":0.3154284189345152,"score_spread":0.2909766775803686,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}