{"id":"W3047253995","doi":"10.1109/lwc.2020.3045005","title":"Faded-Experience Trust Region Policy Optimization for Model-Free Power Allocation in Interference Channel","year":2020,"lang":"en","type":"preprint","venue":"IEEE Wireless Communications Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Memorization; Convergence (economics); Computer science; Interference (communication); Channel (broadcasting); Power (physics); Control (management); Artificial intelligence; Telecommunications; Mathematics; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":[],"category_scores_codex":[0.0002939228,0.0004602508,0.0004694378,0.0006213177,0.0002942366,0.0004645365,0.0106704,0.0003163141,0.000001280768],"category_scores_gemma":[0.0002943614,0.0005555893,0.0001700187,0.0008203085,0.0002735523,0.000720596,0.00399321,0.00107763,0.000007678871],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00061796,"about_ca_system_score_gemma":0.0003898683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000184695,"about_ca_topic_score_gemma":0.00002515943,"domain_scores_codex":[0.9970598,0.0002553447,0.0009093373,0.0008916764,0.0003927179,0.0004911415],"domain_scores_gemma":[0.9923154,0.0002440942,0.0007272979,0.00631313,0.0002683004,0.0001318315],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001363446,0.00005275798,0.00003625968,0.00007440855,0.00002262239,0.000001204411,0.007653694,0.9747513,0.0005369202,0.0155234,0.0007864839,0.0005473295],"study_design_scores_gemma":[0.0004542341,0.00003983407,0.00004239258,0.0003200191,0.00001169753,0.000003332062,0.0001126619,0.9971014,0.0002948157,0.001000954,0.0001115109,0.0005071904],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001997293,0.00004620521,0.9240689,0.07163937,0.0004140644,0.0011869,0.00001216932,0.000321668,0.0003134311],"genre_scores_gemma":[0.8078355,0.0003790367,0.1872207,0.003332748,0.0000663704,0.0009084496,0.0001592596,0.00005134724,0.00004657402],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8058382,"threshold_uncertainty_score":0.9996896,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07360881230310362,"score_gpt":0.3074642812208463,"score_spread":0.2338554689177427,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}