{"id":"W3047253995","doi":"10.1109/lwc.2020.3045005","title":"Faded-Experience Trust Region Policy Optimization for Model-Free Power Allocation in Interference Channel","year":2020,"lang":"en","type":"preprint","venue":"IEEE Wireless Communications Letters","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Reinforcement learning; Memorization; Convergence (economics); Computer science; Interference (communication); Channel (broadcasting); Power (physics); Control (management); Artificial intelligence; Telecommunications; Mathematics; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001471334,0.0007947319,0.001272925,0.0003459837,0.0002623211,0.000785879,0.0008513997,0.0008805999,0.001400033],"category_scores_gemma":[0.006026149,0.0004831826,0.0004944977,0.0003172537,0.00120589,0.0008127189,0.0009933605,0.001347529,0.0002029264],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009962289,"about_ca_system_score_gemma":0.001297747,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007012845,"about_ca_topic_score_gemma":0.002560551,"domain_scores_codex":[0.9994636,0.0002275216,0.00002368041,0.00009004187,0.0001041204,0.00009095654],"domain_scores_gemma":[0.9977036,0.00163989,0.0002050793,0.0001062066,0.0002430708,0.0001021677],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004042964,0.00001842854,0.00019722,0.00002844393,0.00001339796,0.00002769028,0.0000295345,0.9885734,0.0003496814,0.00418612,0.0002227491,0.006312901],"study_design_scores_gemma":[0.000003445788,0.000009778091,0.00001472098,0.000001535152,0.000001390319,0.000002239907,0.000001477636,0.9987847,0.00007201567,0.001068165,0.00003949875,0.000001067512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02408271,0.0003020497,0.9736241,0.0002135056,0.00003024496,0.00002937804,0.00002046643,0.0002162223,0.001481338],"genre_scores_gemma":[0.951317,0.0001897937,0.04655251,0.00008685139,0.00002585652,0.0001003805,0.00003695337,0.00004802916,0.001642668],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007012845,"threshold_uncertainty_score":0.01394403,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07360881230310362,"score_gpt":0.3074642812208463,"score_spread":0.2338554689177427,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}