{"id":"W7124240193","doi":"10.65109/jsvi2318","title":"FedFormer: Contextual Federation with Attention in Reinforcement Learning","year":2023,"lang":"","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Aggregate (composite); Reinforcement; Transformer","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001417785,0.0004532164,0.0004163519,0.0007215359,0.000736035,0.001351442,0.0006298214,0.0001960085,0.0003171446],"category_scores_gemma":[0.0001675106,0.0004179366,0.00008763638,0.002391849,0.0001208641,0.00181559,0.0004868141,0.0007863611,0.002180716],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003669576,"about_ca_system_score_gemma":0.0002801913,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000205436,"about_ca_topic_score_gemma":0.0001409368,"domain_scores_codex":[0.9955066,0.0002332588,0.001059531,0.0008373099,0.00125763,0.001105643],"domain_scores_gemma":[0.9983868,0.000179691,0.000446325,0.0005655784,0.0002429647,0.0001786522],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005865686,0.00002370194,0.01313209,0.00006378045,0.00004607948,0.00004981085,0.001724326,0.9568469,0.0002766121,0.01462067,0.0004684062,0.01268899],"study_design_scores_gemma":[0.001930374,0.001677904,0.0149898,0.0003035616,0.00001667139,0.00001370881,0.001334257,0.9764661,0.0001601909,0.00002426788,0.0025379,0.0005453063],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03056482,0.00002098816,0.9366148,0.001241726,0.0006001518,0.0008340199,1.538627e-7,0.000515341,0.02960799],"genre_scores_gemma":[0.9291329,0.0001652241,0.001733709,0.0002215927,0.00009178789,0.00004634804,0.00006714655,0.00003712707,0.06850416],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9348811,"threshold_uncertainty_score":0.9998273,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02335311171161477,"score_gpt":0.2575649258979304,"score_spread":0.2342118141863156,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}