{"id":"W4367319168","doi":"10.1145/3543873.3587661","title":"Investigating Action-Space Generalization in Reinforcement Learning for Recommendation Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalization; Reinforcement learning; Computer science; Action (physics); Space (punctuation); Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008401311,0.00119124,0.002270854,0.0006743823,0.000712719,0.001441087,0.001979095,0.002217666,0.002495596],"category_scores_gemma":[0.04943075,0.0008104236,0.001212497,0.0006807463,0.002839378,0.003638553,0.001934761,0.003749818,0.0002678345],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002878726,"about_ca_system_score_gemma":0.001798869,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01399254,"about_ca_topic_score_gemma":0.008540316,"domain_scores_codex":[0.9970315,0.001583009,0.0001340256,0.000613171,0.0003583295,0.0002798997],"domain_scores_gemma":[0.9638111,0.03071594,0.001574838,0.001714936,0.001485594,0.0006975327],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009339934,0.000113331,0.002581259,0.000118766,0.00008910341,0.00008491643,0.0002390701,0.9361517,0.0006227701,0.04457053,0.0006153561,0.01471981],"study_design_scores_gemma":[0.00001070019,0.00003195251,0.0002008861,0.00000805634,0.000005532602,0.000008223138,0.00001196317,0.9820616,0.00006482918,0.01746378,0.0001261589,0.000006344339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1199438,0.001109642,0.8723239,0.002061154,0.00007886307,0.0001621742,0.0001178576,0.00040819,0.003794505],"genre_scores_gemma":[0.9529475,0.000512352,0.04366124,0.0003698158,0.00007825062,0.0001937021,0.0001361731,0.00007356871,0.002027433],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01399254,"threshold_uncertainty_score":0.04443085,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0737814705673724,"score_gpt":0.3166772349722313,"score_spread":0.2428957644048589,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}