{"id":"W4367319168","doi":"10.1145/3543873.3587661","title":"Investigating Action-Space Generalization in Reinforcement Learning for Recommendation Systems","year":2023,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalization; Reinforcement learning; Computer science; Action (physics); Space (punctuation); Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007968536,0.0001081995,0.0001151456,0.0003061429,0.0001931099,0.000252661,0.0002412839,0.00005872648,0.000008725799],"category_scores_gemma":[0.0002989513,0.000112655,0.00002431881,0.0009517815,0.00001007883,0.0006511,0.0001218801,0.0001212572,0.00006331337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001504324,"about_ca_system_score_gemma":0.00004495847,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001004755,"about_ca_topic_score_gemma":0.000009458177,"domain_scores_codex":[0.9987946,0.00008611403,0.0003635777,0.0002618925,0.0002037194,0.0002901212],"domain_scores_gemma":[0.9993001,0.0001854508,0.0001891519,0.0001865373,0.00008797369,0.0000507997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[5.566613e-7,0.000001730903,0.001809962,0.0000393282,0.000004789654,1.832401e-7,0.0003754626,0.9583325,0.0007775763,0.03517972,0.001984958,0.001493186],"study_design_scores_gemma":[0.0002868216,0.00007128285,0.0004398978,0.00003638456,0.000001724697,8.661077e-7,0.0002265215,0.9874437,0.0006581579,0.0001164738,0.01058652,0.0001316544],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002375465,0.000003268096,0.9927562,0.0011657,0.0006327577,0.0004389872,7.278709e-8,0.0005221728,0.002105341],"genre_scores_gemma":[0.953083,0.00005169816,0.0326907,0.0002355269,0.0001322679,0.0002162022,0.0001942626,0.00002566863,0.01337064],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9600655,"threshold_uncertainty_score":0.4593938,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0737814705673724,"score_gpt":0.3166772349722313,"score_spread":0.2428957644048589,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}