{"id":"W3199199701","doi":"10.48550/arxiv.2106.00589","title":"Improving Long-Term Metrics in Recommendation Systems using Short-Horizon Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Term (time); Horizon; Computer science; Reinforcement; Recommender system; Artificial intelligence; Machine learning; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00483531,0.001568474,0.002044433,0.0006954127,0.0006156,0.001450711,0.001827049,0.001861424,0.00147396],"category_scores_gemma":[0.01909666,0.0006886005,0.0004774932,0.0007991815,0.001380891,0.002850797,0.001227141,0.002514894,0.0005447138],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00161062,"about_ca_system_score_gemma":0.001785935,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007833154,"about_ca_topic_score_gemma":0.006387927,"domain_scores_codex":[0.9980676,0.0007976855,0.0001104448,0.0005099557,0.0003270964,0.000187323],"domain_scores_gemma":[0.9897885,0.007414469,0.0008373418,0.0007476478,0.0008457102,0.000366226],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002462733,0.0002165922,0.002385834,0.0001121624,0.00009725877,0.0000484443,0.00007600005,0.9157991,0.001689523,0.006879771,0.001273664,0.07117534],"study_design_scores_gemma":[0.00001217986,0.00005993639,0.0001586341,0.000006674725,0.000007199239,0.000007530843,0.000004886552,0.9959309,0.0003605405,0.003316439,0.000128826,0.000006334298],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05598072,0.001311184,0.9396368,0.0005274869,0.00007156959,0.00008501612,0.00009277972,0.0009612769,0.001333179],"genre_scores_gemma":[0.896841,0.0004392267,0.09980778,0.0002219808,0.00008801556,0.0001413457,0.0002130037,0.0001153283,0.002132359],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007833154,"threshold_uncertainty_score":0.02557188,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3178481876547598,"score_gpt":0.3384350574772259,"score_spread":0.02058686982246605,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}