{"id":"W7117323897","doi":"10.1007/s10994-025-06943-6","title":"Extended UCB Policies for Multi-armed Bandit Problems","year":2025,"lang":"en","type":"article","venue":"Machine Learning","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Novelis (Canada)","funders":"","keywords":"Regret; Process (computing); Markov decision process; Simplicity; Order (exchange); Reinforcement learning; Extension (predicate logic)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007665602,0.001715362,0.003312043,0.00148158,0.001006124,0.003590556,0.002785838,0.00317323,0.009099055],"category_scores_gemma":[0.02809129,0.001107897,0.0008827564,0.00212546,0.002053805,0.003678079,0.002951504,0.004284423,0.00223261],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001844045,"about_ca_system_score_gemma":0.002585754,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003590044,"about_ca_topic_score_gemma":0.002302555,"domain_scores_codex":[0.9960942,0.002332142,0.00018084,0.0003520169,0.0005921317,0.0004486909],"domain_scores_gemma":[0.9830089,0.01264182,0.0009575584,0.001240882,0.001432324,0.0007184897],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003417681,0.0001785205,0.0005305912,0.0002736365,0.00008131799,0.00008138409,0.0001455994,0.7842264,0.0006506696,0.1548126,0.006997162,0.05168043],"study_design_scores_gemma":[0.00003715109,0.00003881512,0.00006825851,0.00005388462,0.00001031149,0.0000170678,0.00001465212,0.9495411,0.0001579427,0.04898215,0.001066347,0.00001219802],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01802219,0.001967058,0.969257,0.001096361,0.0002286177,0.0001161884,0.0001931379,0.0004153303,0.008704057],"genre_scores_gemma":[0.6957247,0.003588697,0.2737499,0.0009805163,0.0005851321,0.001104795,0.0007270842,0.0005254673,0.02301383],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009099055,"threshold_uncertainty_score":0.0405401,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1527085351058629,"score_gpt":0.4869636691665585,"score_spread":0.3342551340606956,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}