{"id":"W4404520129","doi":"10.22541/au.173204444.43700477/v1","title":"MaxExp-UCB: Enhanced Regret Bounds for Normalized Exploration Stochastic Multi-Armed Bandit Problems With High Action Spaces","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Reservoir Engineering and Simulation Methods","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Regret; Action (physics); Mathematical optimization; Mathematical economics; Computer science; Mathematics; Machine learning; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0003913674,0.000556359,0.0005560285,0.0003692424,0.00008335504,0.0004114123,0.0002081046,0.0004412575,0.00006336321],"category_scores_gemma":[0.00008022247,0.0004725137,0.0001526324,0.0002516344,0.00002550882,0.0002661602,0.0001053715,0.0006832864,0.00002465204],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002433204,"about_ca_system_score_gemma":0.00007099153,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006025973,"about_ca_topic_score_gemma":0.0001067685,"domain_scores_codex":[0.9981378,0.0000498435,0.0005000424,0.0005679453,0.0003337622,0.0004106],"domain_scores_gemma":[0.9988953,0.0001816304,0.0001017855,0.0005169892,0.0001765942,0.0001276971],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007755474,0.0000178365,0.000001345402,0.002199232,0.0002195167,0.000001305735,0.0006683174,0.9888492,0.006542125,0.0001916368,0.0003279349,0.0009040456],"study_design_scores_gemma":[0.001414679,0.00009482891,0.00002067084,0.0005579857,0.0001322107,0.000002403578,0.0001219232,0.9788172,0.01519992,0.002259839,0.000744857,0.0006334843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09332575,0.0003265546,0.9003965,0.0001156497,0.002189169,0.001516395,0.00004657929,0.001852621,0.0002307278],"genre_scores_gemma":[0.8126058,0.000107083,0.1819229,0.000006534545,0.0003964697,0.001747767,0.0004354386,0.0002049373,0.002573136],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.71928,"threshold_uncertainty_score":0.9997727,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0590815207010501,"score_gpt":0.3127189573357616,"score_spread":0.2536374366347115,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}