{"id":"W4315489128","doi":"10.1109/cdc51059.2022.9992906","title":"Bandit learning with regularized second-order mirror descent","year":2022,"lang":"en","type":"article","venue":"2022 IEEE 61st Conference on Decision and Control (CDC)","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Huawei Technologies","keywords":"Descent (aeronautics); Computer science; Order (exchange); Gradient descent; Artificial intelligence; Mathematical optimization; Applied mathematics; Mathematics; Physics; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002196389,0.00111318,0.001690851,0.0005849386,0.0006484969,0.001679034,0.00157464,0.002070975,0.002264204],"category_scores_gemma":[0.007753974,0.0006349812,0.0006551198,0.0006573282,0.00167748,0.00175916,0.001539036,0.001903452,0.0007084976],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001204109,"about_ca_system_score_gemma":0.001871958,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00391643,"about_ca_topic_score_gemma":0.004225448,"domain_scores_codex":[0.9989226,0.0004961872,0.000051983,0.0001607388,0.000240364,0.0001280149],"domain_scores_gemma":[0.9974036,0.001621395,0.0002679108,0.0002791786,0.0003062209,0.0001217498],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001471757,0.0001256974,0.0007434773,0.0001066444,0.00007761661,0.0001019334,0.0001082937,0.8392695,0.001557784,0.1136291,0.001760134,0.04237275],"study_design_scores_gemma":[0.000006815481,0.00001746146,0.00002648452,0.00000442391,0.000002122016,0.000008277209,0.00000322303,0.9895633,0.000192203,0.009995864,0.0001765681,0.000003276378],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01193219,0.0001824218,0.9849636,0.0002284019,0.00003799719,0.00005293187,0.00002480108,0.0001828085,0.002394793],"genre_scores_gemma":[0.705627,0.0003528027,0.2835946,0.0004648686,0.0001056953,0.0004094815,0.0001643583,0.0001492619,0.009131878],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00391643,"threshold_uncertainty_score":0.01161575,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07159991721118657,"score_gpt":0.3584330874760927,"score_spread":0.2868331702649061,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}