{"id":"W3215887789","doi":"10.1109/tai.2021.3117743","title":"Delayed Reward Bernoulli Bandits: Optimal Policy and Predictive Meta-Algorithm PARDI","year":2021,"lang":"en","type":"article","venue":"IEEE Transactions on Artificial Intelligence","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bernoulli's principle; Computer science; Outcome (game theory); Reinforcement learning; Mathematical optimization; Index (typography); Algorithm; Range (aeronautics); Artificial intelligence; Mathematics; Mathematical economics; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002967824,0.0009722789,0.001652062,0.0008764017,0.000477044,0.001541223,0.0022648,0.00170409,0.00206431],"category_scores_gemma":[0.01028762,0.0005288722,0.0005558888,0.000897427,0.0009337318,0.001553052,0.001134583,0.002242036,0.0004565763],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00178585,"about_ca_system_score_gemma":0.00295378,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004258415,"about_ca_topic_score_gemma":0.003569288,"domain_scores_codex":[0.9988053,0.0004881867,0.00005472134,0.0002225846,0.0002648162,0.0001644109],"domain_scores_gemma":[0.9956127,0.003176387,0.0003437208,0.0002950482,0.0003741323,0.0001979682],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000138342,0.00006923902,0.0006413821,0.00004789095,0.00004404215,0.00002392177,0.0000342298,0.9525262,0.0003998591,0.009745201,0.0008479682,0.03548177],"study_design_scores_gemma":[0.00001345664,0.00002429267,0.00003682707,0.000005878747,0.000006101474,0.000007579884,0.000003244247,0.9967616,0.0002034413,0.002740364,0.0001941214,0.000003080042],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04065998,0.001176707,0.9514865,0.0005966365,0.0001168343,0.0001184828,0.00010157,0.0009253374,0.004818097],"genre_scores_gemma":[0.7474827,0.0004237989,0.2481134,0.0003889251,0.00008330234,0.0002854527,0.0002265593,0.0001351903,0.00286069],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004258415,"threshold_uncertainty_score":0.01569551,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1953330603976514,"score_gpt":0.4366686007048083,"score_spread":0.2413355403071569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}