{"id":"W6888860372","doi":"10.24433/co.9982747.v1","title":"A Thompson sampling algorithm to learn Whittle index policy for restless bandits","year":2022,"lang":"en","type":"other","venue":"Code Ocean","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Markov chain; Class (philosophy); Scheduling (production processes); Markov process; Sampling (signal processing); Index (typography)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002292624,0.001103653,0.002610074,0.000925824,0.00088825,0.001455218,0.002247323,0.002080709,0.006950157],"category_scores_gemma":[0.0107632,0.0008595244,0.0008059841,0.001043376,0.001481183,0.002243837,0.001585845,0.002547681,0.001202225],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001473936,"about_ca_system_score_gemma":0.002844209,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01061089,"about_ca_topic_score_gemma":0.01147316,"domain_scores_codex":[0.9988895,0.0004220126,0.00005666986,0.0002351498,0.0002125747,0.0001841057],"domain_scores_gemma":[0.9950004,0.003779775,0.0002253699,0.000317587,0.0004036355,0.0002731444],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000304165,0.0001351752,0.0007734484,0.00007421802,0.00005698775,0.00006585493,0.0000765722,0.8855247,0.0009368934,0.03434781,0.003928011,0.07377626],"study_design_scores_gemma":[0.00001932499,0.00001597948,0.00002276114,0.00000459822,0.000003128617,0.000003462059,0.000003316924,0.9917994,0.0001088836,0.007862234,0.0001532567,0.000003676763],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02321938,0.0003555917,0.9720564,0.0002672139,0.00008959819,0.0001033104,0.0001375213,0.001093712,0.002677195],"genre_scores_gemma":[0.6267831,0.0003072532,0.3604493,0.0005313993,0.0001840167,0.0004117093,0.0009526737,0.0005604097,0.00982001],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01061089,"threshold_uncertainty_score":0.02325064,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04843255918529042,"score_gpt":0.3466562063849024,"score_spread":0.2982236471996119,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}