{"id":"W6968595109","doi":"10.5281/zenodo.3554750","title":"Statistical Consequences of using Multi-armed Bandits to Conduct Adaptive Educational Experiments","year":2019,"lang":"en","type":"article","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Key (lock); Statistical power; Randomized experiment; Design of experiments; Statistical hypothesis testing; Statistical analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1604071,0.001442907,0.001804197,0.0007121202,0.001318252,0.002796623,0.002420316,0.002821563,0.002270271],"category_scores_gemma":[0.3724174,0.0012089,0.001477216,0.000888871,0.005513244,0.003944513,0.002360751,0.005180381,0.0004703185],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002733784,"about_ca_system_score_gemma":0.002985023,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001976057,"about_ca_topic_score_gemma":0.001930032,"domain_scores_codex":[0.8514693,0.1269524,0.004491127,0.008041765,0.007386815,0.00165857],"domain_scores_gemma":[0.3989277,0.5337731,0.02748873,0.03197604,0.006345691,0.001488793],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.009555975,0.002446014,0.06958719,0.001152239,0.002911145,0.0004346062,0.001684715,0.6137323,0.01098587,0.1143401,0.002605954,0.1705639],"study_design_scores_gemma":[0.001971611,0.005497477,0.02033691,0.0003037213,0.0005138212,0.0001189008,0.0004098232,0.7876242,0.01006903,0.1693476,0.003630571,0.0001762653],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2656183,0.0006542433,0.7225544,0.003051315,0.0002704104,0.002466412,0.0002881165,0.0006165869,0.004480171],"genre_scores_gemma":[0.8078262,0.0001602618,0.1856892,0.001196469,0.00006532211,0.004114388,0.0001401014,0.00005591049,0.0007520724],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1604071,"threshold_uncertainty_score":0.8483241,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3190968909175737,"score_gpt":0.4519885797029352,"score_spread":0.1328916887853615,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}