{"id":"W2796264871","doi":"10.1007/978-3-319-89656-4_3","title":"A Novel Evaluation Methodology for Assessing Off-Policy Learning Methods in Contextual Bandits","year":2018,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Bandit Algorithms Research","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Outcome (game theory); Observational study; Machine learning; Artificial intelligence; Evaluation function; Policy learning; Quality (philosophy)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03642529,0.002418153,0.002778412,0.00390129,0.00126086,0.00434828,0.003269281,0.004122772,0.005741042],"category_scores_gemma":[0.1244929,0.0006789464,0.001275143,0.003031188,0.002933996,0.006036631,0.00483964,0.004399035,0.0008020607],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002251989,"about_ca_system_score_gemma":0.002678981,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002159576,"about_ca_topic_score_gemma":0.002192104,"domain_scores_codex":[0.970902,0.01982263,0.001394619,0.002060147,0.005116874,0.0007036873],"domain_scores_gemma":[0.8993618,0.07745282,0.003727749,0.008456172,0.009306684,0.001694826],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001608557,0.0008110956,0.01035562,0.0009027868,0.0007511093,0.0001068925,0.0003574009,0.4267126,0.003349452,0.1286983,0.0053544,0.4209918],"study_design_scores_gemma":[0.00008225357,0.0006683888,0.001105353,0.0001398343,0.0001044871,0.00005934443,0.00007077257,0.953698,0.002026914,0.04017356,0.001830419,0.00004074046],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02111815,0.000797162,0.973078,0.0002376324,0.000182201,0.0003021133,0.0002187835,0.0005396727,0.00352627],"genre_scores_gemma":[0.3333215,0.0005626922,0.660527,0.0002380051,0.000293597,0.001047666,0.0008802205,0.0004614474,0.002667896],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03642529,"threshold_uncertainty_score":0.1926377,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4237425212056544,"score_gpt":0.5813126106282402,"score_spread":0.1575700894225859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}