{"id":"W3057037158","doi":"10.1007/978-3-030-56150-5_4","title":"Extending Sliding-Step Importance Weighting from Supervised Learning to Reinforcement Learning","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Machine Learning and Algorithms","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Machine learning; Artificial intelligence; Weighting; Stochastic gradient descent; Robustness (evolution); Online machine learning; Supervised learning; Mathematical optimization; Algorithm; Semi-supervised learning; Mathematics; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001874921,0.0006310886,0.001124808,0.0005544567,0.000235256,0.0006572119,0.00147437,0.0008460822,0.002524837],"category_scores_gemma":[0.005585707,0.0004093323,0.0005311516,0.0007687953,0.0006539181,0.001568131,0.001393628,0.001487301,0.0004907912],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004815139,"about_ca_system_score_gemma":0.0006955601,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001885814,"about_ca_topic_score_gemma":0.001980179,"domain_scores_codex":[0.9993448,0.0002090235,0.00004369525,0.0001450069,0.0002121959,0.00004526334],"domain_scores_gemma":[0.9977037,0.001415788,0.0001011875,0.000314594,0.0003914476,0.00007332048],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001261965,0.0002229223,0.0005284514,0.0001707715,0.00009984468,0.00005827088,0.00007638073,0.4648907,0.007100862,0.03556622,0.003037564,0.4881217],"study_design_scores_gemma":[0.000005385951,0.0000256005,0.00005907612,0.00000495618,0.000006271655,0.000009050928,0.00000177438,0.9863771,0.0007087725,0.01228804,0.0005106216,0.000003362012],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003733865,0.0002040628,0.9949272,0.00003627848,0.00005156112,0.00001833842,0.000009080862,0.0001582922,0.0008612067],"genre_scores_gemma":[0.4381138,0.0005485763,0.5534914,0.0001591322,0.0001880397,0.0001655529,0.0001465825,0.0002162039,0.006970645],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.002524837,"threshold_uncertainty_score":0.00991565,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01447910791569336,"score_gpt":0.2424965017675249,"score_spread":0.2280173938518316,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}