{"id":"W2952662670","doi":"10.48550/arxiv.1906.04328","title":"Importance Resampling for Off-policy Prediction","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Resampling; Consistency (knowledge bases); Variance (accounting); Computer science; Reinforcement learning; Sampling (signal processing); Jackknife resampling; Function (biology); Statistics; Artificial intelligence; Machine learning; Econometrics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004552013,0.001022894,0.001342334,0.0007131572,0.0005134154,0.001107947,0.001723172,0.001075557,0.002173658],"category_scores_gemma":[0.02653713,0.0006236212,0.000543117,0.0004640268,0.00126521,0.001943766,0.001316648,0.002166838,0.0003702441],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001072415,"about_ca_system_score_gemma":0.001365272,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003891967,"about_ca_topic_score_gemma":0.004134072,"domain_scores_codex":[0.9976603,0.001031867,0.0001125409,0.0004171004,0.0005804407,0.0001977067],"domain_scores_gemma":[0.9889714,0.007464858,0.0007912026,0.001424766,0.001053963,0.0002938197],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004575955,0.0003018676,0.004149548,0.000137687,0.0001124145,0.000169915,0.0001626365,0.8193266,0.005652366,0.03091138,0.0024533,0.1361646],"study_design_scores_gemma":[0.0000168457,0.00005826627,0.0002219655,0.000009142627,0.000007601961,0.00001568999,0.000008995376,0.9899677,0.001392839,0.007926964,0.0003670631,0.000006919023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03421153,0.0002466288,0.9626913,0.0002511611,0.00007084683,0.0001199822,0.00004201196,0.000659048,0.001707527],"genre_scores_gemma":[0.856603,0.0001502978,0.1406798,0.0002642401,0.00007191511,0.0002169125,0.0001468691,0.0001651722,0.001701843],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004552013,"threshold_uncertainty_score":0.0240736,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08829860924234678,"score_gpt":0.2152403378248559,"score_spread":0.1269417285825091,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}