{"id":"W3041828093","doi":"10.24963/ijcai.2020/706","title":"On Overfitting and Asymptotic Bias in Batch Reinforcement Learning with Partial Observability (Extended Abstract)","year":2020,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Samsung; Institut de Valorisation des Données; Waalse Gewest","keywords":"Overfitting; Observability; Reinforcement learning; Term (time); Context (archaeology); Representation (politics); Computer science; Artificial intelligence; State (computer science); Mathematics; Applied mathematics; Algorithm; Artificial neural network","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005383305,0.0002180085,0.0002426462,0.00005948682,0.0001219707,0.0002420723,0.0003945359,0.00006413305,0.00007419045],"category_scores_gemma":[0.0005398792,0.0001784829,0.00003443185,0.0003821998,0.00005358955,0.0004909723,0.0002825856,0.0004957847,0.00004643618],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007330739,"about_ca_system_score_gemma":0.00007586779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006767461,"about_ca_topic_score_gemma":0.000007112149,"domain_scores_codex":[0.9979882,0.00008170085,0.0004395745,0.0005592866,0.0005138665,0.000417408],"domain_scores_gemma":[0.9988407,0.0004075271,0.0001734152,0.0003424966,0.00004807021,0.0001877393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005105786,0.00001752572,0.0259719,0.00003928232,0.00001203398,0.0000193667,0.001163794,0.9596893,0.00008041883,0.01118461,0.00002988102,0.001740891],"study_design_scores_gemma":[0.0008422212,0.0009699839,0.03782829,0.00005993433,0.000004658098,0.000002777655,0.0001227762,0.9588709,0.0005785395,0.00007927073,0.0003936496,0.0002470347],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3961605,0.000007368975,0.5942374,0.001879442,0.0000766524,0.0003238435,7.3772e-8,0.0002612903,0.007053432],"genre_scores_gemma":[0.9894902,0.00000573568,0.009221252,0.001003796,0.00003211116,0.000008826773,0.000002248651,0.00001299543,0.0002227977],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5933298,"threshold_uncertainty_score":0.7278321,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04988424548801556,"score_gpt":0.2554322823185893,"score_spread":0.2055480368305737,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}