{"id":"W2034960258","doi":"10.1109/devlrn.2012.6400860","title":"Scaling life-long off-policy learning","year":2012,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Estimator; Artificial intelligence; Scaling; Convergence (economics); Machine learning; Scale (ratio); Value (mathematics); Coding (social sciences); Mathematics; Economics; Economic growth","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003741396,0.001066333,0.001298164,0.000618497,0.0006432827,0.001438459,0.001988079,0.001138771,0.002858651],"category_scores_gemma":[0.0200761,0.0005604941,0.0005674079,0.0004472447,0.002044901,0.00398849,0.002174198,0.002601195,0.0005380844],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001408736,"about_ca_system_score_gemma":0.001248355,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001588702,"about_ca_topic_score_gemma":0.001083073,"domain_scores_codex":[0.9985251,0.000418418,0.00009358373,0.0003880484,0.000416004,0.0001588395],"domain_scores_gemma":[0.9888137,0.006083583,0.001037075,0.002343365,0.001180491,0.0005417267],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002161153,0.0002789161,0.002229507,0.0001041129,0.00007529949,0.00009023814,0.0001464652,0.9092439,0.004269886,0.0221651,0.001034626,0.06014588],"study_design_scores_gemma":[0.00001453044,0.00008206074,0.000167879,0.000006876275,0.000007032042,0.00002306404,0.00001438432,0.9828367,0.001383619,0.01495001,0.000505383,0.000008592091],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1427491,0.0005193609,0.8503326,0.0005506453,0.0001741773,0.000145638,0.00005793876,0.001166142,0.004304338],"genre_scores_gemma":[0.902022,0.0002168457,0.09486202,0.0002719854,0.00007741109,0.0001925341,0.0001113405,0.0001922411,0.00205362],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003741396,"threshold_uncertainty_score":0.01978666,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03094341280881763,"score_gpt":0.287813299354403,"score_spread":0.2568698865455853,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}