{"id":"W1460549585","doi":"10.1007/978-3-642-29946-9_13","title":"Regularized Least Squares Temporal Difference Learning with Nested ℓ2 and ℓ1 Penalization","year":2012,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Advanced Multi-Objective Optimization Algorithms","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Overfitting; Computer science; Regularization (linguistics); Reinforcement learning; Solver; Mathematical optimization; Algorithm; Artificial intelligence; Applied mathematics; Mathematics; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002825659,0.000936613,0.001506397,0.0003308853,0.0004340091,0.001062053,0.002795857,0.002423379,0.004176989],"category_scores_gemma":[0.007773041,0.0009278238,0.001142156,0.0006101886,0.001207669,0.001978695,0.00248498,0.003293197,0.001613474],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007009823,"about_ca_system_score_gemma":0.001894015,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002944626,"about_ca_topic_score_gemma":0.003995334,"domain_scores_codex":[0.9988738,0.0003693438,0.0000852071,0.0002741813,0.0002951644,0.0001022288],"domain_scores_gemma":[0.9973922,0.001294065,0.0001619539,0.0004200571,0.0006015204,0.0001301984],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004311235,0.0001777554,0.0006994676,0.000238356,0.0001591478,0.0001112739,0.0001360373,0.6529124,0.009341283,0.0607908,0.008280999,0.2667213],"study_design_scores_gemma":[0.000007629552,0.00002090229,0.00003493328,0.000004663056,0.000005481992,0.00001488206,0.000002941942,0.9944766,0.0007863543,0.003945879,0.0006931516,0.000006644721],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.001551043,0.00007700965,0.9976215,0.00007102722,0.00006216658,0.00001610461,0.000026508,0.0002055576,0.000369006],"genre_scores_gemma":[0.1078446,0.0001728236,0.8830202,0.000249422,0.0001757506,0.000284948,0.0004689333,0.0003701117,0.007413237],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004176989,"threshold_uncertainty_score":0.01494372,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01232845308421654,"score_gpt":0.2322221378745569,"score_spread":0.2198936847903404,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}