{"id":"W2006330826","doi":"10.1007/s10994-011-5254-7","title":"Model selection in reinforcement learning","year":2011,"lang":"en","type":"article","venue":"Machine Learning","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Mathematics; Estimator; Oracle; Bellman equation; Function (biology); Regularization (linguistics); Countable set; Sequence (biology); Combinatorics; Rate of convergence; Discrete mathematics; Mathematical optimization; Computer science; Artificial intelligence; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003000777,0.0008185222,0.001954886,0.0006059599,0.0004927586,0.001321604,0.001505417,0.00181473,0.00364159],"category_scores_gemma":[0.01370685,0.0007742019,0.0006195215,0.0007234466,0.001998662,0.002015246,0.001269325,0.002395649,0.0004426797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001317766,"about_ca_system_score_gemma":0.0008361903,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003297328,"about_ca_topic_score_gemma":0.002238556,"domain_scores_codex":[0.9980781,0.001338518,0.00005620794,0.0001927943,0.0002447383,0.00008965623],"domain_scores_gemma":[0.9912907,0.007714408,0.0002414044,0.0002522329,0.0003683518,0.0001328744],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009259565,0.0000693143,0.0006326678,0.0001382993,0.00008280487,0.00007131924,0.00006421772,0.8254107,0.0003144218,0.1265685,0.002444992,0.04411007],"study_design_scores_gemma":[0.00001816492,0.00001472121,0.0000452044,0.000007917492,0.000007473988,0.00000790178,0.000003321227,0.9451907,0.00009519334,0.05419817,0.0004071816,0.000004069874],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01222015,0.002021311,0.9801543,0.001014156,0.0001414117,0.00003821355,0.00003349072,0.0002034131,0.004173592],"genre_scores_gemma":[0.849001,0.00142138,0.1394624,0.0004587484,0.0003252094,0.0003296143,0.0001346998,0.0001556336,0.008711359],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00364159,"threshold_uncertainty_score":0.01586986,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03659510899503107,"score_gpt":0.2458099770528048,"score_spread":0.2092148680577738,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}