{"id":"W2182877511","doi":"10.82308/46241","title":"Optimal time scales for reinforcement learning behaviour strategies","year":2010,"lang":"en","type":"article","venue":"Open MIND","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Temporal difference learning; Formalism (music); Gradient descent; Representation (politics); Q-learning; Scale (ratio); Machine learning; Mathematical optimization; Artificial neural network; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001871917,0.000790867,0.0007767965,0.0005921947,0.0005761854,0.001620303,0.0009150578,0.001179362,0.004885925],"category_scores_gemma":[0.01243551,0.000459238,0.0005642699,0.0002850553,0.001709384,0.002237814,0.001527788,0.002041544,0.0005086712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001906064,"about_ca_system_score_gemma":0.001196189,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001839506,"about_ca_topic_score_gemma":0.001427581,"domain_scores_codex":[0.9991412,0.0003151173,0.00004969918,0.0001655886,0.0002162255,0.0001122118],"domain_scores_gemma":[0.9967873,0.002230201,0.0003265902,0.0001569182,0.0002659714,0.000233109],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001363442,0.00009551695,0.0008837396,0.0001052108,0.0000313857,0.00008600106,0.000232372,0.6618337,0.002491856,0.3008042,0.001000599,0.03229901],"study_design_scores_gemma":[0.00002219232,0.0000320287,0.0001098889,0.00001451646,0.000005734462,0.000008191371,0.00001734664,0.9272394,0.0003897319,0.07171836,0.0004350506,0.000007631988],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05867064,0.0002995114,0.9313096,0.0004994676,0.00004258123,0.0000902193,0.00005143356,0.0002662427,0.008770403],"genre_scores_gemma":[0.8747287,0.000280984,0.1187893,0.0001201106,0.0000344442,0.0003117411,0.00009231822,0.0001488611,0.005493538],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004885925,"threshold_uncertainty_score":0.01634502,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02682287044972681,"score_gpt":0.3060654353432031,"score_spread":0.2792425648934763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}