{"id":"W2169619645","doi":"10.5555/2484920.2485084","title":"Smart exploration in reinforcement learning using absolute temporal difference errors","year":2013,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Reinforcement learning; Temporal difference learning; Computer science; State (computer science); Function (biology); Artificial intelligence; Function approximation; Control (management); Machine learning; Algorithm; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002582136,0.0008085276,0.001039074,0.0004860105,0.0002595262,0.0008500977,0.001126042,0.0008664833,0.001068099],"category_scores_gemma":[0.0105106,0.0004126291,0.0003986102,0.00041096,0.001678652,0.002003563,0.001535976,0.001419337,0.0001436716],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008490774,"about_ca_system_score_gemma":0.000817303,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001911263,"about_ca_topic_score_gemma":0.001289838,"domain_scores_codex":[0.9990346,0.0004134514,0.00005918626,0.0001525148,0.0002684823,0.00007181858],"domain_scores_gemma":[0.9947207,0.004011448,0.0004317104,0.000286229,0.0003455488,0.0002043418],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001574397,0.00004524558,0.0007917265,0.00007360104,0.00003533576,0.00005279025,0.00007621475,0.923762,0.001859635,0.03672472,0.0002385094,0.03618286],"study_design_scores_gemma":[0.00001076811,0.00002396163,0.00004181733,0.000003476126,0.000002214634,0.000005833208,0.00000182899,0.9922647,0.0003455589,0.007216745,0.00007999109,0.000003091208],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02723811,0.0002244694,0.971386,0.0001060827,0.00002711604,0.00002273341,0.0000103712,0.000154456,0.0008305316],"genre_scores_gemma":[0.8889738,0.0001686599,0.109204,0.0000594968,0.00003397859,0.0001096413,0.00003137017,0.00006312833,0.001355845],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002582136,"threshold_uncertainty_score":0.01365578,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1078690293136804,"score_gpt":0.2926166985086204,"score_spread":0.18474766919494,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}