{"id":"W2158304715","doi":"10.1109/tsmcb.2007.899419","title":"Positive Impact of State Similarity on Reinforcement Learning Performance","year":2007,"lang":"en","type":"article","venue":"IEEE Transactions on Systems Man and Cybernetics Part B (Cybernetics)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Reinforcement learning; Similarity (geometry); Artificial intelligence; Context (archaeology); Computer science; Reinforcement; Function (biology); Bellman equation; Tree (set theory); State (computer science); Action (physics); Machine learning; Value (mathematics); Mathematics; Mathematical optimization; Algorithm; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005013166,0.0008216931,0.001484189,0.000613075,0.0006221985,0.001409774,0.0009282367,0.001436806,0.002172591],"category_scores_gemma":[0.05086755,0.0002899368,0.0003792305,0.0004015602,0.001353662,0.003076951,0.001929792,0.002095353,0.0003985104],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008439706,"about_ca_system_score_gemma":0.001185228,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001080344,"about_ca_topic_score_gemma":0.0009696226,"domain_scores_codex":[0.9958392,0.001362264,0.0003194651,0.0008116799,0.001302456,0.0003650109],"domain_scores_gemma":[0.9497266,0.0391843,0.003669179,0.003478133,0.002472145,0.001469566],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002301173,0.002003912,0.03193978,0.0004015616,0.0004038662,0.000419459,0.0003493269,0.594425,0.02118233,0.02379922,0.001461228,0.3213131],"study_design_scores_gemma":[0.0001043687,0.001672062,0.01100128,0.0000322665,0.00007951412,0.0002387731,0.00009125263,0.9540259,0.009196662,0.0229144,0.000596108,0.00004749925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6549893,0.001005822,0.331729,0.0009371417,0.0001818405,0.0001636954,0.00007587703,0.00144645,0.009470782],"genre_scores_gemma":[0.989744,0.00005837389,0.009695958,0.0000621413,0.00003088935,0.00002559054,0.00003941114,0.000032298,0.0003113555],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005013166,"threshold_uncertainty_score":0.0265125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01715931351416537,"score_gpt":0.2565818502936981,"score_spread":0.2394225367795328,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}