{"id":"W1821963137","doi":"","title":"State similarity based approach for improving performance in RL","year":2007,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Reinforcement learning; Similarity (geometry); Computer science; Context (archaeology); State (computer science); Artificial intelligence; Function (biology); Tree (set theory); Bellman equation; Action (physics); Machine learning; Data mining; Mathematical optimization; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002070128,0.000822932,0.001320517,0.001192112,0.0005699845,0.001063249,0.001651781,0.001335103,0.00228664],"category_scores_gemma":[0.007071697,0.0004382256,0.0006205364,0.0008828659,0.001056177,0.002489565,0.001627046,0.001559342,0.0006156112],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009810551,"about_ca_system_score_gemma":0.001094341,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001729132,"about_ca_topic_score_gemma":0.001585062,"domain_scores_codex":[0.998328,0.0006217933,0.0001163153,0.0002439408,0.0005880674,0.0001019055],"domain_scores_gemma":[0.9965738,0.001930142,0.0002432071,0.0005565878,0.0005847477,0.0001115915],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002945212,0.0003643458,0.002111668,0.0001657798,0.0001293798,0.0001154332,0.0002463933,0.6199774,0.01342364,0.04391294,0.001596583,0.3176619],"study_design_scores_gemma":[0.00001447742,0.0001311741,0.0001887733,0.000005530337,0.00001330726,0.000030547,0.000007642411,0.9882997,0.002833185,0.007893566,0.0005708934,0.00001103843],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0161222,0.00024434,0.9804452,0.00009686779,0.00003316016,0.00004911713,0.00001929952,0.001232537,0.001757317],"genre_scores_gemma":[0.6607089,0.0002487923,0.3364556,0.0001650861,0.00007801076,0.0002092807,0.000120955,0.000204014,0.001809425],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00228664,"threshold_uncertainty_score":0.010948,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02225955871261686,"score_gpt":0.2490177287066525,"score_spread":0.2267581699940356,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}