{"id":"W2169190416","doi":"","title":"Using bisimulation for policy transfer in MDPs (Extended Abstract)","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Bellman equation; Set (abstract data type); Function (biology); Action (physics); Value (mathematics); Computer science; State (computer science); Mathematical economics; Markov process; Artificial intelligence; Mathematics; Algorithm; Machine learning; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002238446,0.00008245969,0.0000851771,0.0002048619,0.00005426785,0.0001078148,0.0003291751,0.00007304892,0.00002151545],"category_scores_gemma":[0.0000774635,0.00007699045,0.00004116069,0.0002742567,0.00001919815,0.0004463372,0.00003087005,0.0001489025,0.00001138943],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000348575,"about_ca_system_score_gemma":0.0000984224,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001717318,"about_ca_topic_score_gemma":0.00006491627,"domain_scores_codex":[0.9992232,0.000007611692,0.0002125628,0.0001841884,0.0001443344,0.0002280762],"domain_scores_gemma":[0.9995111,0.00009425154,0.00002430912,0.0002763485,0.00004925378,0.00004469486],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000004119468,0.0000184801,0.0005489699,0.00001012523,0.000003333392,8.677884e-7,0.0002635423,0.7582277,0.01926488,0.2153023,0.00001147663,0.00634416],"study_design_scores_gemma":[0.0003973588,0.00002998198,0.008543811,0.000004812539,0.000001360368,0.000001911437,0.000003512461,0.9850253,0.003284049,0.001870074,0.0007336409,0.0001042246],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0973371,7.525394e-7,0.89906,0.0004569525,0.0002083197,0.0002069118,2.578636e-7,0.00007694311,0.00265281],"genre_scores_gemma":[0.8774291,4.984719e-7,0.1220742,0.0001657186,0.00007557668,0.000003658781,0.000001489952,0.000007048024,0.0002427318],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.7800919,"threshold_uncertainty_score":0.313958,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04872600221822092,"score_gpt":0.3417722960033684,"score_spread":0.2930462937851475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}