{"id":"W2169190416","doi":"","title":"Using bisimulation for policy transfer in MDPs (Extended Abstract)","year":2010,"lang":"en","type":"article","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Bellman equation; Set (abstract data type); Function (biology); Action (physics); Value (mathematics); Computer science; State (computer science); Mathematical economics; Markov process; Artificial intelligence; Mathematics; Algorithm; Machine learning; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005578457,0.002820526,0.002648564,0.001742126,0.0009217067,0.002640978,0.00270896,0.002637893,0.01418426],"category_scores_gemma":[0.02699168,0.001247588,0.003007824,0.002023193,0.002557985,0.004383248,0.005154373,0.005178037,0.002367799],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003748423,"about_ca_system_score_gemma":0.003674315,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006882444,"about_ca_topic_score_gemma":0.003436452,"domain_scores_codex":[0.9964178,0.00170048,0.00020227,0.0006821735,0.0006838851,0.0003134801],"domain_scores_gemma":[0.9856873,0.01138081,0.0008950141,0.0006995584,0.0009166238,0.000420633],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00008191964,0.00009341439,0.0003259391,0.0002283254,0.00006393338,0.00008279735,0.00009776448,0.823595,0.0004241361,0.1501325,0.001121364,0.02375291],"study_design_scores_gemma":[0.00001665958,0.0000335546,0.00002808294,0.00003882089,0.00001158339,0.00001514689,0.000008253022,0.914894,0.0001964473,0.08361085,0.001135073,0.00001159549],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003829596,0.000503979,0.9896841,0.0003746001,0.00006759519,0.00009729141,0.0001334616,0.0003306982,0.004978672],"genre_scores_gemma":[0.4656967,0.002403845,0.5149169,0.0009148582,0.0003246696,0.00185638,0.001022764,0.0008685903,0.0119953],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01418426,"threshold_uncertainty_score":0.04745108,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04872600221822092,"score_gpt":0.3417722960033684,"score_spread":0.2930462937851475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}