{"id":"W1555338578","doi":"10.5555/1838206.1838401","title":"Using bisimulation for policy transfer in MDPs","year":2010,"lang":"en","type":"article","venue":"Adaptive Agents and Multi-Agents Systems","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Markov decision process; Computer science; Bisimulation; Artificial intelligence; Markov process; Work (physics); Theoretical computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003589092,0.00019336,0.0002271718,0.0002670496,0.0001581891,0.0001818818,0.0003343244,0.0001205586,0.000002545551],"category_scores_gemma":[0.00006086274,0.0001790246,0.00005731546,0.0002719289,0.00004286201,0.000473701,0.00008850333,0.0001677586,0.000008193044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007082217,"about_ca_system_score_gemma":0.00006606551,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00110339,"about_ca_topic_score_gemma":0.0000551378,"domain_scores_codex":[0.9985343,0.00007165417,0.0003829431,0.000400286,0.0002406153,0.0003702175],"domain_scores_gemma":[0.9992909,0.00008075221,0.00009036305,0.0003181046,0.0001035865,0.0001162718],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003427463,0.0001816662,0.02503603,0.0001424466,0.0001134289,0.00001706311,0.004427384,0.9009987,0.00936086,0.05498131,0.00009625358,0.004610579],"study_design_scores_gemma":[0.001419381,0.00007965182,0.01204898,0.00005056735,0.000008773159,0.000005859238,0.00005800115,0.9837282,0.0001253144,0.00002831322,0.002246993,0.0001999363],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1618139,0.00002626436,0.8362215,0.00004938491,0.0008022456,0.0008798485,0.000006339263,0.0000451384,0.0001553874],"genre_scores_gemma":[0.9801678,0.00001168891,0.01909513,0.00009867714,0.0001235239,0.00003271797,0.000005768107,0.00002053577,0.0004441455],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8183539,"threshold_uncertainty_score":0.7300411,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1285799627275599,"score_gpt":0.3554582096881,"score_spread":0.2268782469605402,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}