{"id":"W1504915502","doi":"10.1609/aaai.v24i1.7751","title":"Using Bisimulation for Policy Transfer in MDPs","year":2010,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Office of Naval Research; Natural Sciences and Engineering Research Council of Canada","keywords":"Bisimulation; Computer science; Pessimism; Task (project management); Metric (unit); Markov decision process; Transfer (computing); Action (physics); Quality (philosophy); Theoretical computer science; Artificial intelligence; Markov process; Mathematics; Economics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007882175,0.002238057,0.001968806,0.001868654,0.001047107,0.001956865,0.002273841,0.00217403,0.002672044],"category_scores_gemma":[0.03515433,0.001064351,0.001294177,0.001071349,0.00277464,0.005382282,0.004935751,0.003635802,0.0004140044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003690415,"about_ca_system_score_gemma":0.0027235,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002425946,"about_ca_topic_score_gemma":0.001587715,"domain_scores_codex":[0.995702,0.002318344,0.0003414013,0.0006333038,0.0007110814,0.0002938836],"domain_scores_gemma":[0.982463,0.0127105,0.001603986,0.001627289,0.0009108415,0.0006843841],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001619844,0.00008569885,0.0005597748,0.00009017612,0.000053861,0.00003425929,0.0001449141,0.9099631,0.001381648,0.05936851,0.0002039088,0.02795203],"study_design_scores_gemma":[0.00002610124,0.00008366796,0.00005512581,0.00001553735,0.00000860264,0.000009179269,0.00001449267,0.9539768,0.001409262,0.04409033,0.0002992935,0.00001165014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02952531,0.0001507674,0.9683262,0.0002183936,0.00002353171,0.0000957544,0.00003337372,0.0004736538,0.001152966],"genre_scores_gemma":[0.7091706,0.0002102614,0.28872,0.0001755644,0.00002850704,0.0006063104,0.0001567612,0.0002204187,0.0007114673],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007882175,"threshold_uncertainty_score":0.04168546,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.135113029198047,"score_gpt":0.3565051551249854,"score_spread":0.2213921259269385,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}