{"id":"W2784831250","doi":"10.1109/icra.2018.8462977","title":"Cross-Domain Transfer in Reinforcement Learning Using Target Apprentice","year":2018,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Air Force Office of Scientific Research; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Transfer of learning; Task (project management); Domain (mathematical analysis); Sample (material); Artificial intelligence; Apprenticeship; Negative transfer; Sample complexity; Reuse; Multi-task learning; Machine learning; Policy learning; Engineering; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002436935,0.0009791151,0.001067545,0.0003566543,0.0003427091,0.0008206344,0.001751945,0.001250179,0.003199506],"category_scores_gemma":[0.007391526,0.0004467555,0.000624573,0.0003071861,0.00145073,0.002013561,0.00298697,0.002168401,0.0008134689],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000684366,"about_ca_system_score_gemma":0.0007523918,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001096286,"about_ca_topic_score_gemma":0.0007675143,"domain_scores_codex":[0.999011,0.0004034581,0.00004666716,0.000273458,0.0001744858,0.00009087903],"domain_scores_gemma":[0.9977993,0.001120988,0.0001407894,0.0005014181,0.0002710758,0.0001663407],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003659242,0.0005480032,0.001865316,0.0001792125,0.0001307376,0.0002605397,0.0003085234,0.7596273,0.01145139,0.02735785,0.001326919,0.1965783],"study_design_scores_gemma":[0.00002355652,0.0001738746,0.000226254,0.000007999634,0.00001188879,0.00004198074,0.00001699276,0.9834041,0.002654012,0.01270358,0.0007242116,0.00001163082],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03762227,0.0001806744,0.9588045,0.0001469392,0.00005189863,0.0001034871,0.00001837812,0.0006244735,0.002447263],"genre_scores_gemma":[0.8794459,0.0001301185,0.1156902,0.0001757737,0.00004053117,0.0002509387,0.00006803962,0.0001109758,0.004087415],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003199506,"threshold_uncertainty_score":0.0128879,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03322628946258945,"score_gpt":0.2995726163778216,"score_spread":0.2663463269152322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}