{"id":"W2174786457","doi":"10.48550/arxiv.1511.06342","title":"Actor-Mimic: Deep Multitask and Transfer Reinforcement Learning","year":2015,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":208,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Computer science; Exploit; Artificial intelligence; Transfer of learning; Set (abstract data type); Task (project management); Machine learning; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00127193,0.0007304588,0.0006343292,0.0002545187,0.0002718916,0.0006206107,0.001766773,0.001013901,0.002622169],"category_scores_gemma":[0.003652343,0.0003778806,0.0004145399,0.0002616442,0.001109794,0.001295918,0.001382179,0.001803728,0.0005013078],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008899579,"about_ca_system_score_gemma":0.001061716,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002132785,"about_ca_topic_score_gemma":0.002278087,"domain_scores_codex":[0.9995056,0.0002126898,0.00001783684,0.00009817468,0.0001169827,0.00004861876],"domain_scores_gemma":[0.9991575,0.0004173838,0.0000987707,0.0001571937,0.00008735342,0.0000818663],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001225323,0.0001154397,0.0007952304,0.00007630991,0.00006381912,0.0001000521,0.00007857683,0.8703164,0.003032673,0.05194716,0.003002346,0.07034942],"study_design_scores_gemma":[0.000007421662,0.00001925154,0.00002941905,0.000002640205,0.000002238069,0.000008883296,0.000001998966,0.9875773,0.0004783437,0.01132845,0.0005415728,0.000002564677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01608139,0.0001872172,0.9786723,0.000318353,0.00006556316,0.00005294691,0.00004377806,0.0009716284,0.003606827],"genre_scores_gemma":[0.7859378,0.0001920912,0.206657,0.0002254105,0.00005211891,0.0002368795,0.0001229072,0.0001493791,0.006426381],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002622169,"threshold_uncertainty_score":0.008772016,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07661995159032055,"score_gpt":0.1959416640452457,"score_spread":0.1193217124549251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}