{"id":"W4406132533","doi":"10.1007/s00521-024-10829-4","title":"Do as you teach: a multi-teacher approach to self-play in deep reinforcement learning","year":2025,"lang":"en","type":"article","venue":"Neural Computing and Applications","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reinforcement learning; Computer science; Context (archaeology); Curriculum; Reinforcement; Computational Science and Engineering; Artificial intelligence; State space; Space (punctuation); Baseline (sea); Mathematics education; Machine learning; Psychology; Pedagogy; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001896244,0.0004646228,0.0006302363,0.0002367481,0.0004376545,0.00065254,0.001635056,0.001163648,0.004127193],"category_scores_gemma":[0.00471535,0.0004034952,0.000309672,0.0001881654,0.0009613004,0.001296321,0.001708251,0.002287835,0.000391451],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007518722,"about_ca_system_score_gemma":0.001047201,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003281498,"about_ca_topic_score_gemma":0.005359975,"domain_scores_codex":[0.9994692,0.0002994251,0.0000162906,0.00008164767,0.00007897182,0.00005445868],"domain_scores_gemma":[0.9982948,0.001109175,0.0001107396,0.0001569142,0.0001626712,0.0001656363],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006282298,0.0006613072,0.003353604,0.0001484416,0.0001290459,0.0001483848,0.0006157589,0.6423069,0.005953283,0.1004389,0.005928086,0.239688],"study_design_scores_gemma":[0.00001425271,0.00002628626,0.00005707866,0.000004206339,0.000005617132,0.000006088495,0.00001179623,0.9907163,0.0004230495,0.008361473,0.0003706726,0.000003174403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02650362,0.0001321517,0.9690357,0.0006561167,0.00007288231,0.00006378479,0.00003118421,0.0004982703,0.003006263],"genre_scores_gemma":[0.8256607,0.00008229955,0.1674963,0.0002149201,0.00004290454,0.0001633717,0.00003852685,0.0001410235,0.006159921],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004127193,"threshold_uncertainty_score":0.01380688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0195471957952896,"score_gpt":0.2921890384630919,"score_spread":0.2726418426678023,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}