{"id":"W4403906488","doi":"10.1007/978-3-031-73033-7_9","title":"Learning to Drive via Asymmetric Self-Play","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002747987,0.0004571372,0.0002983314,0.0001635239,0.0002444222,0.0005024121,0.0006219115,0.000433136,0.006726088],"category_scores_gemma":[0.001371079,0.0002015017,0.0002375578,0.0001297453,0.0006181494,0.0008396079,0.001131626,0.0008926743,0.0007438922],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002637297,"about_ca_system_score_gemma":0.0002769631,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005508066,"about_ca_topic_score_gemma":0.0006640007,"domain_scores_codex":[0.9998885,0.0000255399,0.000006692755,0.00002726191,0.00003086558,0.00002110121],"domain_scores_gemma":[0.999622,0.0002232102,0.00003550901,0.0000425217,0.00003540364,0.00004134852],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003331632,0.0001989828,0.0009251895,0.000185384,0.00005751421,0.0001598647,0.0002385559,0.459474,0.01807649,0.272011,0.005116833,0.243223],"study_design_scores_gemma":[0.00001530023,0.00008989553,0.0001521937,0.00001117318,0.000005669367,0.00005821204,0.00001885522,0.9288342,0.001734643,0.0672896,0.001783425,0.000006897885],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09600129,0.0002895978,0.8482579,0.0004199286,0.000137105,0.00006911166,0.00006119644,0.0004895139,0.05427441],"genre_scores_gemma":[0.9515278,0.0001640948,0.02439176,0.00007283984,0.00003416442,0.00008664237,0.00004672365,0.00004884889,0.02362725],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006726088,"threshold_uncertainty_score":0.02250099,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00983572967448916,"score_gpt":0.239412937519269,"score_spread":0.2295772078447798,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}