{"id":"W2900798285","doi":"10.65109/loxq3741","title":"Urban Driving with Multi-Objective Deep Reinforcement Learning","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Reinforcement learning; Computer science; Task (project management); Artificial intelligence; Domain (mathematical analysis); Dual (grammatical number); Markov decision process; Function (biology); Lexicographical order; Collision; Markov process; Machine learning; Computer security; Engineering; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000846913,0.0005207915,0.0005106929,0.0001739139,0.0002026449,0.0004421672,0.0008898489,0.0007392013,0.001817672],"category_scores_gemma":[0.002058855,0.0003350296,0.0002597933,0.0002115349,0.0006944421,0.0006999109,0.0009305014,0.001267266,0.0002388064],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008961518,"about_ca_system_score_gemma":0.0009950281,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006512761,"about_ca_topic_score_gemma":0.006635695,"domain_scores_codex":[0.9997931,0.00006556689,0.000009148242,0.00004996188,0.00004254484,0.00003968597],"domain_scores_gemma":[0.99943,0.0002969725,0.00005868031,0.00005789984,0.0001089519,0.00004745021],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004510696,0.00005323563,0.0005359584,0.0000215233,0.00001812077,0.0000276944,0.00001750897,0.9707963,0.0007668988,0.003741926,0.0005295865,0.02344618],"study_design_scores_gemma":[0.000003769748,0.000008753976,0.0000234907,8.111402e-7,8.057033e-7,0.000001737588,9.824967e-7,0.9985794,0.0001061815,0.001214027,0.00005917232,8.271102e-7],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07895666,0.0002573389,0.9155967,0.0004813308,0.00006544072,0.00005603055,0.00007167653,0.0008108197,0.003703985],"genre_scores_gemma":[0.9459378,0.00005072673,0.05147228,0.0001182817,0.00001842974,0.00005035146,0.00007795732,0.00003223545,0.002242082],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006512761,"threshold_uncertainty_score":0.01294971,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01674804821501111,"score_gpt":0.2469498928797251,"score_spread":0.230201844664714,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}