{"id":"W6989067501","doi":"","title":"Aggressiveness-regulated Multi-agent Stress Testing of Autonomous Vehicles","year":2023,"lang":"en","type":"dissertation","venue":"UWSpace (University of Waterloo)","topic":"Autonomous Vehicle Technology and Safety","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Blackberry (Canada)","funders":"","keywords":"Reinforcement learning; Accident (philosophy); Quality (philosophy); Reinforcement; Task (project management); Stress test","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.00009685486,0.0002895575,0.0005515185,0.0003832153,0.0001353089,0.000005885868,0.0004695676,0.0006420525,0.00004251533],"category_scores_gemma":[0.00002283127,0.0003654064,0.0001478581,0.0003711846,0.0001322239,0.0001012406,0.00007459618,0.0003701496,0.0000519603],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009991417,"about_ca_system_score_gemma":0.00006889932,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006638706,"about_ca_topic_score_gemma":0.01194777,"domain_scores_codex":[0.9989707,0.00002910853,0.0002317242,0.0002927973,0.0001646794,0.0003109922],"domain_scores_gemma":[0.99907,0.00006633678,0.0002875843,0.0003560776,0.0001584345,0.00006149425],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","study_design_scores_codex":[0.000521131,0.000683313,0.03162807,0.01203807,0.003534517,0.001177415,0.2494545,0.2663397,0.2340668,0.0003672576,0.001292502,0.1988966],"study_design_scores_gemma":[0.003283494,0.0003209408,0.373861,0.004499791,0.00091751,0.0000116124,0.1923172,0.1625984,0.2595398,0.0002790685,0.0002657437,0.002105459],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9979405,0.000300761,0.00003864547,0.00003693626,0.0003370199,0.0002169276,0.0001260442,0.0008821382,0.0001209577],"genre_scores_gemma":[0.9276585,0.0001637397,0.002140825,0.000001021407,0.00001635433,0.000001025179,0.0004830356,0.00008308345,0.0694524],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3422329,"threshold_uncertainty_score":0.9999762,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0147286928116283,"score_gpt":0.20237416692376,"score_spread":0.1876454741121316,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}