{"id":"W4416750974","doi":"10.1109/iros60139.2025.11246348","title":"AGENTS-LLM: Augmentative GENeration of Challenging Traffic Scenarios with an Agentic LLM Framework","year":2025,"lang":"","type":"article","venue":"","topic":"Autonomous Vehicle Technology and Safety","field":"Engineering","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Key (lock); Control (management); Domain (mathematical analysis); Quality (philosophy); Scenario testing; Scratch; Scale (ratio)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001235457,0.00143399,0.000500096,0.0009067982,0.0004734001,0.00107634,0.002168266,0.001201061,0.005323053],"category_scores_gemma":[0.003851073,0.000590099,0.001108959,0.0003757356,0.0007874041,0.001449404,0.002261227,0.001519576,0.001562092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008131767,"about_ca_system_score_gemma":0.001112063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00474136,"about_ca_topic_score_gemma":0.009996737,"domain_scores_codex":[0.9992537,0.0003124754,0.00004397313,0.0001637308,0.0001798684,0.00004628261],"domain_scores_gemma":[0.9989893,0.000479346,0.00008418205,0.0002231736,0.0001594186,0.00006455096],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002115384,0.0004111007,0.002238661,0.000441518,0.000132437,0.0005428678,0.0004738484,0.7428439,0.01334292,0.02120684,0.01762485,0.2005295],"study_design_scores_gemma":[0.00002285578,0.00005028462,0.0001274758,0.00002176615,0.00001173412,0.00005946016,0.00004369187,0.9827635,0.003424264,0.0054913,0.007967226,0.00001638714],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01401802,0.0001232771,0.9671648,0.0002165963,0.00009793213,0.0004116463,0.0006384726,0.01240017,0.004929048],"genre_scores_gemma":[0.2115283,0.0001251123,0.7809913,0.0001848195,0.00002906679,0.0006747857,0.002345866,0.0009528286,0.003167956],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.005323053,"threshold_uncertainty_score":0.01780736,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02481226748467035,"score_gpt":0.2696134597636205,"score_spread":0.2448011922789501,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}