{"id":"W4411337688","doi":"10.1109/cain66642.2025.00033","title":"LLM-Based Safety Case Generation for Baidu Apollo: Are We there Yet?","year":2025,"lang":"en","type":"article","venue":"","topic":"Risk and Safety Analysis","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Apollo; Computer science; Astrobiology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004599934,0.0007502736,0.0003149381,0.001762832,0.0007954263,0.002128038,0.001599195,0.001128078,0.008458291],"category_scores_gemma":[0.01779346,0.0006379055,0.0007834231,0.0004846201,0.001118956,0.002242781,0.002360595,0.001420511,0.001817302],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001916368,"about_ca_system_score_gemma":0.002611948,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006756813,"about_ca_topic_score_gemma":0.01368894,"domain_scores_codex":[0.9975932,0.001134358,0.0001769614,0.0003484083,0.0006376845,0.0001093752],"domain_scores_gemma":[0.9874197,0.00903998,0.0007180229,0.001285764,0.001210209,0.0003263552],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001144543,0.001216905,0.01951512,0.002833728,0.0001335388,0.004819328,0.01121832,0.1141849,0.05995984,0.07312915,0.0541408,0.6577038],"study_design_scores_gemma":[0.0003300945,0.0003733365,0.003570434,0.0006943797,0.0001106603,0.001451582,0.00285672,0.7851867,0.03866154,0.03106919,0.1355687,0.0001266675],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.13833,0.0003966153,0.8147598,0.003232936,0.0001939543,0.001996254,0.002153746,0.0211664,0.01777028],"genre_scores_gemma":[0.2232277,0.0001790184,0.7680601,0.0002239256,0.00002416732,0.0005328905,0.002357506,0.001194395,0.004200206],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008458291,"threshold_uncertainty_score":0.02829581,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1572253604098817,"score_gpt":0.4148172320835699,"score_spread":0.2575918716736882,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}