{"id":"W4417515405","doi":"10.1109/tvt.2025.3608811","title":"DriveSOTIF: Advancing SOTIF Through Multimodal Large Language Models","year":2025,"lang":"en","type":"preprint","venue":"IEEE Transactions on Vehicular Technology","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of New Brunswick; University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Benchmarking; Inference; Baseline (sea); Domain (mathematical analysis); Work (physics); Language model; Task analysis; Language understanding","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001507573,0.002187164,0.0007487322,0.001279852,0.0005517348,0.00173835,0.002229671,0.00187481,0.004676202],"category_scores_gemma":[0.005323816,0.0005072177,0.002056699,0.0006551388,0.0005196175,0.002577862,0.002156797,0.002781919,0.00357391],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001477844,"about_ca_system_score_gemma":0.001572517,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03244457,"about_ca_topic_score_gemma":0.04920853,"domain_scores_codex":[0.9990789,0.0002860925,0.00005153082,0.000341924,0.0001391558,0.0001024527],"domain_scores_gemma":[0.9986749,0.0006225393,0.00005882007,0.0002907442,0.0002759857,0.00007704776],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005955926,0.0007735951,0.0112271,0.0008944597,0.0005445663,0.000592357,0.0005880696,0.3262733,0.01810343,0.006576207,0.14274,0.4910913],"study_design_scores_gemma":[0.00006657388,0.0001042632,0.001381483,0.00006753371,0.00004924093,0.0001191866,0.0002068533,0.9685298,0.005766645,0.007041058,0.01661175,0.00005569377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.160077,0.004145748,0.6613215,0.002935344,0.001114079,0.0009362167,0.0408315,0.1129725,0.01566609],"genre_scores_gemma":[0.4889543,0.0009743486,0.3820063,0.00167922,0.0002249432,0.0009064034,0.1097001,0.00302741,0.01252719],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03244457,"threshold_uncertainty_score":0.06451142,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009577541560101646,"score_gpt":0.2838089407496976,"score_spread":0.274231399189596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}