{"id":"W4400233908","doi":"10.1109/noms59830.2024.10575429","title":"Intent Assurance using LLMs guided by Intent Drift","year":2024,"lang":"en","type":"article","venue":"","topic":"Safety Systems Engineering in Autonomy","field":"Engineering","cited_by":18,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006125562,0.001196318,0.0006420616,0.001588303,0.001035435,0.002622789,0.001511788,0.001112948,0.00296464],"category_scores_gemma":[0.02810245,0.0007490191,0.001363166,0.0004243588,0.002586531,0.004066037,0.00439724,0.003352878,0.0009236064],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001959895,"about_ca_system_score_gemma":0.004158546,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006324434,"about_ca_topic_score_gemma":0.006472268,"domain_scores_codex":[0.9941297,0.002048075,0.0004394901,0.0008569057,0.002032116,0.0004938503],"domain_scores_gemma":[0.9842063,0.008053292,0.001632284,0.002749381,0.002808346,0.0005504522],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0008471949,0.0005285498,0.01302981,0.0006881031,0.0001496423,0.001200079,0.00624008,0.3802182,0.06153848,0.2771701,0.009465942,0.2489238],"study_design_scores_gemma":[0.00003970171,0.0001334878,0.0003971639,0.0000886714,0.0000399295,0.0001309982,0.0003154881,0.8805851,0.01724929,0.09250002,0.008467974,0.00005205812],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02495594,0.00009510598,0.965991,0.0006323135,0.00006286086,0.0002674093,0.0001694694,0.005216888,0.002609037],"genre_scores_gemma":[0.5314462,0.0001312436,0.4630439,0.0005272301,0.00005642386,0.0002929324,0.0007108269,0.0009962054,0.002795016],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006324434,"threshold_uncertainty_score":0.03239542,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01541141093161201,"score_gpt":0.2292902315967318,"score_spread":0.2138788206651198,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}