{"id":"W4288391571","doi":"10.1109/tse.2022.3194640","title":"FalsifAI: Falsification of AI-Enabled Hybrid Control Systems Guided by Time-Aware Coverage Criteria","year":2022,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Correctness; Robustness (evolution); Artificial neural network; Notation; Context (archaeology); Hybrid system; Semantics (computer science); Artificial intelligence; Model checking; Theoretical computer science; Programming language; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"not_applicable","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"software","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00315003,0.001354709,0.0009052843,0.001209511,0.0006931254,0.002100108,0.001569403,0.001407962,0.002868589],"category_scores_gemma":[0.01687434,0.0004162077,0.001507886,0.00041264,0.003257018,0.001955347,0.00261808,0.00153952,0.0002726306],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001559846,"about_ca_system_score_gemma":0.001556151,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003673008,"about_ca_topic_score_gemma":0.00228052,"domain_scores_codex":[0.9970518,0.0007520982,0.0001893138,0.0005533762,0.001015178,0.0004381625],"domain_scores_gemma":[0.9888131,0.008006147,0.0009461608,0.0008778189,0.001051927,0.0003048054],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0004209612,0.00008682279,0.00487337,0.0003367555,0.0001483546,0.0009361655,0.0005429609,0.7841551,0.01553355,0.1518863,0.001537975,0.03954169],"study_design_scores_gemma":[0.00001997802,0.00007529673,0.0002437093,0.00004220189,0.00001955993,0.00009293517,0.00004516509,0.9475796,0.005262566,0.04568874,0.0009113516,0.00001885649],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05156304,0.0002933695,0.9412859,0.0004476713,0.00007161962,0.0001164843,0.0001833454,0.001435306,0.004603367],"genre_scores_gemma":[0.9005958,0.0001673169,0.09676307,0.0002096693,0.00004706081,0.0001837814,0.0002766965,0.0002327053,0.001523871],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003673008,"threshold_uncertainty_score":0.01665914,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007528728385960206,"score_gpt":0.225826501456591,"score_spread":0.2182977730706308,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}