{"id":"W4410556104","doi":"10.1145/3716863.3718040","title":"Distributionally Robust Statistical Verification with Imprecise Neural Networks","year":2025,"lang":"en","type":"article","venue":"","topic":"Fault Detection and Control Systems","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Army Research Office; Multidisciplinary University Research Initiative; University of Pennsylvania; U.S. Department of Defense; National Science Foundation","keywords":"Computer science; Artificial neural network; Artificial intelligence; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006038577,0.001281274,0.001419218,0.0009090776,0.0007014424,0.001775621,0.002311182,0.001498609,0.001179128],"category_scores_gemma":[0.03268533,0.0008700009,0.001016387,0.0006158582,0.003186662,0.004803526,0.003963488,0.004104817,0.0002286157],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002304406,"about_ca_system_score_gemma":0.00195275,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003399812,"about_ca_topic_score_gemma":0.003253397,"domain_scores_codex":[0.9950552,0.001342709,0.0002674335,0.001043248,0.001909869,0.0003814847],"domain_scores_gemma":[0.9758958,0.01529976,0.002948829,0.003534716,0.001930199,0.0003906204],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001060086,0.00002321889,0.0007026703,0.00003060171,0.0000339928,0.00004256002,0.00004380249,0.9658747,0.00254577,0.01323326,0.0001717198,0.01719176],"study_design_scores_gemma":[0.000002468842,0.00001114834,0.00006260358,0.000002735507,0.000001916185,0.000005891248,0.000002896246,0.9904341,0.001080946,0.008348932,0.00004227669,0.000004149278],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02159433,0.0000737963,0.9768892,0.0001852671,0.00001593656,0.00001997727,0.00003397724,0.0006024982,0.000584932],"genre_scores_gemma":[0.9080102,0.00008848587,0.09054238,0.0001303384,0.00002995639,0.00008610413,0.00009475773,0.0001333485,0.0008843218],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006038577,"threshold_uncertainty_score":0.03193545,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.004521220126431393,"score_gpt":0.1967482310758849,"score_spread":0.1922270109494535,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}