{"id":"W4409262526","doi":"10.1109/wacv61041.2025.00614","title":"Assessing Visually-Continuous Corruption Robustness of Neural Networks Relative to Human Performance","year":2025,"lang":"en","type":"article","venue":"","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Robustness (evolution); Computer science; Artificial neural network; Artificial intelligence; Language change","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001234273,0.00007720279,0.0001168225,0.0001170311,0.0001934539,0.0001034401,0.0003483676,0.00005292553,0.000008050795],"category_scores_gemma":[0.000006756216,0.00007194957,0.00003952867,0.0005562673,0.00002628458,0.000591182,0.0001639059,0.0001057929,0.000001356028],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003095356,"about_ca_system_score_gemma":0.0000154259,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000248021,"about_ca_topic_score_gemma":0.000006704228,"domain_scores_codex":[0.9993196,0.0000242553,0.0002245819,0.0002189541,0.00008400497,0.0001285911],"domain_scores_gemma":[0.9994335,0.00003243584,0.00008479397,0.0002935258,0.0001231673,0.00003255164],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000009149756,0.0001851753,0.00569248,0.00003777597,0.00002765643,8.165304e-7,0.0002099254,0.1372223,0.007907534,0.3267931,0.001706057,0.5202081],"study_design_scores_gemma":[0.00006947717,0.00007812459,0.02918643,0.00003038445,0.000004658986,0.000001521463,0.00002727968,0.9653037,0.004654657,0.0002492708,0.0003035051,0.00009097655],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2021381,0.000006660046,0.7920047,0.0001974276,0.00007771749,0.0001598312,1.339553e-7,0.0002036129,0.005211774],"genre_scores_gemma":[0.9595791,0.000001842284,0.03891303,0.000152407,0.00002315728,0.00004002904,0.000001349145,0.000003383422,0.001285747],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8280815,"threshold_uncertainty_score":0.2934018,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01966534667968344,"score_gpt":0.3176014962214103,"score_spread":0.2979361495417268,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}