{"id":"W4411337774","doi":"10.1109/cain66642.2025.00027","title":"Debugging and Runtime Analysis of Neural Networks with VLMs (A Case Study)","year":2025,"lang":"en","type":"article","venue":"","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Debugging; Computer science; Artificial neural network; Programming language; Software bug; Software engineering; Artificial intelligence; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002410083,0.0009309425,0.0003910588,0.0006313109,0.0003382686,0.0006584759,0.001433675,0.0008499099,0.0008918848],"category_scores_gemma":[0.01557672,0.0003423307,0.0005056397,0.0003681312,0.0009278017,0.001382639,0.0009036977,0.001298193,0.0001528094],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001288849,"about_ca_system_score_gemma":0.0009329853,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005430921,"about_ca_topic_score_gemma":0.005814601,"domain_scores_codex":[0.9981074,0.0006417277,0.0001002623,0.0004814578,0.000508427,0.0001606933],"domain_scores_gemma":[0.9904953,0.006164314,0.0006419819,0.001581307,0.0009029986,0.0002140043],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001053094,0.0003518401,0.04070496,0.0005114385,0.0001966641,0.002777161,0.001087835,0.7359927,0.02825444,0.01058236,0.007123091,0.1713644],"study_design_scores_gemma":[0.00001995311,0.0001132864,0.001828356,0.00001868116,0.00001767016,0.0001652614,0.00007414546,0.9778415,0.0153857,0.003836347,0.0006872612,0.00001186414],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8039297,0.000565562,0.1770923,0.0008646169,0.0001026491,0.000109206,0.00068874,0.01487901,0.001768095],"genre_scores_gemma":[0.9523958,0.00005378639,0.04637644,0.00008556732,0.000007871034,0.00004035097,0.0003386927,0.0002631595,0.0004383556],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005430921,"threshold_uncertainty_score":0.01274592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.007689746056186294,"score_gpt":0.2534756331331594,"score_spread":0.245785887076973,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}