{"id":"W4406153408","doi":"10.1145/3711816","title":"LLM-Powered Static Binary Taint Analysis","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Web Application Security Vulnerabilities","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Ant Group","keywords":"Taint checking; Computer science; Firmware; Static analysis; Vulnerability (computing); Binary number; Scalability; State (computer science); Computer security; Software; Computer hardware; Operating system; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001856388,0.001214481,0.0007967296,0.002632375,0.0007008471,0.002073478,0.001839297,0.0008635251,0.00564243],"category_scores_gemma":[0.009924416,0.0008330283,0.001438762,0.0007966859,0.00156051,0.006390621,0.003891418,0.001682037,0.002601721],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001134188,"about_ca_system_score_gemma":0.002621555,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001964323,"about_ca_topic_score_gemma":0.003846867,"domain_scores_codex":[0.9967225,0.0008223572,0.0001992286,0.0004405309,0.001511071,0.0003043087],"domain_scores_gemma":[0.9937748,0.00208933,0.0006614258,0.002185071,0.00113493,0.0001545265],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0010012,0.0004396152,0.02351482,0.001537367,0.0002767616,0.001181904,0.001165571,0.04632268,0.1139486,0.09358767,0.04852843,0.6684954],"study_design_scores_gemma":[0.00008371367,0.0003180946,0.003101189,0.0002985411,0.0001960141,0.001164522,0.0003119749,0.7323647,0.1024187,0.08550001,0.07401314,0.0002293616],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04539574,0.0008185274,0.8283787,0.0009367223,0.0001985498,0.0001994776,0.001134223,0.1150756,0.007862436],"genre_scores_gemma":[0.5762294,0.0005144847,0.4006619,0.001347962,0.0001775778,0.0002990692,0.002683188,0.01007839,0.008007945],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00564243,"threshold_uncertainty_score":0.01887584,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04508240882382882,"score_gpt":0.3181407358867413,"score_spread":0.2730583270629125,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}