{"id":"W7126218964","doi":"10.1109/raid67961.2025.00047","title":"Evaluating LLM-Based Detection of Malicious Package Updates in npm","year":2025,"lang":"","type":"article","venue":"","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Flagging; Malware; Exploit; Software; Task (project management); Code (set theory); Preprocessor; Key (lock); Cryptovirology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004861303,0.002275834,0.001111432,0.002365753,0.000728344,0.001943811,0.002146194,0.002477842,0.001868252],"category_scores_gemma":[0.02609793,0.0005464621,0.001349979,0.0007358653,0.001264251,0.003778437,0.001938295,0.003042506,0.001874243],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001894471,"about_ca_system_score_gemma":0.00212249,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01444173,"about_ca_topic_score_gemma":0.01716024,"domain_scores_codex":[0.9967414,0.00113089,0.0002411611,0.0009156189,0.0006998403,0.0002711142],"domain_scores_gemma":[0.9848402,0.01087309,0.0007192965,0.001482946,0.001453606,0.0006308373],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002078073,0.002038268,0.09986348,0.001279312,0.0007467236,0.0006309999,0.0006329907,0.5897628,0.01115113,0.003835201,0.02792897,0.260052],"study_design_scores_gemma":[0.00002183157,0.000148802,0.001971232,0.00003279742,0.00002646663,0.00005406416,0.00009559912,0.9929462,0.00244574,0.001354836,0.0008862502,0.00001619651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8342685,0.004672726,0.1109153,0.003910081,0.0008358042,0.000506411,0.004716247,0.03231447,0.00786058],"genre_scores_gemma":[0.9245179,0.0003810304,0.06440493,0.0008721661,0.000125879,0.000152333,0.007261241,0.0004243132,0.001860171],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01444173,"threshold_uncertainty_score":0.02871531,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0246453377315002,"score_gpt":0.3530087305352491,"score_spread":0.3283633928037489,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}