{"id":"W4361004367","doi":"10.3390/bdcc7020060","title":"MalBERTv2: Code Aware BERT-Based Model for Malware Identification","year":2023,"lang":"en","type":"article","venue":"Big Data and Cognitive Computing","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":49,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Moncton","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Malware; Lexical analysis; Artificial intelligence; Source code; Machine learning; Language model; Cryptovirology; Classifier (UML); Identification (biology); Data mining; Computer security; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005107467,0.001159211,0.0004804451,0.001247977,0.0003965208,0.0008243856,0.001781374,0.0009598407,0.003227888],"category_scores_gemma":[0.002167528,0.0004096805,0.0008994996,0.0005951152,0.0003902259,0.001786994,0.0008562796,0.001496324,0.001918265],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00131966,"about_ca_system_score_gemma":0.00161039,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01394734,"about_ca_topic_score_gemma":0.02540457,"domain_scores_codex":[0.9997661,0.00004041862,0.00001381392,0.00007873234,0.00006599031,0.00003494986],"domain_scores_gemma":[0.9994614,0.0002829238,0.00003870739,0.00006632661,0.0001225587,0.00002807557],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007969364,0.0004330136,0.01148028,0.0003765598,0.0001850401,0.0004883793,0.0002553219,0.6127332,0.01362943,0.01240388,0.03442209,0.3127959],"study_design_scores_gemma":[0.000007812769,0.00003146316,0.0002866039,0.000008856531,0.00001042844,0.00004945344,0.00001174189,0.992417,0.001980793,0.003318218,0.001867631,0.00001000435],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.09944568,0.001220401,0.8347599,0.001347771,0.0003576719,0.0004230023,0.01059297,0.04558212,0.006270534],"genre_scores_gemma":[0.7193713,0.0007318723,0.2394696,0.0006465837,0.0001088771,0.0008804963,0.02164805,0.001404935,0.01573817],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01394734,"threshold_uncertainty_score":0.02773231,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1500337647012573,"score_gpt":0.3537749537692224,"score_spread":0.2037411890679651,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}