{"id":"W4385958000","doi":"10.32604/jcs.2023.042501","title":"Comparative Analysis of Machine Learning Models for PDF Malware Detection: Evaluating Different Training and Testing Criteria","year":2023,"lang":"en","type":"article","venue":"Journal of Cyber Security","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Malware; Support vector machine; Naive Bayes classifier; Random forest; Machine learning; Exploit; Artificial intelligence; Scripting language; Data mining; Computer security; Operating system","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01244476,0.002614772,0.001855033,0.004620836,0.0008115643,0.001641514,0.001484623,0.001688843,0.00081163],"category_scores_gemma":[0.02605462,0.0003850846,0.001600913,0.002283412,0.0005680911,0.002257765,0.0008301463,0.00156291,0.0005381895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00178127,"about_ca_system_score_gemma":0.001333536,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01258341,"about_ca_topic_score_gemma":0.01043573,"domain_scores_codex":[0.9940177,0.002599904,0.0007425847,0.0008233009,0.001455565,0.0003608279],"domain_scores_gemma":[0.9614922,0.03091628,0.00141671,0.001347929,0.004472176,0.000354674],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002277821,0.001389794,0.06081566,0.001136629,0.0009542769,0.0002519269,0.0002543126,0.5822144,0.002904274,0.001417885,0.004842545,0.3415405],"study_design_scores_gemma":[0.00003337232,0.0007001886,0.00918424,0.0001038233,0.0001621284,0.0001134279,0.0001505754,0.9837001,0.004475092,0.0006468452,0.0006887248,0.00004144726],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8428567,0.01350067,0.1290655,0.001246454,0.0005286008,0.0005740732,0.002407645,0.003229104,0.006591122],"genre_scores_gemma":[0.928001,0.001954947,0.064808,0.0001749179,0.00009755202,0.0002936209,0.00314148,0.0001285204,0.001399894],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01258341,"threshold_uncertainty_score":0.06581497,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1281578408210884,"score_gpt":0.3764114026187327,"score_spread":0.2482535617976443,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}