{"id":"W4289313429","doi":"10.1186/s42400-022-00119-8","title":"On building machine learning pipelines for Android malware detection: a procedural survey of practices, challenges and opportunities","year":2022,"lang":"en","type":"article","venue":"Cybersecurity","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University; Huawei Technologies (Canada)","funders":"Huawei Technologies","keywords":"Computer science; Malware; Android (operating system); Software deployment; Malware analysis; Data science; Process (computing); Best practice; Machine learning; Artificial intelligence; Computer security; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01203947,0.002195288,0.001226265,0.01036283,0.001035036,0.004429314,0.003717204,0.002280375,0.002542744],"category_scores_gemma":[0.04292204,0.001318234,0.002028049,0.006474081,0.001611703,0.01202737,0.002458909,0.003665673,0.003718395],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001548636,"about_ca_system_score_gemma":0.002947903,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003498658,"about_ca_topic_score_gemma":0.002723922,"domain_scores_codex":[0.9922726,0.002663623,0.000904288,0.001707047,0.002141292,0.0003111562],"domain_scores_gemma":[0.9551897,0.03093001,0.00129035,0.002640777,0.009529358,0.0004198175],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.00007093672,0.0001741792,0.006888138,0.006249937,0.0002377958,0.0001935589,0.0008052895,0.009382316,0.002488559,0.01859726,0.01494509,0.9399669],"study_design_scores_gemma":[0.00009607479,0.001320385,0.01363955,0.01842923,0.001032109,0.002466225,0.003222116,0.3459691,0.02556152,0.1327712,0.4549897,0.0005028374],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"review","genre_scores_codex":[0.01837214,0.4107502,0.5377953,0.01430661,0.0008738553,0.0003896061,0.000661884,0.004037711,0.01281255],"genre_scores_gemma":[0.1752482,0.3615687,0.4485833,0.004146404,0.001804906,0.000534922,0.00275767,0.0007766894,0.004579199],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.9879605,"threshold_uncertainty_score":0.06367153,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0929005616227369,"score_gpt":0.3118799947807046,"score_spread":0.2189794331579677,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}