{"id":"W2784152338","doi":"10.1109/icmla.2017.0-122","title":"Cybersecurity Automated Information Extraction Techniques: Drawbacks of Current Methods, and Enhanced Extractors","year":2017,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Science and Technology Directorate; Ministère de la Défense Nationale; UT-Battelle; Battelle; U.S. Department of Homeland Security; U.S. Department of Energy","keywords":"Overfitting; Computer science; Leverage (statistics); Information extraction; Relationship extraction; Identification (biology); Information retrieval; Domain (mathematical analysis); Precision and recall; Recall; Artificial intelligence; Natural language processing; Data mining","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01683696,0.002779928,0.001968925,0.00851154,0.001554374,0.006449224,0.003495842,0.001940362,0.002726944],"category_scores_gemma":[0.03664212,0.001496774,0.001735804,0.006730645,0.002090166,0.0134193,0.003870948,0.003333829,0.005895681],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00110318,"about_ca_system_score_gemma":0.002468753,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004174157,"about_ca_topic_score_gemma":0.006395507,"domain_scores_codex":[0.9878293,0.004911309,0.0009521937,0.002387384,0.003603932,0.0003159112],"domain_scores_gemma":[0.9455993,0.03284945,0.002227677,0.01262589,0.006318873,0.0003788474],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002256436,0.0003391168,0.008126525,0.001815115,0.0003628997,0.0001723149,0.00133688,0.01465029,0.01083463,0.009419341,0.01244526,0.940272],"study_design_scores_gemma":[0.0001454527,0.0007004191,0.01579357,0.002159928,0.0008112755,0.002717592,0.002640588,0.6384476,0.08006972,0.07981417,0.1762194,0.0004802834],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04309054,0.01469939,0.9197701,0.00459776,0.0003704132,0.0003609709,0.001622137,0.008851316,0.006637445],"genre_scores_gemma":[0.1889296,0.01210978,0.7815891,0.001097385,0.0007574336,0.0004510715,0.005481049,0.001181179,0.008403362],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01683696,"threshold_uncertainty_score":0.08904338,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0255260958909049,"score_gpt":0.4003089176312616,"score_spread":0.3747828217403567,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}