{"id":"W4281764099","doi":"10.36227/techrxiv.19498769","title":"Detecting Anomalies in Logs by Combining NLP features with Embedding or TF-IDF","year":2022,"lang":"en","type":"preprint","venue":"","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Lakehead University","funders":"Lakehead University","keywords":"Categorization; Computer science; Artificial intelligence; Natural language processing; Set (abstract data type); Sentence; Focus (optics); Embedding; Information retrieval; Text categorization; Text mining; Meaning (existential); Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001303915,0.0009155361,0.0009038695,0.005964607,0.0004203129,0.001495813,0.0007323815,0.0009859613,0.00181708],"category_scores_gemma":[0.005958153,0.0002119262,0.0006287939,0.003386977,0.0003953767,0.002880569,0.000634282,0.0008419168,0.002522335],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005310863,"about_ca_system_score_gemma":0.0004395793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003626038,"about_ca_topic_score_gemma":0.00373132,"domain_scores_codex":[0.998784,0.0002032213,0.0001596728,0.0003224575,0.0004166831,0.0001139723],"domain_scores_gemma":[0.9962626,0.001769161,0.0006060736,0.000487807,0.0007581323,0.0001161279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000487869,0.0005017424,0.02933285,0.0003423805,0.0001153043,0.0004001868,0.0002972065,0.01476639,0.05154327,0.001771332,0.01009461,0.8903468],"study_design_scores_gemma":[0.00003373821,0.0003313698,0.03393065,0.00007158786,0.0000984998,0.0009393088,0.0004716858,0.9102997,0.03155661,0.01085592,0.01131746,0.00009357855],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2028954,0.001174038,0.7718983,0.0009181168,0.0004043562,0.0003033454,0.005315013,0.01357318,0.003518203],"genre_scores_gemma":[0.6097371,0.0005038973,0.3775331,0.0001356321,0.000300424,0.0002185312,0.008570049,0.0004155151,0.00258581],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005964607,"threshold_uncertainty_score":0.007209897,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0231594513300244,"score_gpt":0.2890592490640066,"score_spread":0.2658997977339822,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}