{"id":"W4410553371","doi":"10.1109/saner64311.2025.00036","title":"Preprocessing is All You Need: Boosting the Performance of Log Parsers with a General Preprocessing Framework","year":2025,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Preprocessor; Boosting (machine learning); Computer science; Parsing; Artificial intelligence; Data pre-processing; Natural language processing; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007262018,0.002730573,0.001522918,0.003019805,0.001472849,0.003561099,0.004379265,0.002064173,0.003838231],"category_scores_gemma":[0.02874095,0.00147096,0.00209579,0.00298706,0.001837193,0.01098013,0.004122335,0.004116788,0.003717286],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001776649,"about_ca_system_score_gemma":0.00665593,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008305973,"about_ca_topic_score_gemma":0.0110469,"domain_scores_codex":[0.9932839,0.002134513,0.0006100654,0.001544951,0.00183371,0.0005929598],"domain_scores_gemma":[0.9807478,0.009303783,0.001032559,0.005502968,0.002995459,0.0004173561],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001635299,0.001496326,0.01895116,0.001530637,0.0002744962,0.001016424,0.001485289,0.05003772,0.0591254,0.01828813,0.1093255,0.7368336],"study_design_scores_gemma":[0.000426141,0.0009305406,0.01077794,0.000211782,0.0004142075,0.001190731,0.0007478708,0.7383482,0.118254,0.04291625,0.08541699,0.0003653106],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.08598658,0.001950374,0.5534464,0.003221719,0.0004194681,0.0006915457,0.004488226,0.3426127,0.007183073],"genre_scores_gemma":[0.2949514,0.0007789776,0.6636758,0.002581176,0.0001838363,0.0005553975,0.01674922,0.01552154,0.005002645],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008305973,"threshold_uncertainty_score":0.03840572,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01189127894174018,"score_gpt":0.2776228799812636,"score_spread":0.2657316010395234,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}