From a6f9421be944eb75f50dcaebfaf37bede290faeb Mon Sep 17 00:00:00 2001 From: GoodNightIsabelle Date: Thu, 17 Sep 2026 17:53:05 -0400 Subject: [PATCH 1/3] Add the general preprocessing framework to logparser/utils and uploaded code for PreDrain (Drain with the preprocessing framework enabled). --- .DS_Store | Bin 0 -> 10244 bytes logparser/.DS_Store | Bin 0 -> 14340 bytes logparser/PreDrain/.DS_Store | Bin 0 -> 6148 bytes logparser/PreDrain/PreDrain.py | 95 +++++++++++++++++++ logparser/PreDrain/README.md | 94 +++++++++++++++++++ logparser/PreDrain/__init__.py | 1 + logparser/PreDrain/benchmark.py | 155 +++++++++++++++++++++++++++++++ logparser/PreDrain/demo.py | 15 +++ logparser/utils/preprocessing.py | 79 ++++++++++++++++ 9 files changed, 439 insertions(+) create mode 100644 .DS_Store create mode 100644 logparser/.DS_Store create mode 100644 logparser/PreDrain/.DS_Store create mode 100644 logparser/PreDrain/PreDrain.py create mode 100644 logparser/PreDrain/README.md create mode 100644 logparser/PreDrain/__init__.py create mode 100644 logparser/PreDrain/benchmark.py create mode 100644 logparser/PreDrain/demo.py create mode 100644 logparser/utils/preprocessing.py diff --git a/.DS_Store b/.DS_Store new file mode 100644 index 0000000000000000000000000000000000000000..e87075fab67f5218d3f3d45b3250648249fbf572 GIT binary patch literal 10244 zcmeHN&2Jk;6n~RAc-^#Q(w?K$^m7Ua)Vn4$`co1xdSt zA_of*sz`+@kyi{N)Nwv5BEatM zDVT*OI1s7#cXR)zS5;n%mP$33)^Lt(HS@;o!-al+<5<4Gee^b9cK8!Py0HuRAT8~M@eAv7$d>D}G^JIvW>W|D-1}s7uR$&2a8J6H?*`;+EGfOK)th0aff<1;? z11yNZiphOFOiqoEeD%Wbs*NGB9j?)$iBTQ$)p*D^fA;*NP=~PJKXNg_KQ5>G2c!4D zlm2nuG5$&ZQ2#hz^$$`7=3g3YxQht3zy}xXn3tcSgo<`dy3bb_{d~~g5RYSahTs~L z>hoS4QK#SreoGeN8r*~mwzC*Vqd$fbc1Wo)WH3&Od(FqF0|+txAf^%gyGCs<^yRnm z`P430 z*#TE@8a9gDe%J9DbgxL;UL)kX6~QnvMy6_>?{wz#`ODcWbNPeI+0JV(V}Jh2D+dP| zV>nqhAx8>1~F$ye19@cP1j1SvSu$O>^Q{R=5kzaa|{briI*fOWaPoH^m!kj#N zZgOgJditrS&p-3*h376*&2u%YzS9ht&x_QhLD6m7x0->|sQ6ah@tW(AeK)E&_w1^9 zHYU1aH{L&*THW_iX6k5yn|m{`ZI9bq+S&^3XkF=VRLzO79#Gd^rlAu#)Z4fnBI4!9 zZk4?)iYx9S+DkrTg==;YA|23K->`zf@|t#Wf77*Uw!7pNX%x}cx>Ki~@O*-s#o=;b zGn2Wie3t{{3fBd)xx$R-n2_sXohz(~t})e{l9c2g)F1Hvsz*ILlPfF>P6jjlCCq|v z!y4?sJ@^#9fcx+RJcM824|q({#3UEVB{EOmAXmv-ck~?NI*Km#7OUxe_o7x zE%^|837@>Wn`gd#^f*~6wxd6i`9Hy4Y*!y6sqXBVAGx0i_I4|Q=TfhUd=^;-R&d8J zKnXEQ-1+>h_)%o`DA!KgnE06RleeS>F<2=#r`@L|7S zp67H`ehjVJsDmrSBw1b^BB_1~c)b0El8T*v^thj3AAk|?)cz&ME8%@U@R_3nw+})dYXf)1fi-g&3m6L+3m6L+3m6Oh zGYjx{FCM9$pfp!w0b>DUftCe$e8}L*%bYCdgHnVJyznJFZ|6&RabO+u0C6uP%bYCd zgHm+GKB6m@L02YI3>I|9`xy?$GAGOVpbWYL3%Uc7BQu$yFflsz%Lq8ITu_>;v4F8a z)B=0(mX>>0h9bGQhN({v7Rp|&9*Fyn7om&-*iapIyt0(R-SJa&$48aSonW1+D(lsK zx-7nzpzt{%_5!r=Y98Sni;3Z!gi+jc7=dSCm@;NtHO2~(@MT*v$X99A*akZn8iH_* zK@g@1J^+sJ9LxkeBpTD}5vDC9VG36ls2q)HXftJ`Hda*-gei#UAO+*N2V}vJ*a@=1 zrF;_@w%JF@vk<|T<{EW5-?vJ2$8I#U*&kV=qcizHSE?)3lR8tF_T~!BVsoaEFZ!qW zuL;j{N@2ZNI&NEYQ|aDAp4%+iuH}mkcEv)-=@YhF@#ga0j9Y1lzJ;J6)t%~|N^iV$ zX?HfeBeQETdwECZ(%zj|-@R+k<;&fvzW&GeADcSw)Lid%J|(D!d=z8yo)U10|&+0V~u1&97zhPbPy1u@?haTRzY4heSQ|S$*V)=a4=Zp_| zj_2o{8S7Nlw=0Euv2455$)@#6vtVDerqb&*=rODE;?k{&3w6ZwF5M8~p7kxu74pW; zooiUlNu_>rD!sN*_C3cL_ZoK7_S}=F8))z0rd1nu&v`iFEZTak&YK#?Ex&8UM z&UUk>*#UNh4YLV^?P1TUCAgONGzM*ndt*~V&th7c_^@&GN zO+;AfM2;hH2qrY1z83ykIKq^b02hJ4$a z1Uzm@i6k81$_*-~4NIl}GM1Dh5T-$Z?ttgflH(dn{)$$^vP4maKDVN5X~VA78f!t> zN?F|u>^B2VTsk<=pv=UpSY?*P7okM{=dX&rgF7;VJD>QMuL?yY^OE@gz9f`srf&7v zAHwllkG1MrZ@>BV75QRN`m2r~kDxA=(_Bn@4XlD$3|&4n&;dPgVE_Zv1%ROw*`2<-2+qjhq)S_O;L}D0FnQ{ci?jRFpSVR!E=4`UwVn65pRw!8fA+b z?eK26*lM-5SxAVnX_Q;U?zj-k$7h~=AwifaNXqVmuc(|h43++BJjTAE4hn)Y{S`sD zV={#Ho^6-a6RrfSvfZ}u`CGayjc>TaKxAl)V=*e{ZSQ9}=j3vEJz=aMRlbn==`|{+ z4O^vu8?FjMWeVN7fax&qMDTA)M(Gopz?4B_+Cmbh@Wp^&+G^F9F3qxnh<#dcjn#6}BwHO0>q>OO8!h z-9e$SN<4d2Oj#XFSQ;2tafPLE#5EPMTx=i0m|g#~ootPN=+UtqR|HRVf(ia?L@=5@;d2m!i=CMWj_3gkI3#q*IT{u3K17ihueSuzp|vvXvPA@ z0xN0();Uxd!jwV$PdIL5p3r6RJc}pav{=NGA6OTA7Kd2=n6L;2Q8FUttZS5l$Hbt0Y;eRqe=|H~I{J$ynR}sg-r#`K z`M=)m+GVqA8%78{70c{hg&J^y25aUl@rXSZc*a)OJ;)1JgS>OBn72opjeZ+M7r155 zYt*b4oZlgrHIgrJIyW|5oB?OR8E^)ifgc#)nJqFY483*+oB?OxivigmB8y;|urt)H zgOyx7#V^-r73y+X7F$eMChQF9p#)1MT53p7j9}@s#}rp4>": numOfPar += 1 + return 1, numOfPar + + for token1, token2 in zip(seq1, seq2): + if token1 == '<*>': + numOfPar += 1 + continue + if token1 == token2: + simTokens += 1 + + retVal = float(simTokens) / len(seq1) + + return retVal, numOfPar + + def parse(self, logName): + print('Parsing file: ' + os.path.join(self.path, logName)) + start_time = datetime.now() + self.logName = logName + rootNode = Node() + logCluL = [] + + self.load_data() + + count = 0 + matched_types = [] + use_sequence = [] + for idx, line in self.df_log.iterrows(): + logID = line['LineId'] + + # Add preprocess + if idx < self.estimate: + logmessageL, sequence, matched = preprocess(line['Content'], estimation_stage=True) + logmessageL = logmessageL.strip().split() + matched_types.extend(matched) + use_sequence = sequence + else: + if idx == self.estimate: + matched_types = set(matched_types) + use_sequence = [i for i in use_sequence if i in matched_types] + logmessageL = preprocess(line['Content'], estimation_stage=False, use_sequence=use_sequence).strip().split() + + matchCluster = self.treeSearch(rootNode, logmessageL) + + #Match no existing log cluster + if matchCluster is None: + newCluster = Logcluster(logTemplate=logmessageL, logIDL=[logID]) + logCluL.append(newCluster) + self.addSeqToPrefixTree(rootNode, newCluster) + + #Add the new log message to the existing cluster + else: + newTemplate = self.getTemplate(logmessageL, matchCluster.logTemplate) + matchCluster.logIDL.append(logID) + if ' '.join(newTemplate) != ' '.join(matchCluster.logTemplate): + matchCluster.logTemplate = newTemplate + + count += 1 + if count % 1000 == 0 or count == len(self.df_log): + print('Processed {0:.1f}% of log lines.'.format(count * 100.0 / len(self.df_log))) + + + if not os.path.exists(self.savePath): + os.makedirs(self.savePath) + + self.outputResult(logCluL) + + print('Parsing done. [Time taken: {!s}]'.format(datetime.now() - start_time)) diff --git a/logparser/PreDrain/README.md b/logparser/PreDrain/README.md new file mode 100644 index 00000000..b86220ab --- /dev/null +++ b/logparser/PreDrain/README.md @@ -0,0 +1,94 @@ +# Drain + "Preprocessing is All You Need" + +## What Are the New Framework Features? + +Getting tired of low parsing accuracies? Our log preprocessing framework is here to save your day! Go to ```./benchmark/logparser/utils/preprocessing.py``` to check the implementation details. + +### More Regexes +Our study identified several categories of variables that are not matched by the default Loghub regexes but can be identified using **consistent and generalizable** regexes. Therefore, we enriched the regex set used for log preprocessing. The regexes used in our new framework are introduced in the following table: + +| Semantic | Regex | Introduction | +|----------------|---------------------------------------------------------------------------------------------------------------|------------------------------------------------------------------------| +| IPv4_port | r'(/\|)(\d+\.){3}\d+(:\d+)?' | IPv4 addresses (optional: with port). | +| host_port | r'([\w-]+\.)+[\w-]+\:\d+' | Domain host names with port. | +| package_host | r'([\w-]+\.){2,}[\w-]+(\$[\w-]+)*(\@[\w-]+)?' | Package (optional: with port and node)/Domain host names without port. | +| Mac_address | r'^([0-9A-Fa-f]{2}[:-]){5}([0-9A-Fa-f]{2})$' | MAC addresses. | +| IPv6 | r'(([0-9a-fA-F]{1,4}:){7}([0-9a-fA-F]{1,4}\|:)\|(([0-9a-fA-F]{1,4}:){1,7}\|:):((:[0-9a-fA-F]{1,4}){1,7}\|:))' | IPv6 addresses. | +| path | r'(/\|)(([\w.-]+\|\<\*\>)/)+([\w.-]+\|\<\*\>)' | File paths. | +| size | r'\b\d+\.?\d*\s?([KGTMkgtm]?(B\|b)\|([KGTMkgtm]))\b' | Memory sizes. | +| duration | r'\b\:<\*>" instead of the correct form "<\*>." Therefore, we carefully organized the detection sequence as: +``` +'url', 'IPv4_port', 'host_port', 'package_host', 'IPv6', 'Mac_address', 'time', 'path', 'block', 'date', 'duration', 'size', 'numerical', 'weekday_months' +``` + +### Customizable Masks +Our framework allows users to customize the masks for variables. For example, an IPv4 address with port can be masked as either the finegrained "<\*>:<\*>" or the standard form "<\*>". The framework leverages "<\*>" for default parsing, but customizable masks can be managed using the ```regex_map``` dictionary and enabled in parsers. + +### Easy Knowledge Management +Have some domain specific regexes in your mind? Add it to the regex set! Update the ```regex_match``` dictionary and ```sequence``` list to preprocess your log. + +## Know Your Targets (the variables) +Loghub provides various regexes for log preprocessing. These regexes were selected based on the domain knowledge for each system. We summarized the regexes and de-duplicated them as follows: +![image](./plots/default-regex.png?raw=true) + +According to RQ1, we found that using all these default regexes is insufficient for variable detection in the preprocessing stage. Hence, we carried out a study on the non-matchable variables from Loghub-2k and manually categorized them. The two authors independently labeled a small subset and discussed the category range. Leveraging the range, the two authors then labeled the remaining variables independently and discussed the final labels. The labeled variables can be checked at ``not_matched_variables.csv``. A summary of the variable types and their ratios is available here: + +![image](./plots/non-matchable.png?raw=true) + +Our framework aims to reduce the not-matching number of generalizable (i.e., not customized or system-specific) variables (e.g., IPv6 addresses). + +## Dataset +We used the smaller-scale dataset ``Loghub-2k`` for variable extraction and categorization; the framework is developed based on the findings in this dataset. To replicate the log parsing process and test the generalizability of our findings, we used the ``Loghub 2.0`` dataset for framework impact evaluation. The two datasets contain labeled log messages from 14 different systems. Both 2k and the full 2.0 version log data, along with their detailed introductions, can be found at https://github.com/logpai/loghub-2.0. + +## Parsing Tools +Our work focuses on improving the performance of statistic-based parsers with **manageable, interpretable, and generalizable** knowledge provided in the preprocessing stage. According to the Loghub 2.0 results, only four statistic-based log parsers (i.e., Drain, IPLoM, LFA, and LogCluster) can parse all the full-sized log files in 12 hours. Considering the applicability of these four tools in real-life usage, we only evaluated them in our study. The implementation codes are inherited from the Loghub 2.0 repository. + + + +## Replicate the Results +Result replication is made easy! + +### Overall Performance +Run the following commands to obtain the parsing result and evaluations (GA, PA, FGA, FTA) on all log messages: + +``` +cd benchmark/ +./run_all_full.sh +``` + +We illustrate the evaluation results of the four statistic-based parsers in the following box plot. The blue boxes indicate the parsers with the original preprocessing function, while the yellow boxes show the results of parsers with the new preprocessing framework. The red lines show the medians and the green arrows indicate the means. + +![image](./plots/comparison_full.png?raw=true) + +### Performance on Different Complexity Subgroups +The log messages are divided into three subgroups according to the number of variables in the message: ``#Param=0 (complexity=1)``, ``0<#Param<5 (complexity=2)``, and ``#Param>=5 (complexity=3)``. Run the following commands to obtain the parsing result and evaluations (GA, PA, FGA, FTA) on log messages in different subgroups: + +``` +cd benchmark/ +./run_complexity_full.sh +``` + +The following plot visualizes the average evaluation results of log parsers on logs with different numbers of variables. The red dot lines illustrate the original results obtained with the previous preprocessing function. + +![image](./plots/complexity_full_all.png?raw=true) + +### Performance on Different Frequency Subgroups +We extract the messages with the most frequent 10\% and the least frequent 10\% templates and evaluate the impact brought by our framework. Run the following commands to obtain the parsing result and evaluations (GA, PA, FGA, FTA) on log messages in different subgroups: + +``` +cd benchmark/ +./run_frequency_full.sh +``` + +The following plot visualizes the average evaluation results of log parsers on logs with different frequencies (i.e., the most frequent 10% and the least frequent 10%.) The red dot lines illustrate the original results obtained with the previous preprocessing function. + +![image](./plots/frequency_full_all.png?raw=true) diff --git a/logparser/PreDrain/__init__.py b/logparser/PreDrain/__init__.py new file mode 100644 index 00000000..6a77e624 --- /dev/null +++ b/logparser/PreDrain/__init__.py @@ -0,0 +1 @@ +from .PreDrain import * \ No newline at end of file diff --git a/logparser/PreDrain/benchmark.py b/logparser/PreDrain/benchmark.py new file mode 100644 index 00000000..9e89f4d8 --- /dev/null +++ b/logparser/PreDrain/benchmark.py @@ -0,0 +1,155 @@ +# ========================================================================= +# Copyright (C) 2016-2023 LOGPAI (https://github.com/logpai). +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. +# ========================================================================= + + +import sys +sys.path.append("../../") +from logparser.PreDrain import LogParser +from logparser.utils import evaluator +import os +import pandas as pd + + +input_dir = "../../data/loghub_2k/" # The input directory of log file +output_dir = "PreDrain_result/" # The output directory of parsing results + + +benchmark_settings = { + "HDFS": { + "log_file": "HDFS/HDFS_2k.log", + "log_format": "