Skip to content
Next Next commit
Updates for Drain scripts to make it work with Python 3.9.12
  • Loading branch information
moesjo committed Aug 16, 2022
commit 382ac0e6d7ee20b0effb6ae54b82710bc28f266d
2 changes: 1 addition & 1 deletion benchmark/Drain_benchmark.py
Original file line number Diff line number Diff line change
Expand Up @@ -141,7 +141,7 @@
}

bechmark_result = []
for dataset, setting in benchmark_settings.iteritems():
for dataset, setting in benchmark_settings.items():
print('\n=== Evaluation on %s ==='%dataset)
indir = os.path.join(input_dir, os.path.dirname(setting['log_file']))
log_file = os.path.basename(setting['log_file'])
Expand Down
2 changes: 1 addition & 1 deletion logparser/Drain/Drain.py
Original file line number Diff line number Diff line change
Expand Up @@ -336,7 +336,7 @@ def get_parameter_list(self, row):
template_regex = re.sub(r"<.{1,5}>", "<*>", row["EventTemplate"])
if "<*>" not in template_regex: return []
template_regex = re.sub(r'([^A-Za-z0-9])', r'\\\1', template_regex)
template_regex = re.sub(r'\\ +', r'\s+', template_regex)
template_regex = re.sub(r'\\ +', r'\\s+', template_regex)
template_regex = "^" + template_regex.replace("\<\*\>", "(.*?)") + "$"
parameter_list = re.findall(template_regex, row["Content"])
parameter_list = parameter_list[0] if parameter_list else ()
Expand Down
8 changes: 4 additions & 4 deletions logparser/utils/evaluator.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@
import sys
import pandas as pd
from collections import defaultdict
import scipy.misc
import scipy.special


def evaluate(groundtruth, parsedresult):
Expand Down Expand Up @@ -58,13 +58,13 @@ def get_accuracy(series_groundtruth, series_parsedlog, debug=False):
real_pairs = 0
for count in series_groundtruth_valuecounts:
if count > 1:
real_pairs += scipy.misc.comb(count, 2)
real_pairs += scipy.special.comb(count, 2)

series_parsedlog_valuecounts = series_parsedlog.value_counts()
parsed_pairs = 0
for count in series_parsedlog_valuecounts:
if count > 1:
parsed_pairs += scipy.misc.comb(count, 2)
parsed_pairs += scipy.special.comb(count, 2)

accurate_pairs = 0
accurate_events = 0 # determine how many lines are correctly parsed
Expand All @@ -82,7 +82,7 @@ def get_accuracy(series_groundtruth, series_parsedlog, debug=False):
print('(parsed_eventId, groundtruth_eventId) =', error_eventIds, 'failed', logIds.size, 'messages')
for count in series_groundtruth_logId_valuecounts:
if count > 1:
accurate_pairs += scipy.misc.comb(count, 2)
accurate_pairs += scipy.special.comb(count, 2)

precision = float(accurate_pairs) / parsed_pairs
recall = float(accurate_pairs) / real_pairs
Expand Down