From 41a30f818f4d48bf57c08ec8228ecee5f5177b09 Mon Sep 17 00:00:00 2001 From: "angre.garcia-gomez@ait.ac.at" Date: Thu, 27 Aug 2026 16:42:08 +0200 Subject: [PATCH 01/25] add cleaning method --- src/detectmateperformance/_core/aux.cpp | 16 ++++++++++++---- tests/test_c/test_type.cpp | 15 +++++++++++++++ 2 files changed, 27 insertions(+), 4 deletions(-) diff --git a/src/detectmateperformance/_core/aux.cpp b/src/detectmateperformance/_core/aux.cpp index 6c43933..a681ee5 100644 --- a/src/detectmateperformance/_core/aux.cpp +++ b/src/detectmateperformance/_core/aux.cpp @@ -1,6 +1,17 @@ +#include + #include "aux.h" +std::string clean_string(const std::string& input) { + std::regex pattern( + "[!\"#$%&'()*+,/:;<=>?@\\[\\]^`{|}~\\s]|\\.(?![a-zA-Z0-9])|-(?![a-zA-Z0-9])" + ); + std::string result = std::regex_replace(input, pattern, " "); + + return result; +} + bool do_split(const char* str) { return *str == ' '; } @@ -14,14 +25,11 @@ void remove_empty(std::deque& words) { std::deque preprocessing(std::string message) { std::deque words; + message = clean_string(message); const char* start = message.data(); const char* end = start; while (*end) { - if (std::ispunct(*end)) { - *const_cast(end) = ' '; - } - if (do_split(end)) { words.emplace_back(start, end); start = end + 1; diff --git a/tests/test_c/test_type.cpp b/tests/test_c/test_type.cpp index 9a1eecc..19eda21 100644 --- a/tests/test_c/test_type.cpp +++ b/tests/test_c/test_type.cpp @@ -14,6 +14,21 @@ TEST(MessagesTest, Preprocessing) { EXPECT_EQ(result.size(), 2); EXPECT_EQ(result[0], "Hello"); EXPECT_EQ(result[1], "world"); + + std::string input2 = "Hello. world"; + auto result2 = preprocessing(input2); + + EXPECT_EQ(result2.size(), 2); + EXPECT_EQ(result2[0], "Hello"); + EXPECT_EQ(result2[1], "world"); + + std::string input3 = "Hello. world 129.131.12.12"; + auto result3 = preprocessing(input3); + + EXPECT_EQ(result3.size(), 3); + EXPECT_EQ(result3[0], "Hello"); + EXPECT_EQ(result3[1], "world"); + EXPECT_EQ(result3[2], "129.131.12.12"); } TEST(TemplatesTest, SizeShape) { From 03f281c2b0b8d22bc6591be15bd08e8143498eb3 Mon Sep 17 00:00:00 2001 From: "angre.garcia-gomez@ait.ac.at" Date: Thu, 27 Aug 2026 16:47:32 +0200 Subject: [PATCH 02/25] modify some extra cases --- src/detectmateperformance/_core/aux.cpp | 2 +- tests/test_c/test_type.cpp | 16 ++++++++++++++++ 2 files changed, 17 insertions(+), 1 deletion(-) diff --git a/src/detectmateperformance/_core/aux.cpp b/src/detectmateperformance/_core/aux.cpp index a681ee5..4e6e8fc 100644 --- a/src/detectmateperformance/_core/aux.cpp +++ b/src/detectmateperformance/_core/aux.cpp @@ -5,7 +5,7 @@ std::string clean_string(const std::string& input) { std::regex pattern( - "[!\"#$%&'()*+,/:;<=>?@\\[\\]^`{|}~\\s]|\\.(?![a-zA-Z0-9])|-(?![a-zA-Z0-9])" + "[!\"#$%&'()*+,:;<=>?@\\[\\]^`{|}~\\s]|\\.(?![a-zA-Z0-9])|-(?![a-zA-Z0-9])" ); std::string result = std::regex_replace(input, pattern, " "); diff --git a/tests/test_c/test_type.cpp b/tests/test_c/test_type.cpp index 19eda21..bcac893 100644 --- a/tests/test_c/test_type.cpp +++ b/tests/test_c/test_type.cpp @@ -29,6 +29,22 @@ TEST(MessagesTest, Preprocessing) { EXPECT_EQ(result3[0], "Hello"); EXPECT_EQ(result3[1], "world"); EXPECT_EQ(result3[2], "129.131.12.12"); + + std::string input4 = "Receiving block blk_-1608999687919862906"; + auto result4 = preprocessing(input4); + + EXPECT_EQ(result4.size(), 3); + EXPECT_EQ(result4[0], "Receiving"); + EXPECT_EQ(result4[1], "block"); + EXPECT_EQ(result4[2], "blk_-1608999687919862906"); + + std::string input5 = "Receiving block /home/linux/home/windows"; + auto result5 = preprocessing(input5); + + EXPECT_EQ(result5.size(), 3); + EXPECT_EQ(result5[0], "Receiving"); + EXPECT_EQ(result5[1], "block"); + EXPECT_EQ(result5[2], "/home/linux/home/windows"); } TEST(TemplatesTest, SizeShape) { From 65612932d8457246dbf2d44bc7472a54836a86d9 Mon Sep 17 00:00:00 2001 From: "angre.garcia-gomez@ait.ac.at" Date: Thu, 27 Aug 2026 17:55:42 +0200 Subject: [PATCH 03/25] working in drain regex --- src/detectmateperformance/drain.py | 18 ++++++++++++++++-- tests/test_python/test_drain.py | 21 +++++++++++++++++++++ 2 files changed, 37 insertions(+), 2 deletions(-) diff --git a/src/detectmateperformance/drain.py b/src/detectmateperformance/drain.py index 7cd0de9..d8a50d6 100644 --- a/src/detectmateperformance/drain.py +++ b/src/detectmateperformance/drain.py @@ -7,9 +7,23 @@ import warnings +def clean_strings(df: pl.DataFrame) -> pl.DataFrame: + return df.with_columns( + pl.col("Content") + .str.replace_all(r'[!"#$%&\'()*+,:;<=>?@\[\\\]^`{|}~\s]', " ") + .str.replace_all(r'\.\s|\s\.', " ") + .str.replace_all(r'-\s|\s-', " ") + .str.replace_all( + r'\b(?:\d{1,3}\.){3}\d{1,3}\b|(?:[a-zA-Z]:\\[^\\]*|/[^\\]*|\.\.[^\\]*|\.\/[^\\]*)\b|(?:\d+\s)+\d+', # noqa: E501 + "VAR" + ) + .str.replace_all(r"\b\d+\b", "VAR") + .str.replace_all(r"\s+", " ") + ) + + def cluster_logs_df(df: pl.DataFrame, depth: int = 2, max_child: int = 10) -> dict[str, list[str]]: - df = df.with_columns(pl.col("Content").str.replace_all(r"[^a-zA-Z0-9\s]", " ")) - df = df.with_columns(pl.col("Content").str.replace_all(r"\b\d+\b", "VAR").str.replace_all(r"\s+", " ")) + df = clean_strings(df) df = df.unique() df = df.insert_column(-1, pl.col("Content").str.split(by=" ").list.len().alias("L")) diff --git a/tests/test_python/test_drain.py b/tests/test_python/test_drain.py index 77cd718..321c566 100644 --- a/tests/test_python/test_drain.py +++ b/tests/test_python/test_drain.py @@ -5,6 +5,27 @@ class TestCommonMethods: + def test_clean_strings(self): + result = drain.clean_strings(pl.DataFrame({"Content": [ + "Hello, world", + "Hello. world", + "Hello. world 129.131.12.12", + "Receiving block blk_-1608999687919862906", + "Receiving block /home/linux/home/windows", + ]}))["Content"].to_list() + + print(result) + expected = [ + "Hello world", + "Hello world", + "Hello world VAR", + "Receiving block blk_-1608999687919862906", + "Receiving block VAR", + ] + print(expected) + + assert result == expected + def test_cluster_logs_df(self): df = pl.DataFrame({"Content": ["hello world, 12 1cia2o1", "Ciao! ? bella bella"]}) From 03e46d1d5f0ea40fcf6f476b9611005707ba75ed Mon Sep 17 00:00:00 2001 From: "angre.garcia-gomez@ait.ac.at" Date: Fri, 28 Aug 2026 09:58:34 +0200 Subject: [PATCH 04/25] correct bugs --- src/detectmateperformance/drain.py | 19 ++++++++++++++----- tests/test_python/test_drain.py | 4 ++-- 2 files changed, 16 insertions(+), 7 deletions(-) diff --git a/src/detectmateperformance/drain.py b/src/detectmateperformance/drain.py index d8a50d6..cc97f12 100644 --- a/src/detectmateperformance/drain.py +++ b/src/detectmateperformance/drain.py @@ -5,20 +5,29 @@ import polars as pl import warnings +import re + + +def replace_patterns(text: str) -> str: + text = re.sub(r'\b(?:\d{1,3}\.){3}\d{1,3}\b', "VAR", text) + text = re.sub(r'(?:[a-zA-Z]:\\.[^\\\-]*|/[^\\\-]*|\.\.[^\\\-]*|\.\/[^\\\-]*)\b', "VAR", text) + text = re.sub(r'(? pl.DataFrame: + df = df.with_columns( + pl.concat_str([pl.lit(" "), pl.col("Content"), pl.lit(" ")]).alias("Content") + ) return df.with_columns( pl.col("Content") .str.replace_all(r'[!"#$%&\'()*+,:;<=>?@\[\\\]^`{|}~\s]', " ") .str.replace_all(r'\.\s|\s\.', " ") .str.replace_all(r'-\s|\s-', " ") - .str.replace_all( - r'\b(?:\d{1,3}\.){3}\d{1,3}\b|(?:[a-zA-Z]:\\[^\\]*|/[^\\]*|\.\.[^\\]*|\.\/[^\\]*)\b|(?:\d+\s)+\d+', # noqa: E501 - "VAR" - ) - .str.replace_all(r"\b\d+\b", "VAR") + .map_elements(replace_patterns) + .str.replace_all(r' \d+ ', " VAR ") .str.replace_all(r"\s+", " ") + .str.strip_chars() ) diff --git a/tests/test_python/test_drain.py b/tests/test_python/test_drain.py index 321c566..4098b24 100644 --- a/tests/test_python/test_drain.py +++ b/tests/test_python/test_drain.py @@ -8,7 +8,7 @@ class TestCommonMethods: def test_clean_strings(self): result = drain.clean_strings(pl.DataFrame({"Content": [ "Hello, world", - "Hello. world", + "Hello. world 123", "Hello. world 129.131.12.12", "Receiving block blk_-1608999687919862906", "Receiving block /home/linux/home/windows", @@ -17,7 +17,7 @@ def test_clean_strings(self): print(result) expected = [ "Hello world", - "Hello world", + "Hello world VAR", "Hello world VAR", "Receiving block blk_-1608999687919862906", "Receiving block VAR", From e54c2adafb77c39c152ff8b023b27894b452350c Mon Sep 17 00:00:00 2001 From: "angre.garcia-gomez@ait.ac.at" Date: Fri, 28 Aug 2026 10:06:27 +0200 Subject: [PATCH 05/25] correct path bug --- src/detectmateperformance/drain.py | 2 +- tests/test_python/test_drain.py | 4 +++- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/src/detectmateperformance/drain.py b/src/detectmateperformance/drain.py index cc97f12..d21f373 100644 --- a/src/detectmateperformance/drain.py +++ b/src/detectmateperformance/drain.py @@ -10,7 +10,7 @@ def replace_patterns(text: str) -> str: text = re.sub(r'\b(?:\d{1,3}\.){3}\d{1,3}\b', "VAR", text) - text = re.sub(r'(?:[a-zA-Z]:\\.[^\\\-]*|/[^\\\-]*|\.\.[^\\\-]*|\.\/[^\\\-]*)\b', "VAR", text) + text = re.sub(r'(?:[a-zA-Z]:\\.[^\\\-]*|/[^\\\-\s]*|\.\.[^\\\-]*|\.\/[^\\\-]*)\b', "VAR", text) text = re.sub(r'(? Date: Fri, 28 Aug 2026 10:25:58 +0200 Subject: [PATCH 06/25] add full pipeline to drain --- src/detectmateperformance/drain.py | 10 ++++++++++ tests/test_python/test_drain.py | 7 +++++++ 2 files changed, 17 insertions(+) diff --git a/src/detectmateperformance/drain.py b/src/detectmateperformance/drain.py index d21f373..9855fd6 100644 --- a/src/detectmateperformance/drain.py +++ b/src/detectmateperformance/drain.py @@ -1,6 +1,8 @@ from detectmateperformance.match_tree import TreeMatcher from detectmateperformance.types_ import LogTemplates +from detectmateperformance.pipeline_op import preprocessing + from detectmateperformance.lib.bind_class import drain_generator import polars as pl @@ -103,3 +105,11 @@ def generate_from_df(self, df: pl.DataFrame) -> TreeMatcher: def generate(self) -> TreeMatcher: return self.generate_from_df(pl.DataFrame({"Content": self.buffer})) + + def __call__( + self, logs: list[str] | pl.DataFrame | str, regex: str = r"(?P.*)" + ) -> TreeMatcher: + if not isinstance(logs, pl.DataFrame): + logs = preprocessing(logs=logs, regex=regex) + + return self.generate_from_df(logs) diff --git a/tests/test_python/test_drain.py b/tests/test_python/test_drain.py index f8c8af8..00dc261 100644 --- a/tests/test_python/test_drain.py +++ b/tests/test_python/test_drain.py @@ -90,3 +90,10 @@ def test_num_cases(self): tree_match = drain_.generate() template = tree_match.match_log("Hello world").get_all_templates()[0] assert "Hello VAR" == template + + def test_drain_pipeline(self): + drain_ = drain.Drain(sim=0.3, max_child=1000, depth=2) + regex = r"type=(?P\w+) msg=audit\((?P