From 743e139888b1cb3e79143d3ec326dcd6a5fd534e Mon Sep 17 00:00:00 2001 From: Raphael Boidol Date: Fri, 25 Sep 2026 10:41:44 +0200 Subject: [PATCH] Speed up get_tokens: use findall on capture-free regex Replace re.split with capture groups with a plain findall on a capture-free pattern. The alternatives are mutually exclusive so the token list is identical. Signed-off-by: Raphael Boidol --- src/license_expression/_pyahocorasick.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/src/license_expression/_pyahocorasick.py b/src/license_expression/_pyahocorasick.py index 2a1f5bb..156b7af 100644 --- a/src/license_expression/_pyahocorasick.py +++ b/src/license_expression/_pyahocorasick.py @@ -632,11 +632,11 @@ def overlap(self, other): # tokenize to separate text from parens _tokenizer = re.compile( r""" - (?P[^\s\(\)]+) + (?:[^\s\(\)]+) # text | - (?P\s+) + (?:\s+) # space | - (?P[\(\)]) + (?:[\(\)]) # parens """, re.VERBOSE | re.MULTILINE | re.UNICODE, ) @@ -646,4 +646,4 @@ def get_tokens(tokens_string): """ Return an iterable of strings splitting on spaces and parens. """ - return [match for match in _tokenizer.split(tokens_string.lower()) if match] + return _tokenizer.findall(tokens_string.lower())