diff --git a/CHANGELOG.md b/CHANGELOG.md index 251846bc0..ee09fcdc3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -55,6 +55,25 @@ and this project adheres to package dependencies. The `segment()` APIs remain fully backward compatible. Attacut tests moved from `tests/noauto_torch/` to `tests/noauto_onnx/`. +### Fixed + +- `pythainlp.tokenize.newmm`: fixed a quadratic-time blowup in + `word_tokenize()` (`engine="newmm"`, the default). The `_MAX_GRAPH_SIZE` + cutoff added for #893 only limits how many edges are added from a single + position in one pass; it does not bound how large the ambiguity graph + grows across a long, persistently ambiguous stretch of text. The BFS that + resolves that graph also copied the whole path-so-far on every step, so + one resolution over a large accumulated graph cost O(V x path length) + instead of O(V + E). Together, text with many overlapping dictionary + matches over a long stretch (e.g., `"กรรมกร" * 8000`, 48,000 characters) + made tokenization time grow quadratically with length: 48,000 characters + of this pattern took over 3 seconds. The BFS now rebuilds the path once, + from a predecessor map, instead of copying it at every step; tokenizing + is 6-13x faster on this kind of input and scales close to linearly again. + Checked against the unmodified tokenizer on the full bundled word list + (60k+ words), 20k pairwise concatenations, and known edge cases: token + output is unchanged. + ## [5.3.8] - 2026-09-25 ### Deprecated diff --git a/pythainlp/tokenize/newmm.py b/pythainlp/tokenize/newmm.py index 07542d1bc..4aba730db 100644 --- a/pythainlp/tokenize/newmm.py +++ b/pythainlp/tokenize/newmm.py @@ -17,7 +17,7 @@ from __future__ import annotations import re -from collections import defaultdict +from collections import defaultdict, deque from heapq import heappop, heappush from typing import TYPE_CHECKING, Optional @@ -60,21 +60,33 @@ del _TEXT_SCAN_RIGHT -def _bfs_paths_graph( +def _bfs_shortest_path( graph: defaultdict[int, list[int]], start: int, goal: int -) -> Generator[list[int], None, None]: +) -> list[int]: # visited set prevents re-exploring nodes already reached via a shorter # path, converting worst-case BFS from exponential to O(V + E). + # The path itself is rebuilt once, from a predecessor map, instead of + # being copied (path + [pos]) at every step: that copy made a single + # BFS call cost O(V x path length) instead of O(V + E), which is what + # let long, persistently ambiguous input make word_tokenize() quadratic. visited: set[int] = {start} - queue = [(start, [start])] + predecessor: dict[int, int] = {} + queue: deque[int] = deque([start]) while queue: - (vertex, path) = queue.pop(0) + vertex = queue.popleft() for pos in graph[vertex]: if pos == goal: - yield path + [pos] - elif pos not in visited: + predecessor[pos] = vertex + path = [goal] + while path[-1] != start: + path.append(predecessor[path[-1]]) + path.reverse() + return path + if pos not in visited: visited.add(pos) - queue.append((pos, path + [pos])) + predecessor[pos] = vertex + queue.append(pos) + raise ValueError(f"no path from {start} to {goal} in the ambiguity graph") def _onecut(text: str, custom_dict: Trie) -> Generator[str, None, None]: @@ -90,25 +102,28 @@ def _onecut(text: str, custom_dict: Trie) -> Generator[str, None, None]: len_text = len(text) pos_list = [0] # priority queue of possible breaking positions + pos_set = {0} # mirrors pos_list, for an O(1) membership check end_pos = 0 while pos_list[0] < len_text: begin_pos = heappop(pos_list) + pos_set.discard(begin_pos) for word in custom_dict.prefixes(text, begin_pos): end_pos_candidate = begin_pos + len(word) if valid_poss[end_pos_candidate]: graph[begin_pos].append(end_pos_candidate) graph_size = graph_size + 1 - if end_pos_candidate not in pos_list: + if end_pos_candidate not in pos_set: heappush(pos_list, end_pos_candidate) + pos_set.add(end_pos_candidate) if graph_size > _MAX_GRAPH_SIZE: break len_pos_list = len(pos_list) if len_pos_list == 1: # one candidate, no longer ambiguous - end_pos_candidates = next( - _bfs_paths_graph(graph, end_pos, pos_list[0]) + end_pos_candidates = _bfs_shortest_path( + graph, end_pos, pos_list[0] ) graph_size = 0 graph.clear() @@ -145,6 +160,7 @@ def _onecut(text: str, custom_dict: Trie) -> Generator[str, None, None]: graph.clear() yield text[begin_pos:end_pos] heappush(pos_list, end_pos) + pos_set.add(end_pos) def segment( diff --git a/tests/core/test_tokenize.py b/tests/core/test_tokenize.py index ca14e2bba..8aa912c0a 100644 --- a/tests/core/test_tokenize.py +++ b/tests/core/test_tokenize.py @@ -615,6 +615,26 @@ def test_newmm_ambiguous_performance(self): # Should complete in well under 1 second after the BFS fix. self.assertLess(elapsed, 5.0) + def test_newmm_persistent_ambiguity_performance(self): + # Regression test: the _MAX_GRAPH_SIZE cutoff only limits how many + # edges are added from a single position in one pass, not the total + # size the ambiguity graph accumulates across many positions. Text + # that stays ambiguous for a long stretch (many overlapping + # dictionary prefixes, e.g. "กรรมกร" repeated) let that graph grow + # to the size of the whole stretch before the first resolution, and + # _bfs_shortest_path used to copy the whole path-so-far on every BFS + # step, making that one resolution O(V x path length). Together this + # made word_tokenize() quadratic in text length: 48,000 characters + # of this pattern took over 3 seconds before the fix. + text = "กรรมกร" * 16000 + t = time.perf_counter() + result = word_tokenize(text, engine="newmm") + elapsed = time.perf_counter() - t + self.assertIsInstance(result, list) + self.assertGreater(len(result), 0) + self.assertEqual("".join(result), text) + self.assertLess(elapsed, 5.0) + def test_tcc(self): assert_segment_handles_none_and_empty(self, tcc.segment) self.assertEqual(