Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 19 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,25 @@ and this project adheres to
package dependencies. The `segment()` APIs remain fully backward compatible.
Attacut tests moved from `tests/noauto_torch/` to `tests/noauto_onnx/`.

### Fixed

- `pythainlp.tokenize.newmm`: fixed a quadratic-time blowup in
`word_tokenize()` (`engine="newmm"`, the default). The `_MAX_GRAPH_SIZE`
cutoff added for #893 only limits how many edges are added from a single
position in one pass; it does not bound how large the ambiguity graph
grows across a long, persistently ambiguous stretch of text. The BFS that
resolves that graph also copied the whole path-so-far on every step, so
one resolution over a large accumulated graph cost O(V x path length)
instead of O(V + E). Together, text with many overlapping dictionary
matches over a long stretch (e.g., `"กรรมกร" * 8000`, 48,000 characters)
made tokenization time grow quadratically with length: 48,000 characters
of this pattern took over 3 seconds. The BFS now rebuilds the path once,
from a predecessor map, instead of copying it at every step; tokenizing
is 6-13x faster on this kind of input and scales close to linearly again.
Checked against the unmodified tokenizer on the full bundled word list
(60k+ words), 20k pairwise concatenations, and known edge cases: token
output is unchanged.

## [5.3.8] - 2026-09-25

### Deprecated
Expand Down
38 changes: 27 additions & 11 deletions pythainlp/tokenize/newmm.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,7 @@
from __future__ import annotations

import re
from collections import defaultdict
from collections import defaultdict, deque
from heapq import heappop, heappush
from typing import TYPE_CHECKING, Optional

Expand Down Expand Up @@ -60,21 +60,33 @@
del _TEXT_SCAN_RIGHT


def _bfs_paths_graph(
def _bfs_shortest_path(
graph: defaultdict[int, list[int]], start: int, goal: int
) -> Generator[list[int], None, None]:
) -> list[int]:
# visited set prevents re-exploring nodes already reached via a shorter
# path, converting worst-case BFS from exponential to O(V + E).
# The path itself is rebuilt once, from a predecessor map, instead of
# being copied (path + [pos]) at every step: that copy made a single
# BFS call cost O(V x path length) instead of O(V + E), which is what
# let long, persistently ambiguous input make word_tokenize() quadratic.
visited: set[int] = {start}
queue = [(start, [start])]
predecessor: dict[int, int] = {}
queue: deque[int] = deque([start])
while queue:
(vertex, path) = queue.pop(0)
vertex = queue.popleft()
for pos in graph[vertex]:
if pos == goal:
yield path + [pos]
elif pos not in visited:
predecessor[pos] = vertex
path = [goal]
while path[-1] != start:
path.append(predecessor[path[-1]])
path.reverse()
return path
if pos not in visited:
visited.add(pos)
queue.append((pos, path + [pos]))
predecessor[pos] = vertex
queue.append(pos)
raise ValueError(f"no path from {start} to {goal} in the ambiguity graph")


def _onecut(text: str, custom_dict: Trie) -> Generator[str, None, None]:
Expand All @@ -90,25 +102,28 @@ def _onecut(text: str, custom_dict: Trie) -> Generator[str, None, None]:

len_text = len(text)
pos_list = [0] # priority queue of possible breaking positions
pos_set = {0} # mirrors pos_list, for an O(1) membership check
end_pos = 0
while pos_list[0] < len_text:
begin_pos = heappop(pos_list)
pos_set.discard(begin_pos)
for word in custom_dict.prefixes(text, begin_pos):
end_pos_candidate = begin_pos + len(word)
if valid_poss[end_pos_candidate]:
graph[begin_pos].append(end_pos_candidate)
graph_size = graph_size + 1

if end_pos_candidate not in pos_list:
if end_pos_candidate not in pos_set:
heappush(pos_list, end_pos_candidate)
pos_set.add(end_pos_candidate)

if graph_size > _MAX_GRAPH_SIZE:
break

len_pos_list = len(pos_list)
if len_pos_list == 1: # one candidate, no longer ambiguous
end_pos_candidates = next(
_bfs_paths_graph(graph, end_pos, pos_list[0])
end_pos_candidates = _bfs_shortest_path(
graph, end_pos, pos_list[0]
)
graph_size = 0
graph.clear()
Expand Down Expand Up @@ -145,6 +160,7 @@ def _onecut(text: str, custom_dict: Trie) -> Generator[str, None, None]:
graph.clear()
yield text[begin_pos:end_pos]
heappush(pos_list, end_pos)
pos_set.add(end_pos)


def segment(
Expand Down
20 changes: 20 additions & 0 deletions tests/core/test_tokenize.py
Original file line number Diff line number Diff line change
Expand Up @@ -615,6 +615,26 @@ def test_newmm_ambiguous_performance(self):
# Should complete in well under 1 second after the BFS fix.
self.assertLess(elapsed, 5.0)

def test_newmm_persistent_ambiguity_performance(self):
# Regression test: the _MAX_GRAPH_SIZE cutoff only limits how many
# edges are added from a single position in one pass, not the total
# size the ambiguity graph accumulates across many positions. Text
# that stays ambiguous for a long stretch (many overlapping
# dictionary prefixes, e.g. "กรรมกร" repeated) let that graph grow
# to the size of the whole stretch before the first resolution, and
# _bfs_shortest_path used to copy the whole path-so-far on every BFS
# step, making that one resolution O(V x path length). Together this
# made word_tokenize() quadratic in text length: 48,000 characters
# of this pattern took over 3 seconds before the fix.
text = "กรรมกร" * 16000
t = time.perf_counter()
result = word_tokenize(text, engine="newmm")
elapsed = time.perf_counter() - t
self.assertIsInstance(result, list)
self.assertGreater(len(result), 0)
self.assertEqual("".join(result), text)
self.assertLess(elapsed, 5.0)

def test_tcc(self):
assert_segment_handles_none_and_empty(self, tcc.segment)
self.assertEqual(
Expand Down
Loading