diff --git a/CHANGELOG.md b/CHANGELOG.md index 2192e3ecf..ca141f82e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,15 +20,14 @@ and this project adheres to ## [Unreleased] ### Added +- Thai G2P v4 model via Hugging Face Hub using ONNX Runtime (`thaig2p_v4`), available in `pythainlp.transliterate`. +### Changed - `pythainlp.transliterate.fastthaig2p`: Native FastThaiG2P grapheme-to-phoneme conversion engine without external package dependencies. Supports text normalization (numbers, dates, times, phone numbers, symbols, abbreviations, maiyamok), 62k IPA dictionary lookup, and rule-based fallback. Accessible via `transliterate(text, engine="fastthaig2p")` or `FastThaiG2P` class. - -## Changed - - Improve guardrails in `check_sara()` and `nighit()` - `pythainlp.tokenize.deepcut`: migrated from the TensorFlow-based `deepcut` package to a built-in ONNX inference engine, removing the TensorFlow diff --git a/docs/api/transliterate.rst b/docs/api/transliterate.rst index 6f4d27ad7..7b8c0dda5 100644 --- a/docs/api/transliterate.rst +++ b/docs/api/transliterate.rst @@ -75,6 +75,7 @@ This section includes multiple transliteration engines designed to suit various - **ipa**: Provides International Phonetic Alphabet (IPA) representation of Thai text. - **thaig2p**: (default) Transliterates Thai text into the Grapheme-to-Phoneme (G2P) representation. - **thaig2p_v2**: Transliterates Thai text into the Grapheme-to-Phoneme (G2P) representation. This model is from https://huggingface.co/pythainlp/thaig2p-v2.0 +- **thaig2p_v4**: Transliterates Thai text into the Grapheme-to-Phoneme (G2P) representation. This model is from https://huggingface.co/pythainlp/thaig2p-v4 - **fastthaig2p**: Fast Thai Grapheme-to-Phoneme converter for IPA transcription with text normalization and fallback. - **tltk**: Utilizes the TLTK transliteration system for a specific approach to transliteration. - **iso_11940**: Focuses on the ISO 11940 transliteration standard. diff --git a/pyproject.toml b/pyproject.toml index 09907e343..e651e11b5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -224,6 +224,7 @@ noauto-tensorflow = [ # ONNX Runtime-based dependencies - for tests.noauto_onnx noauto-onnx = [ + "huggingface-hub>=0.16.0", "numpy>=1.26.0", "onnxruntime>=1.10.0", "oskut>=1.3", diff --git a/pythainlp/transliterate/core.py b/pythainlp/transliterate/core.py index 389865acb..8429c8435 100644 --- a/pythainlp/transliterate/core.py +++ b/pythainlp/transliterate/core.py @@ -115,6 +115,8 @@ def transliterate( * *tltk_ipa* - tltk, output is International Phonetic Alphabet (IPA) * *thaig2p_v2* - Thai Grapheme-to-Phoneme, output is IPA. https://huggingface.co/pythainlp/thaig2p-v2.0 + * *thaig2p_v4* - Thai Grapheme-to-Phoneme (v4), + output is IPA. https://huggingface.co/pythainlp/thaig2p-v4 * *umt5_thaig2p* - Thai Grapheme-to-Phoneme, output is IPA, powered by UMT5.\ https://huggingface.co/B-K/umt5-thai-g2p-v2-0.5k @@ -162,6 +164,8 @@ def transliterate( from pythainlp.transliterate.iso_11940 import transliterate # type: ignore[assignment] # noqa: I001 elif engine == "thaig2p_v2": from pythainlp.transliterate.thaig2p_v2 import transliterate # noqa: I001 + elif engine == "thaig2p_v4": + from pythainlp.transliterate.thaig2p_v4 import transliterate # noqa: I001 elif engine == "umt5_thaig2p": from pythainlp.transliterate.umt5_thaig2p import transliterate # noqa: I001 elif engine == "fastthaig2p": diff --git a/pythainlp/transliterate/thaig2p_v4.py b/pythainlp/transliterate/thaig2p_v4.py new file mode 100644 index 000000000..18351c8ed --- /dev/null +++ b/pythainlp/transliterate/thaig2p_v4.py @@ -0,0 +1,200 @@ +# SPDX-FileCopyrightText: 2016-2026 PyThaiNLP Project +# SPDX-FileType: SOURCE +# SPDX-License-Identifier: Apache-2.0 +"""Thai Grapheme-to-Phoneme (Thai G2P) v4 + +Hugging Face: https://huggingface.co/pythainlp/thaig2p-v4 +""" + +from __future__ import annotations + +import json +import os +from typing import TYPE_CHECKING, Optional + +from pythainlp.corpus import get_hf_hub +from pythainlp.tools import safe_path_join + +if TYPE_CHECKING: + import numpy as np + from numpy.typing import NDArray + from onnxruntime import InferenceSession + +_REPO_ID: str = "pythainlp/thaig2p-v4" +_MAX_LEN: int = 80 +_PAD_TOKEN: int = 0 +_SOS_TOKEN: int = 1 +_EOS_TOKEN: int = 2 +_UNK_TOKEN: int = 3 + + +class ThaiG2P: + """Thai Grapheme-to-Phoneme using ONNX model (v4). + + This version uses the pythainlp/thaig2p-v4 model based on ONNX + for converting Thai text to International Phonetic Alphabet (IPA) representation. + + For more information, see: + https://huggingface.co/pythainlp/thaig2p-v4 + """ + + _encoder_session: InferenceSession + _decoder_session: InferenceSession + _input_char2idx: dict[str, int] + _target_idx2char: dict[int, str] + _max_len: int + + def __init__( + self, + providers: Optional[list[str]] = None, + model_path: Optional[str] = None, + ) -> None: + """Initialize Thai G2P v4 model. + + :param Optional[list[str]] providers: ONNX runtime execution providers + (default is ``['CPUExecutionProvider']``). + :param Optional[str] model_path: Hugging Face model repository or local directory path + (default is ``'pythainlp/thaig2p-v4'``). + """ + try: + import numpy as np # noqa: F401 + import onnxruntime as ort + except ModuleNotFoundError as exc: + raise ModuleNotFoundError( + "Please install the required packages via " + "'pip install numpy onnxruntime huggingface-hub'." + ) from exc + + if providers is None: + providers = ["CPUExecutionProvider"] + + self._max_len = _MAX_LEN + + if model_path is not None and os.path.isdir(model_path): + vocab_file = safe_path_join(model_path, "vocab.json") + encoder_file = safe_path_join(model_path, "encoder_thaig2p.onnx") + decoder_file = safe_path_join(model_path, "decoder_thaig2p.onnx") + else: + repo_id = model_path if model_path is not None else _REPO_ID + vocab_file = get_hf_hub(repo_id, "vocab.json") + encoder_file = get_hf_hub(repo_id, "encoder_thaig2p.onnx") + decoder_file = get_hf_hub(repo_id, "decoder_thaig2p.onnx") + + with open(vocab_file, "r", encoding="utf-8") as f: + vocab_data = json.load(f) + + self._input_char2idx = vocab_data["input_char2idx"] + self._target_idx2char = { + int(k): v for k, v in vocab_data["target_idx2char"].items() + } + + self._encoder_session = ort.InferenceSession( + encoder_file, providers=providers + ) + self._decoder_session = ort.InferenceSession( + decoder_file, providers=providers + ) + + def _encode_input(self, text: str) -> "NDArray[np.int64]": + """Encode input text into padded token index sequence. + + :param str text: Thai text. + :return: 2D array of token indices with shape (1, _max_len). + :rtype: numpy.ndarray + """ + import numpy as np + + src_indices = ( + [_SOS_TOKEN] + + [self._input_char2idx.get(c, _UNK_TOKEN) for c in text] + + [_EOS_TOKEN] + ) + if len(src_indices) < self._max_len: + src_indices += [_PAD_TOKEN] * (self._max_len - len(src_indices)) + else: + src_indices = src_indices[: self._max_len] + + return np.array([src_indices], dtype=np.int64) + + def g2p(self, text: str) -> str: + """Transliterate Thai text to IPA using G2P v4 model. + + :param str text: Thai text to be transliterated. + :return: IPA transcription. + :rtype: str + """ + if not text or not isinstance(text, str): + return "" + + import numpy as np + + src_tensor = self._encode_input(text) + enc_outputs = self._encoder_session.run( + output_names=["memory", "src_pad_mask"], + input_feed={"src": src_tensor}, + ) + memory, src_pad_mask = enc_outputs + + trg_indices: list[int] = [_SOS_TOKEN] + for _ in range(self._max_len): + trg_padded = trg_indices + [_PAD_TOKEN] * ( + self._max_len - len(trg_indices) + ) + trg_tensor = np.array([trg_padded[: self._max_len]], dtype=np.int64) + dec_outputs = self._decoder_session.run( + output_names=["output", "cross_attention"], + input_feed={ + "trg": trg_tensor, + "memory": memory, + "src_pad_mask": src_pad_mask, + }, + ) + output = dec_outputs[0] + current_step_idx = len(trg_indices) - 1 + next_token_logits = output[0, current_step_idx, :] + next_token = int(np.argmax(next_token_logits)) + if next_token == _EOS_TOKEN: + break + trg_indices.append(next_token) + if len(trg_indices) >= self._max_len: + break + + result_chars = [ + self._target_idx2char.get(idx, "") + for idx in trg_indices[1:] + ] + return "".join(result_chars) + + +_THAI_G2P: Optional[ThaiG2P] = None + + +def transliterate( + text: str, + providers: Optional[list[str]] = None, + model_path: Optional[str] = None, +) -> str: + """Transliterate Thai text using Thai G2P v4 model. + + :param str text: Thai text to be transliterated. + :param Optional[list[str]] providers: ONNX runtime execution providers + (default is ``['CPUExecutionProvider']``). + :param Optional[str] model_path: Hugging Face model repository or local directory path + (default is ``'pythainlp/thaig2p-v4'``). + :return: IPA transcription. + :rtype: str + """ + global _THAI_G2P + if _THAI_G2P is None: + _THAI_G2P = ThaiG2P(providers=providers, model_path=model_path) + return _THAI_G2P.g2p(text) + + +ThaiG2PV4 = ThaiG2P + +__all__: list[str] = [ + "ThaiG2P", + "ThaiG2PV4", + "transliterate", +] + diff --git a/tests/core/test_transliterate.py b/tests/core/test_transliterate.py index 757d87a4d..9717d903a 100644 --- a/tests/core/test_transliterate.py +++ b/tests/core/test_transliterate.py @@ -3,6 +3,7 @@ # SPDX-License-Identifier: Apache-2.0 import unittest +from unittest.mock import patch from pythainlp.transliterate import pronunciate_pali, romanize, transliterate @@ -105,9 +106,16 @@ def test_romanize_lookup(self): def test_transliterate(self): self.assertEqual(transliterate(""), "") + self.assertEqual(transliterate("", engine="thaig2p_v4"), "") self.assertIsNotNone(transliterate("คน", engine="iso_11940")) self.assertIsNotNone(transliterate("แมว", engine="iso_11940")) + @patch("pythainlp.transliterate.thaig2p_v4.transliterate") + def test_transliterate_thaig2p_v4_dispatch(self, mock_g2p): + mock_g2p.return_value = "/kʰon˧/" + self.assertEqual(transliterate("คน", engine="thaig2p_v4"), "/kʰon˧/") + mock_g2p.assert_called_once_with("คน") + def test_transliterate_iso11940(self): self.assertEqual( transliterate("เชียงใหม่", engine="iso_11940"), "echīyngıh̄m̀" diff --git a/tests/noauto_onnx/testn_transliterate_onnx.py b/tests/noauto_onnx/testn_transliterate_onnx.py index 0576e3f16..f4081824d 100644 --- a/tests/noauto_onnx/testn_transliterate_onnx.py +++ b/tests/noauto_onnx/testn_transliterate_onnx.py @@ -39,3 +39,33 @@ def test_thai2rom_onnx_mixed_text(self): result = romanize("ภาษาไทย") self.assertIsInstance(result, str) self.assertGreater(len(result), 0) + + def test_thaig2p_v4_returns_string(self): + from pythainlp.transliterate.thaig2p_v4 import transliterate + + result = transliterate("สวัสดี") + self.assertIsInstance(result, str) + self.assertGreater(len(result), 0) + + def test_thaig2p_v4_empty_string(self): + from pythainlp.transliterate.thaig2p_v4 import transliterate + + result = transliterate("") + self.assertIsInstance(result, str) + self.assertEqual(result, "") + + def test_thaig2p_v4_model_loaded(self): + from pythainlp.transliterate.thaig2p_v4 import ThaiG2P + + g2p = ThaiG2P() + self.assertIsNotNone(g2p._encoder_session) + self.assertIsNotNone(g2p._decoder_session) + self.assertIsNotNone(g2p._input_char2idx) + self.assertIsNotNone(g2p._target_idx2char) + + def test_transliterate_thaig2p_v4(self): + from pythainlp.transliterate import transliterate + + result = transliterate("คน", engine="thaig2p_v4") + self.assertIsInstance(result, str) + self.assertEqual(result, "/kʰon˧/")