Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion .github/workflows/lint.yml
Original file line number Diff line number Diff line change
Expand Up @@ -26,4 +26,4 @@ jobs:
uses: astral-sh/ruff-action@v3
with:
src: "./pythainlp"
args: check --verbose --line-length 79 --select C901
args: check --fix --verbose --line-length 79 --select I,W,C901,W291,W293
8 changes: 4 additions & 4 deletions docs/conf.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,8 @@
import os
import sys
import traceback
from datetime import datetime
from datetime import date
from datetime import date, datetime

import pythainlp

# -- Path setup --------------------------------------------------------------
Expand All @@ -34,7 +34,7 @@
# -- Get version information and date from Git ----------------------------

try:
from subprocess import check_output, STDOUT
from subprocess import STDOUT, check_output

current_branch = (
os.environ["CURRENT_BRANCH"]
Expand Down Expand Up @@ -69,7 +69,7 @@
# .decode()
# .strip()
# )
except Exception as e:
except Exception:
traceback.print_exc()
release = pythainlp.__version__
# today = "<unknown date>"
Expand Down
6 changes: 2 additions & 4 deletions examples/khavee.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,11 @@
# SPDX-FileCopyrightText: 2016-2026 PyThaiNLP Project
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0
"""
Example of using KhaveeVerifier from pythainlp.khavee
"""Example of using KhaveeVerifier from pythainlp.khavee
"""

from pythainlp.khavee import KhaveeVerifier


kv = KhaveeVerifier()

# การเช็คสระ
Expand Down Expand Up @@ -64,7 +62,7 @@
เรื่องวิศวะเก่งกาจประหลาดใจ เรื่องฟิสิกส์ไร้ผู้ใดมาต่อไป
นริศราอีฟเก่งกว่าใครเพื่อน คอยช่วยเตือนเรื่องงานคอยสั่งสอน
อ่านตำราหาความรู้ไม่ละทอน เป็นคนดีศรีนครของจิตรลดา
ภัสนันท์นาคลออหรือมีมี่ เรื่องเกมเอ่อเก่งกาจไม่กังขา
ภัสนันท์นาคลออหรือมีมี่ เรื่องเกมเอ่อเก่งกาจไม่กังขา
เกมอะไรก็เล่นได้ไม่ลดวา สุดฉลาดมากปัญญามาครบครัน""",
k_type=8,
)
Expand Down
2 changes: 1 addition & 1 deletion notebooks/convert_thai2rom_to_onnx.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,7 @@
"outputs": [],
"source": [
"import torch\n",
"\n",
"from pythainlp.corpus import get_corpus_path\n",
"from pythainlp.transliterate.thai2rom import _MODEL_NAME\n",
"\n",
Expand Down Expand Up @@ -115,7 +116,6 @@
"outputs": [],
"source": [
"import torch\n",
"import numpy as np\n",
"\n",
"input_tensor = torch.Tensor([[30, 19, 8, 30, 38, 37, 10, 3]]).long()\n",
"\n",
Expand Down
4 changes: 2 additions & 2 deletions notebooks/create_words.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -6,8 +6,8 @@
"metadata": {},
"outputs": [],
"source": [
"from pythainlp.transliterate import pronunciate\n",
"from pythainlp import thai_consonants"
"from pythainlp import thai_consonants\n",
"from pythainlp.transliterate import pronunciate"
]
},
{
Expand Down
5 changes: 3 additions & 2 deletions notebooks/test_chat.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,9 @@
},
"outputs": [],
"source": [
"from pythainlp.chat.core import ChatBotModel\n",
"import torch"
"import torch\n",
"\n",
"from pythainlp.chat.core import ChatBotModel"
]
},
{
Expand Down
1 change: 1 addition & 0 deletions notebooks/test_el.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"1\""
]
},
Expand Down
5 changes: 3 additions & 2 deletions notebooks/test_wangchanglm.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,9 @@
},
"outputs": [],
"source": [
"from pythainlp.generate.wangchanglm import WangChanGLM\n",
"import torch"
"import torch\n",
"\n",
"from pythainlp.generate.wangchanglm import WangChanGLM"
]
},
{
Expand Down
2 changes: 1 addition & 1 deletion notebooks/test_wsd.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -80,7 +80,7 @@
},
"outputs": [],
"source": [
"from pythainlp.corpus import get_corpus_path, thai_wsd_dict"
"from pythainlp.corpus import thai_wsd_dict"
]
},
{
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -301,4 +301,4 @@ docstring-code-format = true
[tool.ruff.lint.mccabe]
# Flag errors (`C901`) whenever the complexity level exceeds 5. Default is 10.
# We should aim to gradually reduce this to 10.
max-complexity = 40
max-complexity = 38
15 changes: 15 additions & 0 deletions pythainlp/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,21 @@
โกรธจี๊ดจ๋อยจ่มถ้ำ อยู่เฝ้า “อตฺตา” ๚ะ๛
๑๒ กรกฎาคม ๒๕๕๘"""

__all__ = [
"collate",
"correct",
"pos_tag",
"romanize",
"spell",
"sent_tokenize",
"subword_tokenize",
"soundex",
"thai_strftime",
"transliterate",
"Tokenizer",
"word_tokenize",
]

from pythainlp.soundex import soundex
from pythainlp.spell import correct, spell
from pythainlp.tag import pos_tag
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/ancient/__init__.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
# SPDX-FileCopyrightText: 2016-2026 PyThaiNLP Project
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0
"""
Ancient versions of the Thai language
"""Ancient versions of the Thai language
"""

__all__ = ["aksonhan_to_current", "convert_currency"]
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/ancient/aksonhan.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,8 +23,7 @@


def aksonhan_to_current(word: str) -> str:
"""
Convert AksonHan words to current Thai words
"""Convert AksonHan words to current Thai words

AksonHan (อักษรหัน) writes down two consonants for the \
spelling of the /a/ vowels. (สระ อะ).
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/ancient/currency.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,8 +5,7 @@


def convert_currency(value: float, from_unit: str) -> dict:
"""
Convert ancient Thai currency to other units
"""Convert ancient Thai currency to other units

* เบี้ย (Bia)
* อัฐ (At)
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/augment/__init__.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
# SPDX-FileCopyrightText: 2016-2026 PyThaiNLP Project
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0
"""
Thai text augment
"""Thai text augment
"""

__all__ = ["WordNetAug"]
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/augment/lm/__init__.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
# SPDX-FileCopyrightText: 2016-2026 PyThaiNLP Project
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0
"""
Language Models
"""Language Models
"""

__all__ = [
Expand Down
15 changes: 5 additions & 10 deletions pythainlp/augment/lm/fasttext.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,15 +12,13 @@


class FastTextAug:
"""
Text Augment from fastText
"""Text Augment from fastText

:param str model_path: path of model file
"""

def __init__(self, model_path: str):
"""
:param str model_path: path of model file
""":param str model_path: path of model file
"""
if model_path.endswith(".bin"):
self.model = FastText_gensim.load_facebook_vectors(model_path)
Expand All @@ -31,8 +29,7 @@ def __init__(self, model_path: str):
self.dict_wv = list(self.model.key_to_index.keys())

def tokenize(self, text: str) -> list[str]:
"""
Thai text tokenization for fastText
"""Thai text tokenization for fastText

:param str text: Thai text

Expand All @@ -42,8 +39,7 @@ def tokenize(self, text: str) -> list[str]:
return word_tokenize(text, engine="icu")

def modify_sent(self, sent: str, p: float = 0.7) -> list[list[str]]:
"""
:param str sent: text of sentence
""":param str sent: text of sentence
:param float p: probability
:rtype: List[List[str]]
"""
Expand All @@ -62,8 +58,7 @@ def modify_sent(self, sent: str, p: float = 0.7) -> list[list[str]]:
def augment(
self, sentence: str, n_sent: int = 1, p: float = 0.7
) -> list[tuple[str]]:
"""
Text Augment from fastText
"""Text Augment from fastText

You may want to download the Thai model
from https://fasttext.cc/docs/en/crawl-vectors.html.
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/augment/lm/phayathaibert.py
Original file line number Diff line number Diff line change
Expand Up @@ -57,8 +57,7 @@ def generate(
def augment(
self, text: str, num_augs: int = 3, sample: bool = False
) -> list[str]:
"""
Text augmentation from PhayaThaiBERT
"""Text augmentation from PhayaThaiBERT

:param str text: Thai text
:param int num_augs: an amount of augmentation text needed as an output
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/augment/lm/wangchanberta.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,8 +51,7 @@ def generate(self, sentence: str, num_replace_tokens: int = 3):
return self.sent2

def augment(self, sentence: str, num_replace_tokens: int = 3) -> list[str]:
"""
Text augmentation from WangchanBERTa
"""Text augmentation from WangchanBERTa

:param str sentence: Thai sentence
:param int num_replace_tokens: number replace tokens
Expand Down
3 changes: 1 addition & 2 deletions pythainlp/augment/word2vec/__init__.py
Original file line number Diff line number Diff line change
@@ -1,8 +1,7 @@
# SPDX-FileCopyrightText: 2016-2026 PyThaiNLP Project
# SPDX-FileType: SOURCE
# SPDX-License-Identifier: Apache-2.0
"""
Word2Vec
"""Word2Vec
"""

__all__ = ["Word2VecAug", "Thai2fitAug", "LTW2VAug"]
Expand Down
12 changes: 4 additions & 8 deletions pythainlp/augment/word2vec/bpemb_wv.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,8 +7,7 @@


class BPEmbAug:
"""
Thai Text Augment using word2vec from BPEmb
"""Thai Text Augment using word2vec from BPEmb

BPEmb:
`github.com/bheinzerling/bpemb <https://github.com/bheinzerling/bpemb>`_
Expand All @@ -22,15 +21,13 @@ def __init__(self, lang: str = "th", vs: int = 100000, dim: int = 300):
self.load_w2v()

def tokenizer(self, text: str) -> list[str]:
"""
:param str text: Thai text
""":param str text: Thai text
:rtype: List[str]
"""
return self.bpemb_temp.encode(text)

def load_w2v(self):
"""
Load BPEmb model
"""Load BPEmb model
"""
self.aug = Word2VecAug(
self.model, tokenize=self.tokenizer, type="model"
Expand All @@ -39,8 +36,7 @@ def load_w2v(self):
def augment(
self, sentence: str, n_sent: int = 1, p: float = 0.7
) -> list[tuple[str]]:
"""
Text Augment using word2vec from BPEmb
"""Text Augment using word2vec from BPEmb

:param str sentence: Thai sentence
:param int n_sent: number of sentence
Expand Down
9 changes: 3 additions & 6 deletions pythainlp/augment/word2vec/core.py
Original file line number Diff line number Diff line change
Expand Up @@ -10,8 +10,7 @@ class Word2VecAug:
def __init__(
self, model: str, tokenize: object, type: str = "file"
) -> None:
"""
:param str model: path of model
""":param str model: path of model
:param object tokenize: tokenize function
:param str type: model type (file, binary)
"""
Expand All @@ -29,8 +28,7 @@ def __init__(
self.dict_wv = list(self.model.key_to_index.keys())

def modify_sent(self, sent: str, p: float = 0.7) -> list[list[str]]:
"""
:param str sent: text of sentence
""":param str sent: text of sentence
:param float p: probability
:rtype: List[List[str]]
"""
Expand All @@ -49,8 +47,7 @@ def modify_sent(self, sent: str, p: float = 0.7) -> list[list[str]]:
def augment(
self, sentence: str, n_sent: int = 1, p: float = 0.7
) -> list[tuple[str]]:
"""
:param str sentence: text of sentence
""":param str sentence: text of sentence
:param int n_sent: maximum number of synonymous sentences
:param int p: probability

Expand Down
12 changes: 4 additions & 8 deletions pythainlp/augment/word2vec/ltw2v.py
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,7 @@


class LTW2VAug:
"""
Text Augment using word2vec from LTW2V
"""Text Augment using word2vec from LTW2V

LTW2V:
`github.com/PyThaiNLP/large-thaiword2vec <https://github.com/PyThaiNLP/large-thaiword2vec>`_
Expand All @@ -21,23 +20,20 @@ def __init__(self):
self.load_w2v()

def tokenizer(self, text: str) -> list[str]:
"""
:param str text: Thai text
""":param str text: Thai text
:rtype: List[str]
"""
return word_tokenize(text, engine="newmm")

def load_w2v(self): # insert substitute
"""
Load LTW2V's word2vec model
"""Load LTW2V's word2vec model
"""
self.aug = Word2VecAug(self.ltw2v_wv, self.tokenizer, type="binary")

def augment(
self, sentence: str, n_sent: int = 1, p: float = 0.7
) -> list[tuple[str]]:
"""
Text Augment using word2vec from Thai2Fit
"""Text Augment using word2vec from Thai2Fit

:param str sentence: Thai sentence
:param int n_sent: number of sentence
Expand Down
Loading
Loading