File size: 7,897 Bytes
4d3248c | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 | import re
from pythainlp.util import num_to_thaiword
from pythainlp.transliterate import pronunciate
from pythainlp.tokenize import word_tokenize
class ThaiTextPreprocessor:
def __init__(self, use_g2p=False, use_dataset_spacing=False):
"""
use_g2p: ถ้าเป็น True จะแปลงคำทั้งหมดเป็นคำอ่านแบบสะกดง่าย (เช่น สุทธิกร -> สุดทิกอน)
use_dataset_spacing: ถ้าเป็น True จะแยกคำและจัด spacebar ให้ตรงกับ dataset ที่เทรนมา
"""
self.use_g2p = use_g2p
self.use_dataset_spacing = use_dataset_spacing
# 1. พจนานุกรมคำทับศัพท์ภาษาอังกฤษ (คุณสามารถมาเพิ่มคำฮิตๆ ในนี้ได้เลย)
self.en_th_dict = {
"facebook": "เฟซบุ๊ก", "youtube": "ยูทูบ", "twitter": "ทวิตเตอร์",
"tiktok": "ติ๊กต็อก", "instagram": "อินสตาแกรม", "line": "ไลน์",
"ai": "เอไอ", "update": "อัปเดต", "app": "แอป", "model": "โมเดล",
"infer": "อินเฟอร์", "data": "ดาต้า", "dataset": "ดาต้าเซ็ต",
"ok": "โอเค", "ui": "ยูไอ", "web": "เว็บ", "error": "เออเร่อ"
}
def process(self, text):
if not text:
return text
text = self._replace_english(text)
text = self._replace_numbers(text)
text = self._expand_mai_yamok(text)
# ถ้าเปิดโหมดสะกดง่าย ให้ทำเป็นขั้นตอนสุดท้ายก่อนเว้นวรรค
if self.use_g2p:
text = self._to_simple_spelling(text)
if self.use_dataset_spacing:
text = self._apply_dataset_spacing(text)
return text
def _replace_english(self, text):
# ค้นหาคำภาษาอังกฤษในดิกชันนารีแล้วแทนที่ (ไม่สนตัวพิมพ์เล็ก/ใหญ่)
for en, th in self.en_th_dict.items():
# (?i) คือ Ignore case, \b คือเช็คให้เป็นคำๆ ไม่ใช่ติ่งของคำอื่น
text = re.sub(r'(?i)\b' + en + r'\b', th, text)
return text
def _replace_numbers(self, text):
def _repl(match):
# เอาลูกน้ำออกก่อนแปลงค่า
num_str = match.group(0).replace(',', '')
try:
if '.' in num_str:
return num_to_thaiword(float(num_str))
else:
return num_to_thaiword(int(num_str))
except:
return num_str
# ค้นหาตัวเลขทั้งหมด (รองรับจุดทศนิยมและลูกน้ำ เช่น 1,000 หรือ 5.5)
return re.sub(r'\d+(?:,\d+)*(?:\.\d+)?', _repl, text)
def _expand_mai_yamok(self, text):
if 'ๆ' not in text:
return text
words = word_tokenize(text, engine="newmm")
result = []
for w in words:
if 'ๆ' in w:
clean_w = w.replace('ๆ', '').strip()
count = w.count('ๆ')
if clean_w:
# ถ้าคำมี text ติดมาด้วย เช่น "จริงๆ" -> ก็เบิ้ล "จริง"
result.append(w.replace('ๆ', '')) # เก็บช่องว่างเดิมไว้ถ้ามี
for _ in range(count):
result.append(clean_w)
else:
# ถ้าเจอ "ๆ" โดดๆ ให้ย้อนหาคำก่อนหน้าที่ไม่ใช่ช่องว่าง
last_word = ""
for past_w in reversed(result):
if past_w.strip():
last_word = past_w.strip()
break
result.append(w.replace('ๆ', '')) # เก็บ space ของตัวมันเองไว้ (ถ้ามี)
for _ in range(count):
if last_word:
result.append(last_word)
else:
result.append(w)
return "".join(result)
def _to_simple_spelling(self, text):
# 1. หั่นข้อความเป็นคำๆ ก่อน เพื่อไม่ให้ wunsen เอ๋อกับเครื่องหมายและช่องว่าง
words = word_tokenize(text, engine="newmm")
result = []
for w in words:
# 2. ถ้าไม่ใช่ภาษาไทยล้วน (เช่น ช่องว่าง, !!!, อักษรพิเศษ) ให้ปล่อยผ่านไปเลย
if not re.match(r'^[ก-๙]+$', w):
result.append(w)
continue
try:
# 3. ส่งเฉพาะคำภาษาไทยไปสะกดคำอ่าน
reading = pronunciate(w, engine="wunsen")
if reading:
reading = reading.replace("-", "")
reading = reading.replace('\u0e3a', '') # ลบพินทุ ( ฺ )
result.append(reading)
else:
result.append(w)
except:
# ถ้า wunsen error ให้ใช้คำเดิมกันเหนียว
result.append(w)
return "".join(result)
def _apply_dataset_spacing(self, text):
# 1. แยกประโยคด้วยช่องว่างเดิมก่อน (เพื่อจัดการ double space ทีหลัง)
parts = text.split(" ")
segmented_parts = []
for part in parts:
if not part.strip():
continue
# 2. ตัดคำในแต่ละส่วน
words = word_tokenize(part, engine="newmm")
# 3. เชื่อมด้วย 2 ช่องว่าง (เพิ่มจากเดิม 1)
segmented_parts.append(" ".join(words))
# 4. เชื่อมแต่ละส่วนด้วย 3 ช่องว่าง (เพิ่มจากเดิม 2)
return " ".join(segmented_parts)
# --- ส่วนทดสอบ ---
if __name__ == "__main__":
preprocessor = ThaiTextPreprocessor(use_g2p=True)
test_text = "เตือน!!! ภาคใต้ตอนล่าง ตั้งแต่จังหวัด พัทลุง สงขลา ตรัง สตูล ยะลา ปัตตานี นราธิวาส วันนี้มีฝนหนัก หลายพื้นที่ ช่วงบ่าย 3 โมง เป็นต้นไป ภาพพื้นที่เสี่ยงใต้โพส"
print("G2P Mode:", preprocessor.process(test_text)) |