File size: 7,897 Bytes
4d3248c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
import re
from pythainlp.util import num_to_thaiword
from pythainlp.transliterate import pronunciate
from pythainlp.tokenize import word_tokenize

class ThaiTextPreprocessor:
    def __init__(self, use_g2p=False, use_dataset_spacing=False):
        """

        use_g2p: ถ้าเป็น True จะแปลงคำทั้งหมดเป็นคำอ่านแบบสะกดง่าย (เช่น สุทธิกร -> สุดทิกอน)

        use_dataset_spacing: ถ้าเป็น True จะแยกคำและจัด spacebar ให้ตรงกับ dataset ที่เทรนมา

        """
        self.use_g2p = use_g2p
        self.use_dataset_spacing = use_dataset_spacing
        
        # 1. พจนานุกรมคำทับศัพท์ภาษาอังกฤษ (คุณสามารถมาเพิ่มคำฮิตๆ ในนี้ได้เลย)
        self.en_th_dict = {
            "facebook": "เฟซบุ๊ก", "youtube": "ยูทูบ", "twitter": "ทวิตเตอร์",
            "tiktok": "ติ๊กต็อก", "instagram": "อินสตาแกรม", "line": "ไลน์",
            "ai": "เอไอ", "update": "อัปเดต", "app": "แอป", "model": "โมเดล",
            "infer": "อินเฟอร์", "data": "ดาต้า", "dataset": "ดาต้าเซ็ต",
            "ok": "โอเค", "ui": "ยูไอ", "web": "เว็บ", "error": "เออเร่อ"
        }

    def process(self, text):
        if not text:
            return text
        
        text = self._replace_english(text)
        text = self._replace_numbers(text)
        text = self._expand_mai_yamok(text)
        
        # ถ้าเปิดโหมดสะกดง่าย ให้ทำเป็นขั้นตอนสุดท้ายก่อนเว้นวรรค
        if self.use_g2p:
            text = self._to_simple_spelling(text)
            
        if self.use_dataset_spacing:
            text = self._apply_dataset_spacing(text)
            
        return text

    def _replace_english(self, text):
        # ค้นหาคำภาษาอังกฤษในดิกชันนารีแล้วแทนที่ (ไม่สนตัวพิมพ์เล็ก/ใหญ่)
        for en, th in self.en_th_dict.items():
            # (?i) คือ Ignore case, \b คือเช็คให้เป็นคำๆ ไม่ใช่ติ่งของคำอื่น
            text = re.sub(r'(?i)\b' + en + r'\b', th, text)
        return text

    def _replace_numbers(self, text):
        def _repl(match):
            # เอาลูกน้ำออกก่อนแปลงค่า
            num_str = match.group(0).replace(',', '')
            try:
                if '.' in num_str:
                    return num_to_thaiword(float(num_str))
                else:
                    return num_to_thaiword(int(num_str))
            except:
                return num_str
        
        # ค้นหาตัวเลขทั้งหมด (รองรับจุดทศนิยมและลูกน้ำ เช่น 1,000 หรือ 5.5)
        return re.sub(r'\d+(?:,\d+)*(?:\.\d+)?', _repl, text)

    def _expand_mai_yamok(self, text):
        if 'ๆ' not in text:
            return text
            
        words = word_tokenize(text, engine="newmm")
        result = []
        for w in words:
            if 'ๆ' in w:
                clean_w = w.replace('ๆ', '').strip()
                count = w.count('ๆ')
                
                if clean_w:
                    # ถ้าคำมี text ติดมาด้วย เช่น "จริงๆ" -> ก็เบิ้ล "จริง"
                    result.append(w.replace('ๆ', '')) # เก็บช่องว่างเดิมไว้ถ้ามี
                    for _ in range(count):
                        result.append(clean_w)
                else:
                    # ถ้าเจอ "ๆ" โดดๆ ให้ย้อนหาคำก่อนหน้าที่ไม่ใช่ช่องว่าง
                    last_word = ""
                    for past_w in reversed(result):
                        if past_w.strip():
                            last_word = past_w.strip()
                            break
                    
                    result.append(w.replace('ๆ', '')) # เก็บ space ของตัวมันเองไว้ (ถ้ามี)
                    for _ in range(count):
                        if last_word:
                            result.append(last_word)
            else:
                result.append(w)
                
        return "".join(result)

    def _to_simple_spelling(self, text):
        # 1. หั่นข้อความเป็นคำๆ ก่อน เพื่อไม่ให้ wunsen เอ๋อกับเครื่องหมายและช่องว่าง
        words = word_tokenize(text, engine="newmm")
        result = []
        
        for w in words:
            # 2. ถ้าไม่ใช่ภาษาไทยล้วน (เช่น ช่องว่าง, !!!, อักษรพิเศษ) ให้ปล่อยผ่านไปเลย
            if not re.match(r'^[ก-๙]+$', w):
                result.append(w)
                continue
                
            try:
                # 3. ส่งเฉพาะคำภาษาไทยไปสะกดคำอ่าน
                reading = pronunciate(w, engine="wunsen")
                if reading:
                    reading = reading.replace("-", "")
                    reading = reading.replace('\u0e3a', '') # ลบพินทุ ( ฺ )
                    result.append(reading)
                else:
                    result.append(w)
            except:
                # ถ้า wunsen error ให้ใช้คำเดิมกันเหนียว
                result.append(w)
        
        return "".join(result)

    def _apply_dataset_spacing(self, text):
        # 1. แยกประโยคด้วยช่องว่างเดิมก่อน (เพื่อจัดการ double space ทีหลัง)
        parts = text.split(" ")
        segmented_parts = []
        
        for part in parts:
            if not part.strip():
                continue
            # 2. ตัดคำในแต่ละส่วน
            words = word_tokenize(part, engine="newmm")
            # 3. เชื่อมด้วย 2 ช่องว่าง (เพิ่มจากเดิม 1)
            segmented_parts.append("  ".join(words))
        
        # 4. เชื่อมแต่ละส่วนด้วย 3 ช่องว่าง (เพิ่มจากเดิม 2)
        return "   ".join(segmented_parts)

# --- ส่วนทดสอบ ---
if __name__ == "__main__":
    preprocessor = ThaiTextPreprocessor(use_g2p=True)
    test_text = "เตือน!!! ภาคใต้ตอนล่าง ตั้งแต่จังหวัด พัทลุง สงขลา ตรัง สตูล ยะลา ปัตตานี นราธิวาส วันนี้มีฝนหนัก หลายพื้นที่ ช่วงบ่าย 3 โมง เป็นต้นไป ภาพพื้นที่เสี่ยงใต้โพส"
    print("G2P Mode:", preprocessor.process(test_text))