text-statistics-dashboard / preprocess.py
psina116's picture
Update preprocess.py
24bb467 verified
Raw
History Blame Contribute Delete
1.05 kB
import re
# очистка
def clean_text(text):
text = re.sub(r'\[[^\]]*\]', '', text) # []
text = re.sub(r'<[^>]*>', '', text) # <>
text = re.sub(r'\d+', '', text) # цифры
text = re.sub(r'\s+', ' ', text) # пробелы
return text.strip().lower()
# токенизирует на предложения
def split_sentences(text):
text = clean_text(text)
sentences = re.split(r'[.!?;:]+', text)
return [s.strip() for s in sentences if s.strip()]
# токенизирует на слова в предложениях
def split_words(sentence):
words = sentence.split()
cleaned = []
for w in words:
w = w.strip('.,!?;:()"\'—')
if w:
cleaned.append(w)
return cleaned
# токенизирует на слова сразу
def split_words_from_text(text):
text = clean_text(text)
words = text.split()
cleaned = []
for w in words:
w = w.strip('.,!?;:()"\'—')
if w:
cleaned.append(w)
return cleaned