File size: 8,074 Bytes
80dcb4a
 
ab28afa
80dcb4a
 
 
 
 
 
 
 
 
 
 
 
ab28afa
2aae7f1
ab28afa
80dcb4a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f1d9664
 
 
2aae7f1
80dcb4a
 
 
 
f1d9664
80dcb4a
 
 
 
 
 
 
 
 
 
 
 
318ea53
80dcb4a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2aae7f1
 
 
 
 
80dcb4a
 
 
 
 
 
2aae7f1
 
 
 
80dcb4a
 
 
 
 
 
 
 
 
 
 
 
 
 
f1d9664
80dcb4a
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
import gradio as gr
from collections import Counter
import nltk
import torch
import numpy as np
import pandas as pd
import re
import string
import json
from typing import Tuple
from dataclasses import dataclass
from nltk.corpus import stopwords
from pymystem3 import Mystem
from torch import nn

# Загрузка необходимых ресурсов NLTK
nltk.download('stopwords', quiet=True)

# Регулярное выражение для поиска стандартных эмодзи и текстовых смайлов
emoji_pattern = re.compile(
    u"(["
    u"\U0001F600-\U0001F64F"  # смайлики
    u"\U0001F300-\U0001F5FF"  # символы и пиктограммы
    u"\U0001F680-\U0001F6FF"  # транспортные и маппинг символы
    u"\U0001F1E0-\U0001F1FF"  # флаги
    u"]+)|"
    r"(:\)|:\(|;\)|:\-\)|:\-\(|\(:|\):|:D|:P|:\]|O_O|XD|\^\^|<3|:\*|\(\(|\)\))"  # текстовые смайлы
)

# Функция для извлечения смайлов
def extract_emojis(text):
    matches = emoji_pattern.findall(text)
    return ' '.join([match[0] if match[0] else match[1] for match in matches])

# Предобработка текста
mystem = Mystem()
stop_words = set(stopwords.words('russian')) - {'не', 'ни'}

# Загрузка списка матерных слов
with open('list.txt', 'r', encoding='utf-8') as f:
    swear_words = set(f.read().splitlines())

# Загрузка данных
data = pd.read_csv('data.csv')

# Функция для извлечения смайлов
data['emojis'] = data['text'].apply(extract_emojis)

print(data[['text', 'emojis']].head(10))

# Подсчет наиболее часто используемых смайлов
positive_emojis = ' '.join(data[data['sentiment'] == 1]['emojis'])
negative_emojis = ' '.join(data[data['sentiment'] == 0]['emojis'])

positive_emojis_count = Counter(positive_emojis.split())
negative_emojis_count = Counter(negative_emojis.split())

print("Most common emojis in positive tweets:", positive_emojis_count.most_common(10))
print("Most common emojis in negative tweets:", negative_emojis_count.most_common(10))

# Паттерн для поиска имен пользователей (если нужен)
username_pattern = re.compile(r"@\w+")

def data_preprocessing(text: str, positive_emoji_placeholder: str = 'отлично', negative_emoji_placeholder: str = 'плохо') -> Tuple[str, int]:
    """Preprocessing string: lowercase, removing html-tags, punctuation, stopwords, and replacing emojis.
       Also calculates the count of swear words.
    """
    def replace_emoji(match):
        emoji = match.group()
        if emoji in positive_emojis:
            return f" {positive_emoji_placeholder} "
        elif emoji in negative_emojis:
            return f" {negative_emoji_placeholder} "
        return emoji
    
    text = emoji_pattern.sub(replace_emoji, text.lower())
    text = re.sub(r"<.*?>", "", text)
    text = "".join([c if c not in string.punctuation else ' ' for c in text])
    text = re.sub(r'\b[a-zA-Z]+\b', '', text)
    text = username_pattern.sub('', text)
    text = re.sub(r'\s+', ' ', text).strip()
    text = "".join(mystem.lemmatize(text)).strip()
    text = re.sub(r'\b\w*\d\w*\b', '', text)
    text = " ".join(word for word in text.split() if word not in stop_words)
    swear_count = sum(1 for word in text.split() if word in swear_words)
    return text, swear_count

# Преобразование текста в числовое представление
def preprocess_for_model(text: str, vocab_to_int: dict, seq_len: int) -> np.array:
    """Convert cleaned text to padded numerical representation."""
    words = text.split()
    indices = [vocab_to_int.get(word, 0) for word in words]
    if len(indices) < seq_len:
        indices = [0] * (seq_len - len(indices)) + indices
    else:
        indices = indices[:seq_len]
    return np.array(indices)

# Инициализация конфигурации и модели
@dataclass
class ConfigRNN:
    vocab_size: int
    device: str
    n_layers: int
    embedding_dim: int
    hidden_size: int
    seq_len: int
    bidirectional: bool

# Сначала создаем конфигурацию
net_config = ConfigRNN(
    vocab_size=10000,  # Placeholder, так как vocab_to_int еще не загружен
    device='cpu',
    n_layers=1,
    embedding_dim=8,
    hidden_size=16,
    seq_len=110,
    bidirectional=False
)

# Загрузка модели и словаря
class RNNNet(nn.Module):
    def __init__(self, rnn_conf):
        super().__init__()
        self.rnn_conf = rnn_conf
        self.embedding = nn.Embedding(self.rnn_conf.vocab_size, self.rnn_conf.embedding_dim)
        self.rnn_cell = nn.RNN(
            input_size=self.rnn_conf.embedding_dim,
            hidden_size=self.rnn_conf.hidden_size,
            batch_first=True,
            bidirectional=self.rnn_conf.bidirectional,
            num_layers=self.rnn_conf.n_layers
        )
        self.bidirect_factor = 2 if self.rnn_conf.bidirectional else 1
        self.linear = nn.Sequential(
            nn.Linear(self.rnn_conf.hidden_size * self.bidirect_factor, 16),
            nn.Tanh(),
            nn.Linear(16, 1)
        )
    
    def forward(self, x):
        x = self.embedding(x.to(self.rnn_conf.device))
        output, hidden = self.rnn_cell(x)
        hidden = torch.cat((hidden[-2,:,:], hidden[-1,:,:]), dim=1) if self.rnn_conf.bidirectional else hidden[-1,:,:]
        out = self.linear(hidden)
        return out

def load_vocab_and_model():
    # Загрузка словаря
    try:
        with open('vocab_to_int.json', 'r', encoding='utf-8') as f:
            vocab_to_int = json.load(f)
    except FileNotFoundError:
        raise RuntimeError("Файл vocab_to_int.json не найден.")
    
    # Обновление конфигурации с реальным размером словаря
    net_config.vocab_size = len(vocab_to_int) + 1
    
    # Загрузка модели
    model = RNNNet(net_config)
    try:
        model.load_state_dict(torch.load("model_rnn.pth", map_location=torch.device('cpu')))
    except FileNotFoundError:
        raise RuntimeError("Файл модели model_rnn.pth не найден.")
    model.eval()
    
    return model, vocab_to_int

# Теперь загружаем модель и словарь
model, vocab_to_int = load_vocab_and_model()

def predict(text: str):
    cleaned_text, _ = data_preprocessing(text)
    numerical_representation = preprocess_for_model(cleaned_text, vocab_to_int, net_config.seq_len)
    input_tensor = torch.from_numpy(numerical_representation).unsqueeze(0)  # Add batch dimension
    with torch.no_grad():
        output = model(input_tensor)
    prediction = torch.sigmoid(output).item()
    return "Positive 👍" if prediction > 0.45 else "Negative 👎"

# Создание интерфейса с кнопкой Submit и другими элементами
interface = gr.Interface(
    fn=predict, 
    inputs=gr.Textbox(lines=2, placeholder="Введите текст здесь..."), 
    outputs=gr.Textbox(label="Предсказание"),
    title="Текстовый Анализатор Настроений",
    description="Введите текст, чтобы узнать, является ли он положительным или отрицательным. Попробуйте добавить эмодзи для более интересного анализа!",
    theme="compact",
    examples=[
        ["Сегодня был отличный день! 🌟 Прекрасное настроение и много позитивных эмоций! 😊 Спасибо всем за поддержку и вдохновение!"],
        ["Не могу поверить, как всё пошло не так сегодня. все очень плохо((#тяжелыйдень #"]
    ]
)

# Запуск приложения
interface.launch(share=True)