svm-model / utils.py
amitansu1024's picture
adds app and utils
d8e1c3b
Raw
History Blame Contribute Delete
4.37 kB
from sklearn import svm
import re
import nltk
import pandas as pd
from nltk.corpus import stopwords
from nltk.stem import WordNetLemmatizer
from nltk.stem import PorterStemmer
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder
from sklearn.metrics import classification_report
import pickle
""""
*** USAGE ***
Proprocessing and model loading should be done at the initialization of the app,
/model.pkl/ can be found at ./app/ml/ alongside the dataset, can be done by
/init_model/
from utils import init_model, inference
model = init_model("BooksDataSet.csv", "model.pkl")
Then, for inference we can simply call /inference/ which returns a string (genre)
print(inference(model, "spaceship armada")) # returns "Science Fiction"
"""
tfidf_vectorizer = TfidfVectorizer(max_df=0.8, max_features=10000)
LE = LabelEncoder()
def setup_vectorizer(xtrain, xval):
"vectorizer used for converting words to vector"
xtrain_tfidf = tfidf_vectorizer.fit_transform(xtrain.values.astype('U'))
xval_tfidf = tfidf_vectorizer.transform(xval.values.astype('U'))
return xtrain_tfidf, xval_tfidf
def setup_nltk():
"""
Initialize nltk before preprocessing, needs to be called only once for python
dependencies
"""
nltk.download('stopwords')
nltk.download('wordnet')
def lematizing(sentence):
lemma=WordNetLemmatizer()
stemSentence = ""
for word in sentence.split():
stem = lemma.lemmatize(word)
stemSentence += stem
stemSentence += " "
stemSentence = stemSentence.strip()
return stemSentence
def remove_stopwords(text):
"""
function to remove stopwords
"""
stop_words = set(stopwords.words('english'))
no_stopword_text = [w for w in text.split() if not w in stop_words]
return ' '.join(no_stopword_text)
def stemming(sentence):
stemmer = PorterStemmer()
stemSentence = ""
for word in sentence.split():
stem = stemmer.stem(word)
stemSentence += stem
stemSentence += " "
stemSentence = stemSentence.strip()
return stemSentence
def clean(text):
"""
remove backslash-apostrophe, whitespace and convert to lowercase
"""
text = re.sub("\'", "", text)
text = re.sub("[^a-zA-Z]"," ",text)
text = ' '.join(text.split())
text = text.lower()
return text
def train_model(xtrain, xval, ytrain, yval):
"""
Train the model on the dataset
"""
svc = svm.SVC(kernel='linear').fit(xtrain, ytrain)
return svc
def validate_model(model, xval, yval):
"""
Validate the model for accuracy
"""
svpred=model.predict(xval)
# for accuracy reports
print(classification_report(yval,svpred))
return svpred
def save_model(model):
pickle.dump(model, open("model.pkl", 'wb'))
def preprocess_data(filename: str):
"""
Data preprocessing for variability and cleanliness
"""
books=pd.read_csv(filename)
books=pd.DataFrame(books,columns=['book_id','book_name','genre','summary'])
books.loc[:,'summary']=books.loc[:,'summary'].apply(lambda x: clean(x))
books['summary'] = books['summary'].apply(lambda x: remove_stopwords(x))
books['summary'] = books['summary'].apply(lambda x: lematizing(x))
books['summary'] = books['summary'].apply(lambda x: stemming(x))
y=LE.fit_transform(books['genre'])
xtrain, xval, ytrain, yval = train_test_split(books['summary'], y, test_size=0.15, random_state=557)
xtrain, xval = setup_vectorizer(xtrain, xval)
# print(books['summary'])
return [xtrain, xval, ytrain, yval]
def load_model(filename: str):
"load a model from a filename"
model = pickle.load(open("model.pkl", 'rb'))
return model
def inference(model, q):
"enter a string to return the genre of the summary"
q = clean(q)
q = remove_stopwords(q)
q = lematizing(q)
q = stemming(q)
q_vec = tfidf_vectorizer.transform([q])
q_pred = model.predict(q_vec)
return LE.inverse_transform(q_pred)[0]
def init_model(dataset: str, model: str):
"""
Initializes the model should be called at the start of the application
"""
setup_nltk()
xtrain, xval, ytrain, yval = preprocess_data(dataset)
model = load_model(model)
return model