Spaces:
Sleeping
Sleeping
| import solara | |
| import pandas as pd | |
| import gensim.downloader as api | |
| from umap import UMAP | |
| from cluestar import plot_text | |
| model = api.load('word2vec-google-news-300') | |
| most_frequent = model.index_to_key[:2000] | |
| embeddings = [model[most_frequent[i]] for i in range(2000)] | |
| reducer = UMAP() | |
| X = reducer.fit_transform(embeddings) | |
| data = pd.DataFrame() | |
| data["words"] = list((model.key_to_index.keys())) | |
| word = solara.reactive("France") | |
| positive_word1 = solara.reactive("France") | |
| positive_word2 = solara.reactive("Madrid") | |
| negative_word = solara.reactive("Paris") | |
| def Page(): | |
| def my_check(text): | |
| if text != "" and text not in list(data["words"].values): | |
| solara.Info(f"{text} is not in the list of available words.") | |
| with solara.Column(margin=10): | |
| solara.Markdown("#Word2Vec") | |
| solara.Markdown("#Nearest neighbors of a word") | |
| solara.InputText("Search for a word:", value=word, continuous_update=True) | |
| my_check(word.value) | |
| if word.value != "" and word.value in list(data["words"].values): | |
| solara.Markdown(f"##Nearest neighbors of the word '{word.value}':") | |
| output = model.most_similar(f"{word.value}", restrict_vocab=25000) | |
| df_output = pd.DataFrame() | |
| df_output["neighbors"] = [output[i][0] for i in range(10)] | |
| df_output["cosine similarity"] = [output[i][1] for i in range(10)] | |
| solara.DataFrame(df_output, items_per_page=10) | |
| if word.value in most_frequent: | |
| id_word = model.key_to_index.get(word.value) | |
| indexes_nn = [model.key_to_index.get(output[i][0]) for i in range(10) if model.key_to_index.get(output[i][0])<2000] | |
| color_array = ["words" for i in range(2000)] | |
| color_array[id_word] = "selected_word" | |
| for index in indexes_nn: | |
| color_array[index] = "neighbors" | |
| solara.AltairChart(plot_text(X, most_frequent, color_array=color_array).configure_range(category=['#0000ff', '#ff0000', '#a0aab4'])) | |
| else: | |
| solara.Info("Not a frequent word") | |
| solara.AltairChart(plot_text(X, most_frequent)) | |
| solara.Markdown("#Word-analogy") | |
| solara.InputText("Word1:", value=negative_word, continuous_update=True) | |
| my_check(negative_word.value) | |
| solara.InputText("Word2:", value=positive_word1, continuous_update=True) | |
| my_check(positive_word1.value) | |
| solara.InputText("Word3:", value=positive_word2, continuous_update=True) | |
| my_check(positive_word2.value) | |
| solara.Markdown(f"#{negative_word} is to {positive_word1} as {positive_word2} is to:") | |
| if positive_word1.value != "" and positive_word2.value != "" and negative_word.value != "" \ | |
| and positive_word1.value in list(data["words"].values) and positive_word2.value in list(data["words"].values) and negative_word.value in list(data["words"].values): | |
| output = model.most_similar(positive=[positive_word1.value, positive_word2.value], negative=[f'{negative_word.value}'], restrict_vocab=25000) | |
| df_output = pd.DataFrame() | |
| df_output["neighbors"] = [output[i][0] for i in range(10)] | |
| df_output["cosine similarity"] = [output[i][1] for i in range(10)] | |
| solara.DataFrame(df_output, items_per_page=10) | |