Skip to content

About

Word2vec algorithm on Rick and Morty dataset using word embeddings

Resources

Stars

1 star

Watchers

1 watching

Forks

Latest commit

 

History

2 Commits

Folders and files

Repository files navigation

NLPword2vec

Word2vec algorithm on Rick and Morty dataset using word embeddings

import numpy as np # linear algebra import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv) import re # For preprocessing import pandas as pd # For data handling from time import time # To time our operations from collections import defaultdict # For word frequency

import spacy # For preprocessing

import logging # Setting up the loggings to monitor gensim logging.basicConfig(format="%(levelname)s - %(asctime)s: %(message)s", datefmt= '%H:%M:%S', level=logging.INFO)

#automatically detect common phrases (bigrams) from a list of sentences. from gensim.models.phrases import Phrases, Phraser

from gensim.models import Word2Vec

df = pd.read_csv('../input/rickmorty-scripts/RickAndMortyScripts.csv') df.shape

#to check missing values df.isnull().sum()

#preprocess the data brief_cleaning = (re.sub("[^A-Za-z']+", ' ', str(row)).lower() for row in df['line'])

sent = [row.split() for row in df_clean['clean']]

phrases = Phrases(sent, min_count=30, progress_per=10000)

bigram = Phraser(phrases) print(bigram)

pre_sentences = bigram[sent]

#build the model w2v_model = Word2Vec(min_count=20, window=2, vector_size=300, sample=6e-5, alpha=0.03, min_alpha=0.0007, negative=20, )

t = time()

w2v_model.build_vocab(sentences, progress_per=10000)

print('Time to build vocab: {} mins'.format(round((time() - t) / 60, 2)))

t = time()

w2v_model.train(sentences, total_examples=w2v_model.corpus_count, epochs=30, report_delay=1)

print('Time to train the model: {} mins'.format(round((time() - t) / 60, 2)))

w2v_model.wv.most_similar(positive=["rick"])

w2v_model.wv.most_similar('morty')

w2v_model.wv.save_word2vec_format('output.txt')

About

Word2vec algorithm on Rick and Morty dataset using word embeddings

Resources

Stars

1 star

Watchers

1 watching

Forks

Releases

Packages

Contributors

Languages