import gensim from gensim.utils import simple_preprocess import nltk nltk.download('stopwords') from nltk.corpus import stopwords stop_words = stopwords.words('english') stop_words.extend(['from', 'subject', 're', 'edu', 'use']) def sent_to_words(sentences): for sentence in sentences: # deacc=True removes punctuations yield(gensim.utils.simple_preprocess(str(sentence), deacc=True)) def remove_stopwords(texts): return [[word for word in simple_preprocess(str(doc)) if word not in stop_words] for doc in texts] data = papers.paper_text_processed.values.tolist() data_words = list(sent_to_words(data)) # remove stop words data_words = remove_stopwords(data_words) print(data_words[:1][0][:30])