def preprocess(text): """ This function takes in a string, coverts it to lowercase, cleans it (remove special character and numbers), and tokenizes it. """ #convert to lowercase text = text.lower() #remove special character and digits text = re.sub(r'd+', '', text) text = re.sub(r'[^ws]', '', text) #tokenize tokens = nltk.word_tokenize(text) return(tokens)