#convert the preprocessed text to a list documents = list(data["clean_text"]) #seperate by ' ' to tokenize each article texts = [x.split(' ') for x in documents] #construct word ID mappings id2word = Dictionary(texts) #use word ID mappings to build corpus corpus = [id2word.doc2bow(text) for text in texts]