"""Breaking wikipedia articles into sentences and paragraphs """ import itertools num_articles = 10000 #getting n articles articles = list(itertools.islice(dataset_iter, num_articles)) #getting paragraphs paragraphs = [] for article in articles: paragraphs.extend(article['text'].splitlines()) #filtering paragraphs so they're hopefully actually paragraphs paragraps = [p for p in paragraphs if len(p)>50] #dividing paragraphs into sentences divided_paragraphs = [] for p in paragraphs: divided_paragraphs.append(nltk.sent_tokenize(p)) #only using paragraphs with 3 or more sentences divided_paragraphs = [pls for pls in divided_paragraphs if len(pls)>=3] divided_paragraphs