from sklearn.feature_extraction.text import CountVectorizer bigram_vectorizer = CountVectorizer( ngram_range=(2, 2), # only bigrams max_features=1000 # top 1000 bigrams by frequency )