vectorizer = TfidfVectorizer( stop_words='english', # Remove common English stop words max_features=10000, # Only keep the 10,000 most frequent words ngram_range=(1, 2), # Include both single words and two-word phrases min_df=2 # Ignore words that appear in fewer than 2 documents )