from sklearn.feature_extraction.text import TfidfVectorizer # Your documents documents = [ "The cat sat on the mat", "The cat sat on the hat", "The dog lay on the mat" ] # Create and fit the vectorizer vectorizer = TfidfVectorizer() tfidf_matrix = vectorizer.fit_transform(documents) # See the vocabulary print(vectorizer.get_feature_names_out()) # ['cat', 'dog', 'hat', 'lay', 'mat', 'on', 'sat', 'the'] # See the TF-IDF scores for document 1 print(tfidf_matrix[0].toarray())