from gensim.models import FastText # Simple dataset: each sentence is a list of words. corpus = [ ["cat", "sat", "on", "the", "mat"], ["dog", "sat", "beside", "the", "cat"], ["dogs", "and", "cats", "are", "friends"], ["the", "dog", "jumped", "over", "the", "log"] ] # Train a FastText model. Kept small for clarity. model = FastText(sentences=corpus, vector_size=10, window=3, min_count=1) # Common word print("Vector for 'cat':", model.wv['cat']) # Rare word (seen once) print("Vector for 'jumped':", model.wv['jumped']) # Plausible word never seen in training print("Vector for 'jumping':", model.wv['jumping']) # Made-up or misspelled word print("Vector for 'cattz':", model.wv['cattz'])