import numpy as np def most_similar(query_word, vocab, embeddings, top_n=3): # Get vector for the query word if query_word not in vocab: raise ValueError('Query word not in vocabulary!') query_vec = embeddings[vocab.index(query_word)] # Normalize query vector query_vec = query_vec / np.linalg.norm(query_vec) similarities = [] for word, vec in zip(vocab, embeddings): # Normalize each embedding vec_norm = vec / np.linalg.norm(vec) # Cosine similarity (since normalized) sim = np.dot(query_vec, vec_norm) similarities.append((word, sim)) # Sort by similarity, descending, skip the query word itself similarities = sorted(similarities, key=lambda x: x[1], reverse=True) return [w for w, s in similarities if w != query_word][:top_n] # Example usage: vocab = ['cat', 'dog', 'car', 'apple'] embeddings = np.array([ [1, 2], # cat [0.9, 2.1], # dog [-2, 0.5], # car [0.2, -1], # apple ]) print(most_similar('cat', vocab, embeddings, top_n=2)) # Output: ['dog', 'car']