import numpy as np import urllib.request import zipfile import os # Download a small GloVe embedding set (50 dimensions, ~70MB zipped) glove_url = "http://nlp.stanford.edu/data/glove.6B.zip" glove_zip_path = "glove.6B.zip" if not os.path.exists(glove_zip_path): urllib.request.urlretrieve(glove_url, glove_zip_path) # Extract the smallest file (glove.6B.50d.txt) with zipfile.ZipFile(glove_zip_path, 'r') as z: if not os.path.exists("glove.6B.50d.txt"): z.extract("glove.6B.50d.txt") # Load GloVe vectors into a dictionary embeddings = {} with open("glove.6B.50d.txt", encoding="utf-8") as f: for line in f: parts = line.strip().split() word = parts[0] vec = np.array(parts[1:], dtype=np.float32) embeddings[word] = vec # Compute cosine similarity between two words def cosine_similarity(x, y): return np.dot(x, y) / (np.linalg.norm(x) * np.linalg.norm(y)) word1, word2 = "king", "queen" vec1 = embeddings[word1] vec2 = embeddings[word2] sim = cosine_similarity(vec1, vec2) print(f"Cosine similarity between '{word1}' and '{word2}': {sim:.3f}")