import numpy as np # Example 'word' vectors (embeddings): 3 words, each with 2 features word_vectors = np.array([ [1.0, 0.0], # word 1 [0.0, 1.0], # word 2 [1.0, 1.0] # word 3 ]) # In scaled dot-product attention, you have 'queries', 'keys', and 'values' # For simplicity, we'll just use the word_vectors for all three Q = word_vectors K = word_vectors V = word_vectors # Compute raw attention scores: Q @ K.T attention_scores = np.dot(Q, K.T) # shape (3, 3) # Scale scores (normally by sqrt of vector size) d_k = Q.shape[1] scaled_scores = attention_scores / np.sqrt(d_k) # Softmax to get attention weights for each word def softmax(x): exp_x = np.exp(x - np.max(x, axis=-1, keepdims=True)) return exp_x / np.sum(exp_x, axis=-1, keepdims=True) attention_weights = softmax(scaled_scores) # shape (3, 3) # Weighted sum: each row is the new context vector for a word context_vectors = np.dot(attention_weights, V) print("Attention Weights:\n", attention_weights) print("Context Vectors:\n", context_vectors)