# Reshape for multi-head attention batch_size, seq_len = 1, 6 Q = Q.view(batch_size, seq_len, num_heads, d_k).transpose(1, 2) # [1, 8, 6, 8] K = K.view(batch_size, seq_len, num_heads, d_k).transpose(1, 2) # [1, 8, 6, 8] V = V.view(batch_size, seq_len, num_heads, d_k).transpose(1, 2) # [1, 8, 6, 8] # Attention scores: How much should each word pay attention to others? scores = torch.matmul(Q, K.transpose(-2, -1)) / (d_k ** 0.5) print(f"Attention scores shape: {scores.shape}") # [1, 8, 6, 6] # Example: How much does "cat" attend to each word? cat_attention = scores[0, 0, 1, :] # First head, "cat" position words = ["the", "cat", "sat", "on", "the", "mat"] for i, word in enumerate(words): print(f"cat -> {word}: {cat_attention[i]:.3f}")