#defining sample value matrix #[batch_size x sequence_len x (query_key_dim * n_heads)] #in this matrix, [0,1,2,3] represents the values for 2 heads across a single word vector samp_val = torch.tensor([[[0,1,2,3],[4,5,6,7]],[[0,-1,-2,-3],[-4,-5,-6,-7]]]) #dividing into two heads #[batch_size x sequence_len x query_key_dim x n_heads] samp_val = samp_val.view(2,2,2,2) #moving the head dimension next to the batch dimension #[batch_size x n_heads x sequence_len x query_key_dim] samp_val = samp_val.permute(0, 3, 1, 2) #combining batch and head dimension #[batch_size*n_heads x sequence_len x query_key_dim] samp_val = samp_val.reshape(-1, 2, 2) #that would be the input into mhsa, which would give back the same shape output #now I want to unpack the mhsa back into the original shape #[batch_size x sequence_len x (query_key_dim * n_heads)] #if I do this right, the values should be exactly identical #seperating heads #[batch_size x n_heads x sequence_len x query_key_dim] samp_val = samp_val.reshape(2,2,2,2) #moving the head dimension to the end #[batch_size x sequence_len x query_key_dim x n_heads] samp_val = samp_val.permute(0, 2, 3, 1) #combining the last dim to effectively concatonate the result of the heads #[batch_size x sequence_len x query_key_dim*n_heads] samp_val = samp_val.reshape(2, 2, -1) samp_val