kv_heads = 4 head_size = 8 standard_keys = kv_heads * head_size standard_values = kv_heads * head_size standard_total = standard_keys + standard_values mla_latent = 8 mla_rotary_key = 4 mla_total = mla_latent + mla_rotary_key print("standard cache:", standard_keys, "key +", standard_values, "value") print("MLA cache:", mla_latent, "latent +", mla_rotary_key, "rotary key") print("stored scalars per token:", standard_total, "->", mla_total) print(f"example reduction: {standard_total / mla_total:.2f}x") # Output: """ standard cache: 32 key + 32 value MLA cache: 8 latent + 4 rotary key stored scalars per token: 64 -> 12 example reduction: 5.33x """