tokens = 32 kv_heads = 2 head_size = 8 one_layer_cache = 2 * tokens * kv_heads * head_size shared_cache = {"stored_scalars": one_layer_cache} layer_caches = [shared_cache, shared_cache] separate_total = 2 * one_layer_cache shared_total = shared_cache["stored_scalars"] print("layers point to same cache:", layer_caches[0] is layer_caches[1]) print("two separate layer caches:", separate_total, "scalars") print("one shared layer cache:", shared_total, "scalars") # Output: """ layers point to same cache: True two separate layer caches: 2048 scalars one shared layer cache: 1024 scalars """