from transformers import AutoModel import deep_gemm model = AutoModel.from_pretrained("meta-llama/Llama-2-7b").cuda() # 替换 FFN 层的矩阵乘法 def fp8_linear(x, weight, scales_x, scales_w): # x: [seq_len, hidden_dim], weight: [hidden_dim, out_dim] x_fp8 = (x.to(torch.float8_e4m3fn), scales_x) # 伪代码,需实际量化 w_fp8 = (weight.to(torch.float8_e4m3fn), scales_w) out = torch.empty(x.shape[0], weight.shape[1], dtype=torch.bfloat16, device="cuda") deep_gemm.gemm_fp8_fp8_bf16_nt(x_fp8, w_fp8, out) return out # 替换原模型的 forward 计算 model.layers[0].mlp.dense_h_to_4h = fp8_linear