[](https://hamel.dev/notes/llm/inference/inference.html#cb13-1)model_id = "meta-llama/Llama-2-7b-hf" [](https://hamel.dev/notes/llm/inference/inference.html#cb13-2)tokenizer = AutoTokenizer.from_pretrained(model_id) [](https://hamel.dev/notes/llm/inference/inference.html#cb13-3)nf4_config = BitsAndBytesConfig( [](https://hamel.dev/notes/llm/inference/inference.html#cb13-4) load_in_4bit=True, [](https://hamel.dev/notes/llm/inference/inference.html#cb13-5) bnb_4bit_quant_type="nf4", [](https://hamel.dev/notes/llm/inference/inference.html#cb13-6) bnb_4bit_compute_dtype=torch.bfloat16 [](https://hamel.dev/notes/llm/inference/inference.html#cb13-7)) [](https://hamel.dev/notes/llm/inference/inference.html#cb13-8)model_nf4 = AutoModelForCausalLM.from_pretrained(model_id, quantization_config=nf4_config)