[](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-1)def download_model_to_folder(): [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-2) from huggingface_hub import snapshot_download [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-3) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-4) snapshot_download( [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-5)- "meta-llama/Llama-2-13b-chat-hf", [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-6)+ "meta-llama/Llama-2-70b-chat-hf", [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-7) local_dir="/model", [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-8) token=os.environ["HUGGINGFACE_TOKEN"], [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-9) ) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-10) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-11)image = ( [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-12) Image.from_dockerhub("nvcr.io/nvidia/pytorch:22.12-py3") [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-13) .pip_install("torch==2.0.1", index_url="https://download.pytorch.org/whl/cu118") [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-14)+ # Pin vLLM to 8/2/2023 [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-15)+ .pip_install("vllm @ git+https://github.com/vllm-project/vllm.git@79af7e96a0e2fc9f340d1939192122c3ae38ff17") [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-16)- # Pin vLLM to 07/19/2023 [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-17)- .pip_install("vllm @ git+https://github.com/vllm-project/vllm.git@bda41c70ddb124134935a90a0d51304d2ac035e8") [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-18) # Use the barebones hf-transfer package for maximum download speeds. No progress bar, but expect 700MB/s. [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-19)- .pip_install("hf-transfer~=0.1") [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-20)+ #Force a rebuild to invalidate the cache (you can remove `force_build=True` after the first time) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-21)+ .pip_install("hf-transfer~=0.1", force_build=True) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-22) .run_function( [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-23) download_model_to_folder, [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-24) secret=Secret.from_name("huggingface"), [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-25) timeout=60 * 20) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-26)) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-27)... [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-28) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-29)-@stub.cls(gpu="A100", secret=Secret.from_name("huggingface")) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-30)+# You need a minimum of 4 A100s that are the 40GB version [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-31)+@stub.cls(gpu=gpu.A100(count=4, memory=40), secret=Secret.from_name("huggingface")) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-32)class Model: [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-33) def __enter__(self): [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-34) from vllm import LLM [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-35) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-36) # Load the model. Tip: MPT models may require `trust_remote_code=true`. [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-37)- self.llm = LLM(MODEL_DIR) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-38)+ self.llm = LLM(MODEL_DIR, tensor_parallel_size=4) [](https://hamel.dev/notes/llm/inference/big_inference.html#cb1-39)...