docker run --gpus all -ti --shm-size 1g --ipc=host --rm -p 8000:80 \ -e MODEL_ID=meta-llama/Meta-Llama-3.1-8B-Instruct \ -e NUM_SHARD=1 \ -e MAX_INPUT_TOKENS=8000 \ -e MAX_TOTAL_TOKENS=8192 \ -e HF_TOKEN=$(cat ~/.cache/huggingface/token) \ ghcr.io/huggingface/text-generation-inference:2.2.0