%%bash num_gpus=1 model_id=philschmid/llama-3-1-8b-math-orca-spectrum-10k-ep1 # replace with your model id docker run --name tgi --gpus ${num_gpus} -d -ti -p 8080:80 --shm-size=2GB \ -e HF_TOKEN=$(cat ~/.cache/huggingface/token) \ ghcr.io/huggingface/text-generation-inference:3.0.1 \ --model-id ${model_id} \ --num-shard ${num_gpus}