model="/home/ubuntu/test-gptq" num_shard=1 quantize="gptq" max_input_length=1562 max_total_tokens=4096 # 4096 !docker run --gpus all -ti -p 8080:80 \ -e MODEL_ID=$model \ -e QUANTIZE=$quantize \ -e NUM_SHARD=$num_shard \ -e MAX_INPUT_LENGTH=$max_input_length \ -e MAX_TOTAL_TOKENS=$max_total_tokens \ -v $model:$model \ ghcr.io/huggingface/text-generation-inference:1.0.3