import json from sagemaker.huggingface import HuggingFaceModel # sagemaker config instance_type = "ml.g5.12xlarge" number_of_gpu = 4 health_check_timeout = 300 # Define Model and Endpoint configuration parameter config = { 'HF_MODEL_ID': "/opt/ml/model", # path to where sagemaker stores the model 'SM_NUM_GPUS': json.dumps(number_of_gpu), # Number of GPU used per replica 'MAX_INPUT_LENGTH': json.dumps(8000), # Max length of input text 'MAX_BATCH_PREFILL_TOKENS': json.dumps(16384), # Number of tokens for the prefill operation. 'MAX_TOTAL_TOKENS': json.dumps(16384), # Max length of the generation (including input text) 'QUANTIZE': "awq" # Quantization method } # create HuggingFaceModel with the image uri llm_model = HuggingFaceModel( role=role, # path to s3 bucket with model, we are not using a compressed model model_data={'S3DataSource':{'S3Uri': s3_path + "/",'S3DataType': 'S3Prefix','CompressionType': 'None'}}, image_uri=llm_image, env=config )