# tell llmperf that we are using the messages api !MESSAGES_API=true python llmperf/token_benchmark_ray.py \ --model {llm.endpoint_name} \ --llm-api "sagemaker" \ --max-num-completed-requests 50 \ --timeout 600 \ --num-concurrent-requests 5 \ --results-dir "results"