# login into the huggingface hub to access gated models, like llama huggingface-cli login --token [API_TOKEN] # compile model with optimum for batch size 4 and sequence length 2048 optimum-cli export neuron -m meta-llama/Meta-Llama-3-70B-Instruct --batch_size 4 --sequence_length 2048 --num_cores 24 --auto_cast_type fp16 ./llama-70b-chat-neuron # push model to hub [repo_id] [local_path] [path_in_repo] huggingface-cli upload aws-neuron/Llama-3-70b-chat-seqlen-2048-bs-4 ./llama-70b-chat-neuron ./ --exclude "checkpoint/**" # Move tokenizer to neuron model repository python -c "from transformers import AutoTokenizer; AutoTokenizer.from_pretrained('meta-llama/Meta-Llama-3-70B-Instruct').push_to_hub('aws-neuron/Llama-3-70b-chat-seqlen-2048-bs-4')"