# precompilation command !MALLOC_ARENA_MAX=64 neuron_parallel_compile torchrun --nproc_per_node=32 scripts/run_clm.py \ --model_id {model_id} \ --dataset_path {dataset_path} \ --bf16 True \ --learning_rate 5e-5 \ --output_dir dolly_llama \ --overwrite_output_dir True \ --per_device_train_batch_size 1 \ --gradient_checkpointing True \ --tensor_parallel_size 8 \ --max_steps 10 \ --logging_steps 10 \ --gradient_accumulation_steps 16