# The model that you want to train from the Hugging Face hubmodel_name = "NousResearch/Llama-2-7b-chat-hf"# The instruction dataset to usedataset_name = "mlabonne/guanaco-llama2-1k"# Fine-tuned model namenew_model = "llama-2-7b-miniguanaco"################################################################################# QLoRA parameters################################################################################# LoRA attention dimensionlora_r = 64# Alpha parameter for LoRA scalinglora_alpha = 16# Dropout probability for LoRA layerslora_dropout = 0.1################################################################################# bitsandbytes parameters################################################################################# Activate 4-bit precision base model loadinguse_4bit = True# Compute dtype for 4-bit base modelsbnb_4bit_compute_dtype = "float16"# Quantization type (fp4 or nf4)bnb_4bit_quant_type = "nf4"# Activate nested quantization for 4-bit base models (double quantization)use_nested_quant = False################################################################################# TrainingArguments parameters################################################################################# Output directory where the model predictions and checkpoints will be storedoutput_dir = "./results"# Number of training epochsnum_train_epochs = 1# Enable fp16/bf16 training (set bf16 to True with an A100)fp16 = Falsebf16 = False# Batch size per GPU for trainingper_device_train_batch_size = 4# Batch size per GPU for evaluationper_device_eval_batch_size = 4# Number of update steps to accumulate the gradients forgradient_accumulation_steps = 1# Enable gradient checkpointinggradient_checkpointing = True# Maximum gradient normal (gradient clipping)max_grad_norm = 0.3# Initial learning rate (AdamW optimizer)learning_rate = 2e-4# Weight decay to apply to all layers except bias/LayerNorm weightsweight_decay = 0.001# Optimizer to useoptim = "paged_adamw_32bit"# Learning rate schedule (constant a bit better than cosine)lr_scheduler_type = "constant"# Number of training steps (overrides num_train_epochs)max_steps = -1# Ratio of steps for a linear warmup (from 0 to learning rate)warmup_ratio = 0.03# Group sequences into batches with same length# Saves memory and speeds up training considerablygroup_by_length = True# Save checkpoint every X updates stepssave_steps = 25# Log every X updates stepslogging_steps = 25################################################################################# SFT parameters################################################################################# Maximum sequence length to usemax_seq_length = None# Pack multiple short examples in the same input sequence to increase efficiencypacking = False# Load the entire model on the GPU 0device_map = {"": 0}