# Specify hyperparameters training_args = TrainingArguments( output_dir=OUTPUT_DIR, # where checkpoints and logs are written eval_strategy="epoch", # run evaluation once per epoch save_strategy="epoch", # save checkpoint once per epoch per_device_train_batch_size=8, # samples per GPU per step per_device_eval_batch_size=16, # larger batch is fine — no gradients gradient_accumulation_steps=4, # effective batch = 8 × 4 = 32 num_train_epochs=15, # total passes over the training data learning_rate=1e-4, # peak LR after warmup bf16=True, # bfloat16 mixed precision optim="adamw_8bit", # 8-bit AdamW warmup_ratio=0.05, # first 5 % of steps ramp LR from 0 to peak lr_scheduler_type="cosine", # cosine decay from peak LR to ~0 logging_steps=25, # print loss/LR to console every 25 steps logging_first_step=True, # also log step 1 to catch early instability load_best_model_at_end=True, # restore best checkpoint after training ends metric_for_best_model="macro_f1", # criterion used to select the best checkpoint greater_is_better=True, # higher macro_f1 is better in evaluation gradient_checkpointing=False, remove_unused_columns=False, # keep input_embeds column save_total_limit=15, # keep all checkpoints on disk to load the best model weight_decay=0.01, # L2 regularisation on all trainable parameters )