config: name: "path/to/sft_model" max_new_tokens: 300 # reasoning + answer token budget exploration_batchsize: 8 # number of questions per batch during rollout G: 6 # num responses per group temperature: 0.7 batch_size: 16 # minibatch size during training gradient_accumulation_steps: 12 learning_rate: 0.000001 # Advisable to keep this low, like 1e-6 or 1e-7 top_p: 0.95 buffer_size: 500