# Response is an array of string of length [B*G] # B is the number of questions, G is the number of responses per question correctness_reward = score_fn(response, validation_object) format_reward = calculate_format_reward(response) # Total reward is a weighted sum of correctness and formatting rewards reward = correctness_reward * 0.85 + format_reward * 0.15 # Convert rewards from [B*G, 1] -> [B, G] rewards = rewards.reshape(B, G) # Calculate advantages advantages = (rewards - np.mean(rewards, axis=1, keepdims=True)) / ( np.std(rewards, axis=1, keepdims=True) + 1e-8 ) advantages = advantages.reshape(-1, 1)