total_rewards.append(total_reward) self.update_policy(log_probs, rewards) print(f"Episode {episode}, Total Reward: {total_reward}") if episode % 5 == 0 and episode > 0: print(f"Episode {episode}, Average Reward: {sum(total_rewards) / len(total_rewards)}") if save_model: torch.save(self.policy_network.state_dict(), self.model_path) print("Saved model to disk") del log_probs, rewards, state, action_probs, action gc.collect()