def backward(self, dy, caches, clip_value=1.0): """ Backward pass through the GRU network. Parameters: - dy: np.ndarray, gradient of the loss with respect to the output - caches: list, caches from the forward pass - clip_value: float, value to clip gradients to (default: 1.0) Returns: - tuple, gradients of the loss with respect to the parameters """ # Initialize gradients dWz, dWr, dWh = [np.zeros_like(w) for w in (self.wz, self.wr, self.wh)] dbz, dbr, dbh = [np.zeros_like(b) for b in (self.bz, self.br, self.bh)] dWhy = np.zeros_like(self.why) dby = np.zeros_like(self.by) # Ensure dy is reshaped to match output size dy = dy.reshape(self.output_size, -1) dh_next = np.zeros((self.hidden_size, 1)) # shape must match hidden_size # Backpropagate through time for cache in reversed(caches): h_prev, z, r, h_, x_t, combined, combined_r, h = cache # Add gradient from next step to current output gradient dh = np.dot(self.why.T, dy) + dh_next dh_ = dh * z * self.dtanh(h_) dz = dh * (h_ - h_prev) * self.dsigmoid(z) dr = np.dot(self.wh[:, :self.hidden_size].T, dh_) * h_prev * self.dsigmoid(r) # Update gradients with respect to weights and biases dWz += np.dot(dz, combined.T) dWr += np.dot(dr, combined.T) dWh += np.dot(dh_, combined_r.T) dbz += dz.sum(axis=1, keepdims=True) dbr += dr.sum(axis=1, keepdims=True) dbh += dh_.sum(axis=1, keepdims=True) # Compute the gradient for the previous time step dh_prev = dh * (1 - z) + np.dot(self.wz[:, :self.hidden_size].T, dz) + np.dot(self.wr[:, :self.hidden_size].T, dr) dh_next = dh_prev # Update output layer weights and biases dWhy += np.dot(dy, h.T) dby += dy gradients = (dWz, dWr, dWh, dbz, dbr, dbh, dWhy, dby) # Gradient clipping for i in range(len(gradients)): np.clip(gradients[i], -clip_value, clip_value, out=gradients[i]) return gradients