lr = 0.2 # Learning rate: tells us how much we should update the weights with torch.no_grad(): W = W - lr * dW # Updating W with Gradient descent B = B - lr * dB # Updating B with Gradient descent