import numpy as np import matplotlib.pyplot as plt from sklearn.linear_model import LinearRegression from sklearn.preprocessing import PolynomialFeatures from sklearn.model_selection import train_test_split from sklearn.metrics import r2_score # Generate random data np.random.seed(0) X = np.linspace(-3, 3, 100) y = 0.5 * X**2 + X + np.random.normal(0, 1, 100) # Reshape the input data X = X.reshape(-1, 1) # Normalize the data X_normalized = (X - np.mean(X)) / np.std(X) y_normalized = (y - np.mean(y)) / np.std(y) # Split the data into train and test sets X_train, X_test, y_train, y_test = train_test_split( X_normalized, y_normalized, test_size=0.2, random_state=0 ) # Plot the original data plt.scatter(X_normalized, y_normalized, color='b', label='Data') # Fit polynomial regression models of different degrees degrees = [1, 4, 15] colors = ['r', 'g', 'm'] for degree, color in zip(degrees, colors): # Create polynomial features polynomial_features = PolynomialFeatures(degree=degree) X_poly_train = polynomial_features.fit_transform(X_train) X_poly_test = polynomial_features.transform(X_test) # Fit the polynomial regression model model = LinearRegression() model.fit(X_poly_train, y_train) # Predict the values for train and test sets y_pred_train = model.predict(X_poly_train) y_pred_test = model.predict(X_poly_test) # Calculate R² scores r2_train = r2_score(y_train, y_pred_train) r2_test = r2_score(y_test, y_pred_test) # Plot the fitted curve plt.plot(X_normalized, model.predict(polynomial_features.transform(X_normalized)), color=color, linewidth=2, label=f'Degree {degree} (Train R²={r2_train:.2f}, Test R²={r2_test:.2f})') # Add labels and title to the plot plt.xlabel('X') plt.ylabel('y') plt.title('Polynomial Regression - Overfitting Example (Normalized)') plt.legend(loc='upper left') # Display the plot plt.show()