import numpy as np import pandas as pd from sklearn.model_selection import train_test_split from sklearn.linear_model import LinearRegression from sklearn.metrics import r2_score from sklearn.preprocessing import StandardScaler # Generating random features with at least 50 occurrences each np.random.seed(42) feature1 = np.random.randint(0, 10, size=100) feature2 = np.random.randint(0, 10, size=100) feature3 = np.random.randint(0, 10, size=100) feature4 = np.random.randint(0, 10, size=100) feature5 = np.random.randint(0, 10, size=100) target = np.random.randint(0, 100, size=100) # Creating a Pandas data frame data = { 'Feature1': feature1, 'Feature2': feature2, 'Feature3': feature3, 'Feature4': feature4, 'Feature5': feature5, 'Target': target } df = pd.DataFrame(data) # Splitting the data into features and target X = df.drop('Target', axis=1) y = df['Target'] # Normalizing the data using StandardScaler scaler = StandardScaler() X_scaled = scaler.fit_transform(X) # Splitting the normalized data into train and test sets X_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.2, random_state=42) # Fitting the linear regression model reg_model = LinearRegression() reg_model.fit(X_train, y_train) # Predicting the target variable for both train and test sets y_train_pred = reg_model.predict(X_train) y_test_pred = reg_model.predict(X_test) # Calculating R² scores for train and test sets r2_train = r2_score(y_train, y_train_pred) r2_test = r2_score(y_test, y_test_pred) # Printing the R² scores print("R² score for train set:", r2_train) print("R² score for test set:", r2_test)