import matplotlib.pyplot as plt from sklearn.manifold import TSNE from imblearn.over_sampling import SMOTE # Set the font family for matplotlib plt.rcParams['font.family'] = 'Verdana' # Split your data into features and labels X = df.drop('target', axis=1) y = df['target'] # Define the SMOTE resampling method smote = SMOTE(random_state=42) # Resample the data using SMOTE X_resampled, y_resampled = smote.fit_resample(X, y) # Use TSNE for dimensionalilty reduction X_tsne = TSNE(n_components=2).fit_transform(X) X_resampled_tsne = TSNE(n_components=2).fit_transform(X_resampled) # Get the class ratios original_class_ratio = np.bincount(y) / len(y) resampled_class_ratio = np.bincount(y_resampled) / len(y_resampled) # Create subplots fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(12, 6)) # Plot the original data ax1.scatter(X_tsne[:, 0], X_tsne[:, 1], c=y, cmap='rainbow',alpha=0.4) ax1.set_title(f'Original DatanClass Ratio: {original_class_ratio[1] * 100:.1f} : {original_class_ratio[0] * 100:.1f}') # Plot the resampled data ax2.scatter(X_resampled_tsne[:, 0], X_resampled_tsne[:, 1], c=y_resampled, cmap='rainbow',alpha=0.4) ax2.set_title(f'Resampled DatanClass Ratio: {resampled_class_ratio[1] * 100:.0f} : {resampled_class_ratio[0] * 100:.0f}') plt.show()