# Select all bad parts bad_parts = data[data['Label'] == 'Bad'] # Randomly select an equal number of good parts good_parts = data[data['Label'] == 'Good'].sample(n=len(bad_parts), random_state=42) # Combine both subsets to create a balanced dataset balanced_data = pd.concat([bad_parts, good_parts]) # Shuffle the combined dataset balanced_data = balanced_data.sample(frac=1, random_state=42).reset_index(drop=True) # Display the number of good and bad parts in the balanced dataset print("Number of Good and Bad Parts in the balanced dataset:") print(balanced_data['Label'].value_counts())