from sklearn.feature_extraction.text import CountVectorizer from sklearn.naive_bayes import MultinomialNB from sklearn.model_selection import train_test_split from sklearn.metrics import accuracy_score # Simple SMS data. For real tasks, use much more data! texts = [ "Win a free iPhone now", "Call me as soon as you can", "Cheap meds available online", "Hey, are we meeting today?", "Congratulations! You won cash.", "Let's catch up for lunch." ] labels = ["spam", "ham", "spam", "ham", "spam", "ham"] # Indices to keep track of which text is which text_indices = list(range(len(texts))) # Step 1: Convert text to bag-of-words counts vectorizer = CountVectorizer() X = vectorizer.fit_transform(texts) # Show how a text line becomes a vector sample = texts[0] print(f"Sample text: '{sample}'") sample_vec = vectorizer.transform([sample]).toarray() print("Vector representation:", sample_vec) print("Vocabulary mapping:", vectorizer.vocabulary_) # Step 2: Train/test split (keep indices to trace back to text) X_train, X_test, y_train, y_test, idx_train, idx_test = train_test_split( X, labels, text_indices, test_size=0.5, random_state=42 ) # Step 3: Train Naive Bayes model nb_clf = MultinomialNB() nb_clf.fit(X_train, y_train) # Step 4: Make predictions and evaluate y_pred = nb_clf.predict(X_test) print("Naive Bayes accuracy:", accuracy_score(y_test, y_pred)) # Show predictions for each text for i in range(len(X_test)): text = texts[idx_test[i]] label = y_test[i] pred = y_pred[i] print(f"MSG: '{text}' | TRUE: {label} | NB-PREDICTED: {pred}")