import tensorflow as tf from tensorflow.keras import layers, models # Sample data - in practice, load from your dataset texts = tf.constant([ "i love this product amazing quality", "terrible service very disappointed", "excellent customer support helpful staff", "poor quality broke immediately", "fantastic experience highly recommend", "worst purchase ever complete waste" ]) labels = tf.constant([1, 0, 1, 0, 1, 0], dtype=tf.float32) # Configuration max_tokens = 10000 sequence_length = 20 embedding_dim = 64 # Text preprocessing pipeline text_vectorizer = layers.TextVectorization( max_tokens=max_tokens, output_mode="int", output_sequence_length=sequence_length, standardize="lower_and_strip_punctuation" ) # Fit the vectorizer on training data text_vectorizer.adapt(texts.batch(8)) # Model architecture def create_text_classifier(): # Input: raw strings text_input = layers.Input(shape=(), dtype=tf.string, name='text') # Convert to integer sequences x = text_vectorizer(text_input) # Shape: (batch, sequence_length) # Embed tokens into dense vectors x = layers.Embedding( input_dim=max_tokens, output_dim=embedding_dim, mask_zero=True, # Handle padding name='token_embedding' )(x) # Shape: (batch, sequence_length, embedding_dim) # Pool across sequence dimension x = layers.GlobalAveragePooling1D()(x) # Shape: (batch, embedding_dim) # Classification head x = layers.Dense(128, activation='relu')(x) x = layers.Dropout(0.2)(x) output = layers.Dense(1, activation='sigmoid', name='prediction')(x) return models.Model(text_input, output) # Build and compile model model = create_text_classifier() model.compile( optimizer='adam', loss='binary_crossentropy', metrics=['accuracy', 'auc'] ) # Train the model history = model.fit( texts, labels, batch_size=2, epochs=10, validation_split=0.2, verbose=1 ) # Inspect learned embeddings embedding_layer = model.get_layer('token_embedding') vocab = text_vectorizer.get_vocabulary() # Get embeddings for specific tokens token_index = vocab.index('excellent') token_embedding = embedding_layer.get_weights()[0][token_index] print(f"Embedding for 'excellent': {token_embedding[:5]}...") __ __