TensorFlow: Text Classification
Generated Python
Source-only: run it in Google Colab
# Generated by FEMOS Python Blocks
# Run this file in Google Colab. FEMOS does not execute it on the server.
import pathlib
import shutil
import matplotlib.pyplot as plt
import tensorflow as tf
SEED = 42
tf.keras.utils.set_random_seed(SEED)
IMDB_URL = "https://ai.stanford.edu/~amaas/data/sentiment/aclImdb_v1.tar.gz"
archive_path = tf.keras.utils.get_file(
"aclImdb_v1", IMDB_URL, untar=True, cache_dir=".", cache_subdir=""
)
archive_path = pathlib.Path(archive_path)
dataset_candidates = [
archive_path / "aclImdb",
archive_path,
archive_path.parent / "aclImdb",
]
DATA_DIR = next(
(path for path in dataset_candidates if (path / "train").is_dir()),
None,
)
if DATA_DIR is None:
discovered_train_dirs = list(pathlib.Path(".").glob("**/aclImdb/train"))
if not discovered_train_dirs:
raise FileNotFoundError("IMDb was downloaded, but its train folder was not found.")
DATA_DIR = discovered_train_dirs[0].parent
shutil.rmtree(DATA_DIR / "train" / "unsup", ignore_errors=True)
print("Movie reviews ready:", DATA_DIR)
VALIDATION_SPLIT = 0.2
train_data = tf.keras.utils.text_dataset_from_directory(
DATA_DIR / "train", validation_split=VALIDATION_SPLIT, subset="training",
seed=SEED, batch_size=32, label_mode="binary",
)
validation_data = tf.keras.utils.text_dataset_from_directory(
DATA_DIR / "train", validation_split=VALIDATION_SPLIT, subset="validation",
seed=SEED, batch_size=32, label_mode="binary",
)
test_data = tf.keras.utils.text_dataset_from_directory(
DATA_DIR / "test", batch_size=32, label_mode="binary", shuffle=False,
)
print("Split: 80% training, 20% validation")
VOCAB_SIZE = 5000
SEQUENCE_LENGTH = 100
vectorizer = tf.keras.layers.TextVectorization(
max_tokens=VOCAB_SIZE, output_mode="int",
output_sequence_length=SEQUENCE_LENGTH,
)
training_text = train_data.map(lambda text, label: text)
vectorizer.adapt(training_text)
print("Vocabulary learned from training text only.")
example_text = "I really loved this movie"
print("Original sentence:", example_text)
print("Token IDs:", vectorizer(tf.constant([example_text])).numpy()[0])
EMBEDDING_DIM = 32
HIDDEN_UNITS = 32
model = tf.keras.Sequential([
tf.keras.Input(shape=(), dtype=tf.string),
vectorizer,
tf.keras.layers.Embedding(VOCAB_SIZE, EMBEDDING_DIM),
tf.keras.layers.GlobalAveragePooling1D(),
tf.keras.layers.Dense(HIDDEN_UNITS, activation="relu"),
tf.keras.layers.Dense(1, activation="sigmoid"),
])
model.compile(optimizer="adam", loss="binary_crossentropy", metrics=["accuracy"])
model.summary()
EPOCHS = 5
BATCH_SIZE = 32
train_data = train_data.unbatch().batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)
validation_data = validation_data.unbatch().batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)
history = model.fit(train_data, validation_data=validation_data, epochs=EPOCHS)
test_loss, test_accuracy = model.evaluate(test_data, verbose=0)
print(f"Test accuracy: {test_accuracy:.2%}")
epochs = range(1, len(history.history["accuracy"]) + 1)
plt.plot(epochs, history.history["accuracy"], marker="o", label="Training")
plt.plot(epochs, history.history["val_accuracy"], marker="o", label="Validation")
plt.title("Text Classifier Accuracy")
plt.xlabel("Epoch")
plt.ylabel("Accuracy")
plt.legend()
plt.grid(alpha=0.3)
plt.show()
sentence = "I loved this movie"
positive_probability = float(model.predict(tf.constant([sentence]), verbose=0)[0][0])
prediction = "POSITIVE" if positive_probability >= 0.5 else "NEGATIVE"
confidence = positive_probability if prediction == "POSITIVE" else 1 - positive_probability
print(f"{sentence}\nPrediction: {prediction} ({confidence:.1%} confidence)")
Loading Python AI blocks…