Femos AI
Python · source-only · copy to ColabPython Core1 imported Sign in to save
TensorFlow: Text Classification

Generated Python

Source-only: run it in Google Colab

# Generated by FEMOS Python Blocks
# Run this file in Google Colab. FEMOS does not execute it on the server.

import pathlib
import shutil

import matplotlib.pyplot as plt

import tensorflow as tf

SEED = 42
tf.keras.utils.set_random_seed(SEED)

IMDB_URL = "https://ai.stanford.edu/~amaas/data/sentiment/aclImdb_v1.tar.gz"
archive_path = tf.keras.utils.get_file(
    "aclImdb_v1", IMDB_URL, untar=True, cache_dir=".", cache_subdir=""
)
archive_path = pathlib.Path(archive_path)
dataset_candidates = [
    archive_path / "aclImdb",
    archive_path,
    archive_path.parent / "aclImdb",
]
DATA_DIR = next(
    (path for path in dataset_candidates if (path / "train").is_dir()),
    None,
)
if DATA_DIR is None:
    discovered_train_dirs = list(pathlib.Path(".").glob("**/aclImdb/train"))
    if not discovered_train_dirs:
        raise FileNotFoundError("IMDb was downloaded, but its train folder was not found.")
    DATA_DIR = discovered_train_dirs[0].parent
shutil.rmtree(DATA_DIR / "train" / "unsup", ignore_errors=True)
print("Movie reviews ready:", DATA_DIR)

VALIDATION_SPLIT = 0.2
train_data = tf.keras.utils.text_dataset_from_directory(
    DATA_DIR / "train", validation_split=VALIDATION_SPLIT, subset="training",
    seed=SEED, batch_size=32, label_mode="binary",
)
validation_data = tf.keras.utils.text_dataset_from_directory(
    DATA_DIR / "train", validation_split=VALIDATION_SPLIT, subset="validation",
    seed=SEED, batch_size=32, label_mode="binary",
)
test_data = tf.keras.utils.text_dataset_from_directory(
    DATA_DIR / "test", batch_size=32, label_mode="binary", shuffle=False,
)
print("Split: 80% training, 20% validation")

VOCAB_SIZE = 5000
SEQUENCE_LENGTH = 100
vectorizer = tf.keras.layers.TextVectorization(
    max_tokens=VOCAB_SIZE, output_mode="int",
    output_sequence_length=SEQUENCE_LENGTH,
)

training_text = train_data.map(lambda text, label: text)
vectorizer.adapt(training_text)
print("Vocabulary learned from training text only.")

example_text = "I really loved this movie"
print("Original sentence:", example_text)
print("Token IDs:", vectorizer(tf.constant([example_text])).numpy()[0])

EMBEDDING_DIM = 32
HIDDEN_UNITS = 32
model = tf.keras.Sequential([
    tf.keras.Input(shape=(), dtype=tf.string),
    vectorizer,
    tf.keras.layers.Embedding(VOCAB_SIZE, EMBEDDING_DIM),
    tf.keras.layers.GlobalAveragePooling1D(),
    tf.keras.layers.Dense(HIDDEN_UNITS, activation="relu"),
    tf.keras.layers.Dense(1, activation="sigmoid"),
])
model.compile(optimizer="adam", loss="binary_crossentropy", metrics=["accuracy"])
model.summary()

EPOCHS = 5
BATCH_SIZE = 32
train_data = train_data.unbatch().batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)
validation_data = validation_data.unbatch().batch(BATCH_SIZE).prefetch(tf.data.AUTOTUNE)
history = model.fit(train_data, validation_data=validation_data, epochs=EPOCHS)
test_loss, test_accuracy = model.evaluate(test_data, verbose=0)
print(f"Test accuracy: {test_accuracy:.2%}")

epochs = range(1, len(history.history["accuracy"]) + 1)
plt.plot(epochs, history.history["accuracy"], marker="o", label="Training")
plt.plot(epochs, history.history["val_accuracy"], marker="o", label="Validation")
plt.title("Text Classifier Accuracy")
plt.xlabel("Epoch")
plt.ylabel("Accuracy")
plt.legend()
plt.grid(alpha=0.3)
plt.show()

sentence = "I loved this movie"
positive_probability = float(model.predict(tf.constant([sentence]), verbose=0)[0][0])
prediction = "POSITIVE" if positive_probability >= 0.5 else "NEGATIVE"
confidence = positive_probability if prediction == "POSITIVE" else 1 - positive_probability
print(f"{sentence}\nPrediction: {prediction} ({confidence:.1%} confidence)")

Loading Python AI blocks…