To perform sentiment analysis on the IMDB movie review dataset, a recurrent neural network (LSTM) is built using MindSpore. The task is a binary classification problem: Positive or Negative. The workflow covers data acquisition, preprocessing, model construction, training, evaluation, and inference on custom input sentences.
1. Data Acquisition
The IMDB dataset is fetched from a remote server using a custom downloader that supports progress visualization and temporary file handling.
import os
import shutil
import requests
import tempfile
from tqdm import tqdm
from pathlib import Path
DATA_HOME = Path.home() / '.mindspore_examples'
def fetch_file(url, temp_fp):
resp = requests.get(url, stream=True)
total_size = resp.headers.get('Content-Length')
total_size = int(total_size) if total_size else None
bar = tqdm(unit='B', total=total_size)
for chunk in resp.iter_content(1024):
if chunk:
bar.update(len(chunk))
temp_fp.write(chunk)
bar.close()
def retrieve_dataset(filename, url):
if not os.path.exists(DATA_HOME):
os.makedirs(DATA_HOME)
full_path = os.path.join(DATA_HOME, filename)
if not os.path.exists(full_path):
with tempfile.NamedTemporaryFile() as tmp_file:
fetch_file(url, tmp_file)
tmp_file.flush()
tmp_file.seek(0)
with open(full_path, 'wb') as final_file:
shutil.copyfileobj(tmp_file, final_file)
return full_path
imdb_path = retrieve_dataset('aclImdb_v1.tar.gz',
'https://mindspore-website.obs.myhuaweicloud.com/notebook/datasets/aclImdb_v1.tar.gz')
1.1 Loading the IMDB Dataset
The downloaded archive (tar.gz) is read with Python’s tarfile. The raw dataset directory structure separates train and test folders, each containing pos and neg subdirectories. A custom IMDBLoader class tokenizes the reviews by splitting, removing punctuation, and converting to lowercase.
import re
import six
import string
import tarfile
class IMDBLoader:
label_mapping = {"pos": 1, "neg": 0}
def __init__(self, archive_path, subset='train'):
self.subset = subset
self.archive_path = archive_path
self.texts, self.labels = [], []
self._load_reviews('pos')
self._load_reviews('neg')
def _load_reviews(self, sentiment):
pattern = re.compile(r"aclImdb/{}/{}/.*\.txt$".format(self.subset, sentiment))
with tarfile.open(self.archive_path) as archive:
member = archive.next()
while member:
if pattern.match(member.name):
raw = archive.extractfile(member).read()
clean = raw.rstrip(six.b("\n\r"))
clean = clean.translate(None, six.b(string.punctuation)).lower()
self.texts.append(clean.split())
self.labels.append([self.label_mapping[sentiment]])
member = archive.next()
def __getitem__(self, idx):
return self.texts[idx], self.labels[idx]
def __len__(self):
return len(self.texts)
Encoding the dataset into MindSpore GeneratorDataset objects:
import mindspore.dataset as ds
def build_imdb_datasets(imdb_archive):
train_set = ds.GeneratorDataset(IMDBLoader(imdb_archive, 'train'),
column_names=['text', 'label'],
shuffle=True, num_samples=10000)
test_set = ds.GeneratorDataset(IMDBLoader(imdb_archive, 'test'),
column_names=['text', 'label'],
shuffle=False)
return train_set, test_set
imdb_train, imdb_test = build_imdb_datasets(imdb_path)
1.2 Pre-trained Word Vectors (GloVe)
The 100-dimensional GloVe embeddings are13 downloaded and processed to create a vocabulary and an embedding matrix. Two special tokens <unk> and <pad> are appended with random and zero vectors respectively.
import zipfile
import numpy as np
def prepare_glove(glove_archive):
glove_txt = os.path.join(DATA_HOME, 'glove.6B.100d.txt')
if not os.path.exists(glove_txt):
with zipfile.ZipFile(glove_archive) as zf:
zf.extractall(DATA_HOME)
token_list = []
weight_list = []
with open(glove_txt, encoding='utf-8') as f:
for line in f:
word, vec = line.split(maxsplit=1)
token_list.append(word)
weight_list.append(np.fromstring(vec, dtype=np.float32, sep=' '))
weight_list.append(np.random.rand(100))
weight_list.append(np.zeros(100, dtype=np.float32))
vocab = ds.text.Vocab.from_list(token_list, special_tokens=['<unk>', '<pad>'], special_first=False)
weights = np.array(weight_list).astype(np.float32)
return vocab, weights
glove_path = retrieve_dataset('glove.6B.zip',
'https://mindspore-website.obs.myhuaweicloud.com/notebook/datasets/glove.6B.zip')
vocab, embeddings = prepare_glove(glove_path)
2. Preprocesssing Pipeline
Each tokenized review is converted into a sequence of indices using Lookup, and the sequences are padded (or truncated) to a fixed length of 500 via PadEnd. Labels are cast to float32.
import mindspore as ms
lookup_op = ds.text.Lookup(vocab, unknown_token='<unk>')
pad_op = ds.transforms.PadEnd([500], pad_value=vocab.tokens_to_ids('<pad>'))
type_cast_op = ds.transforms.TypeCast(ms.float32)
imdb_train = imdb_train.map(operations=[lookup_op, pad_op], input_columns=['text'])
imdb_train = imdb_train.map(operations=[type_cast_op], input_columns=['label'])
imdb_test = imdb_test.map(operations=[lookup_op, pad_op], input_columns=['text'])
imdb_test = imdb_test.map(operations=[type_cast_op], input_columns=['label'])
A validation split (70% train, 30% validation) is12 created, and both sets are batched.
imdb_train, imdb_valid = imdb_train.split([0.7, 0.3])
imdb_train = imdb_train.batch(64, drop_remainder=True)
imdb_valid = imdb_valid.batch(64, drop_remainder=True)
3. Model Architecture
The model consists of an Embedding layer (initialized with GloVe weights), a bidirectional LSTM, and a final Dense layer for binary classification.
import math
import mindspore.nn as nn
import mindspore.ops as ops
from mindspore.common.initializer import Uniform, HeUniform
class SentimentRNN(nn.Cell):
def __init__(self, embed_weights, hidden_dim, output_dim, num_layers,
bidirectional, padding_idx):
super().__init__()
vocab_size, embedding_dim = embed_weights.shape
self.embeddings = nn.Embedding(vocab_size, embedding_dim,
embedding_table=ms.Tensor(embed_weights),
padding_idx=padding_idx)
self.rnn = nn.LSTM(embedding_dim, hidden_dim, num_layers=num_layers,
bidirectional=bidirectional, batch_first=True)
init_range = math.sqrt(5)
self.classifier = nn.Dense(hidden_dim * 2, output_dim,
weight_init=HeUniform(init_range),
bias_init=Uniform(1.0 / math.sqrt(hidden_dim * 2)))
def construct(self, inputs):
embedded = self.embeddings(inputs)
_, (h_n, _) = self.rnn(embedded)
hidden = ops.concat((h_n[-2, :, :], h_n[-1, :, :]), axis=1)
out = self.classifier(hidden)
return out
Encoding the last states of the forward and backward directions provides the representation for classification.
3.1 Training Setup
The loss function is BCEWithLogitsLoss, and the optimizer is Adam.
hidden_size = 256
output_size = 1
num_layers = 2
bidirectional = True
lr = 0.001
pad_idx = vocab.tokens_to_ids('<pad>')
model = SentimentRNN(embeddings, hidden_size, output_size, num_layers, bidirectional, pad_idx)
loss_fn = nn.BCEWithLogitsLoss(reduction='mean')
optimizer = nn.Adam(model.trainable_params(), learning_rate=lr)
3.2 Training and Evaluation Loops
A function run_one_epoch performs a full pass over the training data, while evaluate_model computes loss and accuracy (using a simple rounding of the sigmoid output).
def loss_and_grads(data, label):
logits = model(data)
loss = loss_fn(logits, label)
return loss, ms.ops.grad(loss_fn)(logits, label)
def train_step(data, label):
loss, grads = loss_and_grads(data, label)
optimizer(grads)
return loss
def run_one_epoch(model, train_dataset, epoch=0):
model.set_train(True)
total_batches = train_dataset.get_dataset_size()
running_loss = 0.0
steps = 0
progress = tqdm(total=total_batches)
progress.set_description(f'Epoch {epoch}')
for batch in train_dataset.create_tuple_iterator():
loss_val = train_step(*batch)
running_loss += loss_val.asnumpy()
steps += 1
progress.set_postfix(loss=running_loss / steps)
progress.update(1)
progress.close()
def binary_accuracy(preds, y):
rounded = np.around(ops.sigmoid(preds).asnumpy())
correct = (rounded == y).astype(np.float32)
return correct.sum() / len(correct)
def evaluate_model(model, dataset, criterion, epoch=0):
model.set_train(False)
total_batches = dataset.get_dataset_size()
epoch_loss = 0.0
epoch_acc = 0.0
steps = 0
progress = tqdm(total=total_batches)
progress.set_description(f'Evaluation Epoch {epoch}')
for batch in dataset.create_tuple_iterator():
preds = model(batch[0])
loss = criterion(preds, batch[1])
epoch_loss += loss.asnumpy()
epoch_acc += binary_accuracy(preds, batch[1])
steps += 1
progress.set_postfix(loss=epoch_loss / steps, acc=epoch_acc / steps)
progress.update(1)
progress.close()
return epoch_loss / total_batches
4. Training and Checkpoint Save
The model is trained for 2 epochs; the checkpoint with the lowest validation loss is retained.
num_epochs = 2
best_valid_loss = float('inf')
checkpoint_path = os.path.join(DATA_HOME, 'sentiment-analysis.ckpt')
for epoch in range(num_epochs):
run_one_epoch(model, imdb_train, epoch)
val_loss = evaluate_model(model, imdb_valid, loss_fn, epoch)
if val_loss < best_valid_loss:
best_valid_loss = val_loss
ms.save_checkpoint(model, checkpoint_path)
5. Model Loading and Test Evaluation
After loading the best checkpoint, performance is measured on test set.
param_dict = ms.load_checkpoint(checkpoint_path)
ms.load_param_into_net(model, param_dict)
imdb_test = imdb_test.batch(64)
evaluate_model(model, imdb_test, loss_fn)
6. Custom Sentence Prediction
A helper function tokenizes an input sentence, converts it to indices, and returns the predicted sentiment.
label_names = {1: 'Positive', 0: 'Negative'}
def predict_sentiment(model, vocab, sentence):
model.set_train(False)
tokens = sentence.lower().split()
indices = vocab.tokens_to_ids(tokens)
tensor = ms.Tensor(indices, ms.int32)
tensor = tensor.expand_dims(0)
output = model(tensor)
pred_label = int(np.round(ops.sigmoid(output).asnumpy()))
return label_names[pred_label]
print(predict_sentiment(model, vocab, "This film is terrible")) # Negative
print(predict_sentiment(model, vocab, "This film is great")) # Positive