This document outlines a baseline approach for the Inclusion Global Multimedia Deepfake Detection competition. The competition tasks participants with classifying images to detect deepfakes, using the Area Under the ROC Curve (AUC) as the primary evaluation metric. If AUC scores are tied, True Positive Rate at a False Positive Rate of 1E-3 (TPR@FPR=1E-3) will be used as a tie-breaker.
Model Definition
The baseline utilizes a pre-trained ResNet18 model loaded via the timm library, configured for binary classification.
import timm
import torch
# Initialize a pre-trained ResNet18 model for 2 classes (deepfake/real)
model = timm.create_model('resnet18', pretrained=True, num_classes=2)
# Move the model to the GPU if available
model = model.cuda()
Data Handling and Augmentation
A custom PyTorch Dataset, FFDIDataset, is implemented to load image data and labels. Data augmentation, including resizing, random horizontal and vertical flips, and normalization, is applied to the training set using torchvision.transforms.
from torch.utils.data import Dataset
from PIL import Image
import numpy as np
import torch
import torchvision.transforms as transforms
class FFDIDataset(Dataset):
def __init__(self, image_paths, image_labels, transform=None):
self.image_paths = image_paths
self.image_labels = image_labels
self.transform = transform if transform else transforms.Compose([]) # Use empty transform if none provided
def __getitem__(self, index):
# Open image and convert to RGB
img = Image.open(self.image_paths[index]).convert('RGB')
# Apply transformations if they exist
if self.transform:
img = self.transform(img)
# Return image and its corresponding label as a PyTorch tensor
return img, torch.tensor(self.image_labels[index], dtype=torch.long)
def __len__(self):
return len(self.image_paths)
# Define data transformations
train_transforms = transforms.Compose([
transforms.Resize((256, 256)),
transforms.RandomHorizontalFlip(),
transforms.RandomVerticalFlip(),
transforms.ToTensor(),
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])
])
validation_transforms = transforms.Compose([
transforms.Resize((256, 256)),
transforms.ToTensor(),
transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225])
])
# Assuming train_label and val_label are pandas DataFrames with 'path' and 'target' columns
# Create DataLoaders for training and validation sets
batch_size = 40
num_workers = 4
pin_memory = True
train_dataset = FFDIDataset(train_label['path'].head(1000).tolist(), train_label['target'].head(1000).tolist(), transform=train_transforms)
train_loader = torch.utils.data.DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers, pin_memory=pin_memory)
val_dataset = FFDIDataset(val_label['path'].head(1000).tolist(), val_label['target'].head(1000).tolist(), transform=validation_transforms)
val_loader = torch.utils.data.DataLoader(val_dataset, batch_size=batch_size, shuffle=False, num_workers=num_workers, pin_memory=pin_memory)
Training and Validation Loop
The training process involves iterating through the training data, computing the loss using a specified criterion (e.g., Cross-Entropy Loss), performing backpropagation, and updating model weights. The validation function evaluates the model's performance on the vlaidation set, calculating accuracy.
Helper classes AverageMeter and ProgressMeter are used to track metrics like time, loss, and accuracy during training and validation.
import time
from tqdm.notebook import tqdm_notebook # Assuming this is for notebook environments
# Utility classes for tracking metrics
class AverageMeter(object):
def __init__(self, name, fmt=':f'):
self.name = name
self.fmt = fmt
self.reset()
def reset(self):
self.val = 0
self.avg = 0
self.sum = 0
self.count = 0
def update(self, val, n=1):
self.val = val
self.sum += val * n
self.count += n
self.avg = self.sum / self.count
def __str__(self):
fmtstr = '{name} {val' + self.fmt + '} ({avg' + self.fmt + '})'
return fmtstr.format(**self.__dict__)
class ProgressMeter(object):
def __init__(self, num_batches, *meters, prefix=""):
self.batch_fmtstr = self._get_batch_fmtstr(num_batches)
self.meters = meters
self.prefix = prefix
def display(self, batch):
entries = [self.prefix + self.batch_fmtstr.format(batch)]
entries += [str(meter) for meter in self.meters]
print('\t'.join(entries))
def _get_batch_fmtstr(self, num_batches):
num_digits = len(str(num_batches))
fmt = '{:' + str(num_digits) + 'd}'
return '[' + fmt + '/' + fmt.format(num_batches) + ']'
# Define loss function, optimizer, and learning rate scheduler
criterion = torch.nn.CrossEntropyLoss().cuda()
optimizer = torch.optim.SGD(model.parameters(), lr=0.01, momentum=0.9, weight_decay=1e-4)
scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.9) # Example scheduler
def train_one_epoch(train_loader, model, criterion, optimizer, epoch):
batch_time_meter = AverageMeter('Time', ':6.3f')
loss_meter = AverageMeter('Loss', ':.4e')
accuracy_meter = AverageMeter('Acc@1', ':6.2f')
progress_meter = ProgressMeter(len(train_loader), batch_time_meter, loss_meter, accuracy_meter, prefix=f"Epoch: {epoch}")
model.train() # Set model to training mode
end_time = time.time()
for i, (inputs, targets) in enumerate(train_loader):
inputs = inputs.cuda(non_blocking=True)
targets = targets.cuda(non_blocking=True)
# Forward pass
outputs = model(inputs)
loss = criterion(outputs, targets)
# Compute and update metrics
loss_meter.update(loss.item(), inputs.size(0))
preds = outputs.argmax(dim=1)
acc = (preds == targets).float().mean() * 100.0
accuracy_meter.update(acc, inputs.size(0))
# Backward pass and optimize
optimizer.zero_grad()
loss.backward()
optimizer.step()
# Measure elapsed time
batch_time_meter.update(time.time() - end_time)
end_time = time.time()
if i % 50 == 0: # Log progress every 50 batches
progress_meter.display(i)
def evaluate(val_loader, model, criterion):
batch_time_meter = AverageMeter('Time', ':6.3f')
loss_meter = AverageMeter('Loss', ':.4e')
accuracy_meter = AverageMeter('Acc@1', ':6.2f')
progress_meter = ProgressMeter(len(val_loader), batch_time_meter, loss_meter, accuracy_meter, prefix='Validation:')
model.eval() # Set model to evaluation mode
total_correct = 0
total_samples = 0
with torch.no_grad(): # Disable gradient calculation for evaluation
end_time = time.time()
for i, (inputs, targets) in enumerate(tqdm_notebook(val_loader, desc="Evaluating")):
inputs = inputs.cuda(non_blocking=True)
targets = targets.cuda(non_blocking=True)
# Forward pass
outputs = model(inputs)
loss = criterion(outputs, targets)
# Compute and update metrics
loss_meter.update(loss.item(), inputs.size(0))
preds = outputs.argmax(dim=1)
acc = (preds == targets).float().mean() * 100.0
accuracy_meter.update(acc, inputs.size(0))
# Measure elapsed time
batch_time_meter.update(time.time() - end_time)
end_time = time.time()
if i % 50 == 0: # Log progress every 50 batches
progress_meter.display(i)
print(f' * Validation Accuracy: {accuracy_meter.avg:.3f}')
return accuracy_meter # Return the meter object for potential use
# Training loop
best_accuracy = 0.0
num_epochs = 2
for epoch in range(num_epochs):
scheduler.step() # Adjust learning rate
print(f"--- Epoch {epoch+1}/{num_epochs} ---")
# Train for one epoch
train_one_epoch(train_loader, model, criterion, optimizer, epoch)
# Evaluate on validation set
val_accuracy = evaluate(val_loader, model, criterion)
# Save best model
if val_accuracy.avg > best_accuracy:
best_accuracy = val_accuracy.avg
torch.save(model.state_dict(), f'./best_model_acc_{best_accuracy:.2f}.pt')
print(f"Saved best model with accuracy: {best_accuracy:.2f}%")
Prediction and Submission
After training, the model is used to generate predictions on the test set. These prediction are then formatted into a CSV file suitable for submission to the competition platform.
# Assuming test_loader is defined similarly to val_loader
# and predict function is available
# Placeholder for prediction logic
def predict(test_loader, model):
model.eval()
predictions = []
with torch.no_grad():
for inputs, _ in tqdm_notebook(test_loader, desc="Predicting"):
inputs = inputs.cuda()
outputs = model(inputs)
# Assuming binary classification, take probability of the positive class
probabilities = torch.softmax(outputs, dim=1)[:, 1]
predictions.extend(probabilities.cpu().numpy())
return np.array(predictions)
# Assuming val_label is a DataFrame and test_loader is prepared
# test_predictions = predict(test_loader, model)
# val_label['prediction_score'] = test_predictions # Or appropriate column name
# Create submission file
# submission_df = val_label[['img_name', 'prediction_score']] # Adjust column names as needed
# submission_df.to_csv('submission.csv', index=False)
# Example submission generation for the validation set as a placeholder
# In a real scenario, you would use a separate test_loader
validation_predictions = predict(val_loader, model) # Use val_loader for demonstration
val_label['y_pred'] = validation_predictions
# Assuming val_label has an 'img_name' column
val_label[['img_name', 'y_pred']].to_csv('submit.csv', index=None)
print("Submission file 'submit.csv' created.")
Autogradient Demonstration
PyTorch's automatic differentiation engine, Autograd, tracks operations involving tensors that have requires_grad=True. This allows for efficient gradient computation, crucial for training neural networks.
import torch
# Initialize a tensor with requires_grad=True
x = torch.tensor([[1.0, 2.0], [3.0, 4.0]], requires_grad=True)
print("Input tensor x:\n", x)
# Perform some operations
y = x + 2
print("\nTensor y (x + 2):\n", y)
print("y's grad_fn:", y.grad_fn) # Shows that y was created by an Add operation
z = y * y * 3
print("\nTensor z (y*y*3):\n", z)
print("z's grad_fn:", z.grad_fn) # Shows that z was created by a Multiply operation
out = z.mean()
print("\nScalar output (mean of z):\n", out)
print("out's grad_fn:", out.grad_fn) # Shows that out was created by a Mean operation
# Compute gradients of 'out' with respect to 'x'
# The computation graph is traversed backwards from 'out'
out.backward()
# Gradients of 'out' with respect to 'x'
print("\nGradients of out with respect to x (x.grad):\n", x.grad)
# Manual gradient calculation for verification:
# out = mean(3 * (x + 2)^2)
# d(out)/dx = d(mean(3 * (x_ij + 2)^2))/dx_kl
# For scalar mean, it's 1/N * sum(d(3 * (x_ij + 2)^2)/dx_kl)
# d(3 * (x_ij + 2)^2)/dx_ij = 3 * 2 * (x_ij + 2) * 1 = 6 * (x_ij + 2)
# d(3 * (x_ij + 2)^2)/dx_kl = 0 if k != i or l != j
# So, d(out)/dx_ij = (1/4) * 6 * (x_ij + 2) = 1.5 * (x_ij + 2)
# For x = [[1, 2], [3, 4]]:
# dx[0,0] = 1.5 * (1 + 2) = 4.5
# dx[0,1] = 1.5 * (2 + 2) = 6.0
# dx[1,0] = 1.5 * (3 + 2) = 7.5
# dx[1,1] = 1.5 * (4 + 2) = 9.0
# This matches the output of x.grad.
Function Fitting Example
This section demonstrates setting up matplotlib for plotting within an IPython environment.
import matplotlib.pyplot as plt
from IPython.display import set_matplotlib_formats
# Set matplotlib backend to SVG for better rendering in notebooks
set_matplotlib_formats('svg')
# Enable inline plotting for displaying plots directly in the notebook
# %matplotlib inline # This line is typically uncommented in a Jupyter environment