Files
AI-Red-Teaming-CSCD94/privacy/dp-sgd.ipynb
T
2026-07-26 23:12:07 -04:00

15 KiB

In [1]:
import os
import json
import torch
import torch.optim as optim
from safetensors.torch import save_file

from htb_ai_library import (
    set_reproducibility, use_htb_style,
    CIFAR10CNN,
    get_cifar10_loaders,
    train_baseline_sgd,
    train_dp_sgd,
    evaluate_accuracy,
    compute_mia_advantage,
    plot_accuracy_comparison,
    plot_privacy_utility_tradeoff,
)
RANDOM_SEED = 1337
BATCH_SIZE = 256
BASELINE_EPOCHS = 20
BASELINE_LR = 0.1
DP_EPOCHS = 20
DP_LR = 0.1
MAX_GRAD_NORM = 1.0
DELTA = 1e-5

set_reproducibility(RANDOM_SEED)
use_htb_style()
device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')

os.makedirs("figs", exist_ok=True)
os.makedirs("output", exist_ok=True)
os.makedirs("models", exist_ok=True)
In [2]:
print("=" * 80)
print("  DP-SGD PRIVACY MITIGATION DEMONSTRATION")
print("=" * 80)
print(f"\nDevice: {device}")
print(f"Random seed: {RANDOM_SEED}")

print("\nLoading CIFAR-10 dataset...")
train_dataset, test_dataset, train_loader, test_loader = get_cifar10_loaders(batch_size=BATCH_SIZE, download=True)

print(f"Training samples: {len(train_dataset):,}")
print(f"Test samples: {len(test_dataset):,}")
print(f"Batch size: {BATCH_SIZE}")

print("\n" + "=" * 80)
print("  TRAINING: BASELINE MODEL (No Privacy Protection)")
print("=" * 80)

baseline_model = CIFAR10CNN().to(device)
baseline_model = train_baseline_sgd(baseline_model, train_loader, device, epochs=BASELINE_EPOCHS, learning_rate=BASELINE_LR)
================================================================================
  DP-SGD PRIVACY MITIGATION DEMONSTRATION
================================================================================

Device: cpu
Random seed: 1337

Loading CIFAR-10 dataset...
Files already downloaded and verified
Files already downloaded and verified
Training samples: 50,000
Test samples: 10,000
Batch size: 256

================================================================================
  TRAINING: BASELINE MODEL (No Privacy Protection)
================================================================================
Epoch [1/20] Loss: nan | Acc: 37.60%
Epoch [2/20] Loss: nan | Acc: 10.00%
---------------------------------------------------------------------------
KeyboardInterrupt                         Traceback (most recent call last)
Cell In[2], line 19
     16 print("=" * 80)
     18 baseline_model = CIFAR10CNN().to(device)
---> 19 baseline_model = train_baseline_sgd(baseline_model, train_loader, device, epochs=BASELINE_EPOCHS, learning_rate=BASELINE_LR)

File ~/.conda/envs/ai/lib/python3.11/site-packages/htb_ai_library/training/loops.py:244, in train_baseline_sgd(model, train_loader, device, epochs, learning_rate, momentum)
    242 output = model(data)
    243 loss = criterion(output, target)
--> 244 loss.backward()
    245 optimizer.step()
    246 running_loss += loss.item()

File ~/.conda/envs/ai/lib/python3.11/site-packages/torch/_tensor.py:581, in Tensor.backward(self, gradient, retain_graph, create_graph, inputs)
    571 if has_torch_function_unary(self):
    572     return handle_torch_function(
    573         Tensor.backward,
    574         (self,),
   (...)    579         inputs=inputs,
    580     )
--> 581 torch.autograd.backward(
    582     self, gradient, retain_graph, create_graph, inputs=inputs
    583 )

File ~/.conda/envs/ai/lib/python3.11/site-packages/torch/autograd/__init__.py:347, in backward(tensors, grad_tensors, retain_graph, create_graph, grad_variables, inputs)
    342     retain_graph = create_graph
    344 # The reason we repeat the same comment below is that
    345 # some Python versions print out the first line of a multi-line function
    346 # calls in the traceback and some print out the last line
--> 347 _engine_run_backward(
    348     tensors,
    349     grad_tensors_,
    350     retain_graph,
    351     create_graph,
    352     inputs,
    353     allow_unreachable=True,
    354     accumulate_grad=True,
    355 )

File ~/.conda/envs/ai/lib/python3.11/site-packages/torch/autograd/graph.py:825, in _engine_run_backward(t_outputs, *args, **kwargs)
    823     unregister_hooks = _register_logging_hooks_on_whole_graph(t_outputs)
    824 try:
--> 825     return Variable._execution_engine.run_backward(  # Calls into the C++ engine to run the backward pass
    826         t_outputs, *args, **kwargs
    827     )  # Calls into the C++ engine to run the backward pass
    828 finally:
    829     if attach_logging_hooks:

KeyboardInterrupt: 
In [3]:
train_acc_baseline = evaluate_accuracy(baseline_model, train_loader, device)
test_acc_baseline = evaluate_accuracy(baseline_model, test_loader, device)

print("\nBaseline Model Performance:")
print(f"  Training accuracy: {train_acc_baseline:.2f}%")
print(f"  Test accuracy: {test_acc_baseline:.2f}%")
print(f"  Overfitting gap: {train_acc_baseline - test_acc_baseline:.2f}%")

torch.save(baseline_model.state_dict(), "output/baseline_model.pth")
print("\nSaved baseline model to output/baseline_model.pth")

print("\n" + "=" * 80)
print("  MEMBERSHIP INFERENCE MEASUREMENT: Baseline Model")
print("=" * 80)

mia_acc_baseline, mia_adv_baseline = compute_mia_advantage(
    baseline_model, train_loader, test_loader, device
)

print("\nMIA Results (Baseline):")
print(f"  Attack accuracy: {mia_acc_baseline:.4f}")
print(f"  Attack advantage: {mia_adv_baseline:.4f}")
print("  Random baseline: 0.5000")
Baseline Model Performance:
  Training accuracy: 10.00%
  Test accuracy: 10.00%
  Overfitting gap: 0.00%

Saved baseline model to output/baseline_model.pth

================================================================================
  MEMBERSHIP INFERENCE MEASUREMENT: Baseline Model
================================================================================

MIA Results (Baseline):
  Attack accuracy: 0.5000
  Attack advantage: 0.0000
  Random baseline: 0.5000
In [4]:
from opacus import PrivacyEngine
from opacus.validators import ModuleValidator

print("\n" + "=" * 80)
print("  TRAINING: DP-SGD MODEL (Target ε=10)")
print("=" * 80)

TARGET_EPSILON_10 = 10.0

_, _, train_loader_dp, test_loader_dp = get_cifar10_loaders(batch_size=BATCH_SIZE, download=False)

dp_model_10 = CIFAR10CNN().to(device)
dp_model_10 = ModuleValidator.fix(dp_model_10)
optimizer_dp = optim.SGD(dp_model_10.parameters(), lr=DP_LR, momentum=0.9)

privacy_engine = PrivacyEngine(accountant="rdp")
dp_model_10, optimizer_dp, train_loader_dp = privacy_engine.make_private_with_epsilon(
    module=dp_model_10,
    optimizer=optimizer_dp,
    data_loader=train_loader_dp,
    target_epsilon=TARGET_EPSILON_10,
    target_delta=DELTA,
    epochs=DP_EPOCHS,
    max_grad_norm=MAX_GRAD_NORM,
)

print(f"\nConfiguration:")
print(f"  Target epsilon: {TARGET_EPSILON_10}")
print(f"  Delta: {DELTA}")
print(f"  Max gradient norm: {MAX_GRAD_NORM}")
================================================================================
  TRAINING: DP-SGD MODEL (Target ε=10)
================================================================================
/home/jeremy/.conda/envs/ai/lib/python3.11/site-packages/opacus/privacy_engine.py:96: UserWarning: Secure RNG turned off. This is perfectly fine for experimentation as it allows for much faster training performance, but remember to turn it on and retrain one last time before production with ``secure_mode`` turned on.
  warnings.warn(
/home/jeremy/.conda/envs/ai/lib/python3.11/site-packages/opacus/accountants/analysis/rdp.py:332: UserWarning: Optimal order is the largest alpha. Please consider expanding the range of alphas to get a tighter privacy bound.
  warnings.warn(
Configuration:
  Target epsilon: 10.0
  Delta: 1e-05
  Max gradient norm: 1.0
In [ ]: