One handwritten digit in, ten classes out. The network produces one raw score — a logit — per digit.
Every logit sits in the shared denominator, so every output depends on every input.
Backprop never needs ∂P̂/∂Ŷ itself — only the product ∂L/∂P̂ · ∂P̂/∂Ŷ. And every entry of that product reuses one sum.
The only difference is how many jobs each output has.
The negative log-likelihood that the model assigns every input the correct label in the dataset.
p is the true probability of each class. For MNIST it’s one-hot (1 on the correct digit, 0 elsewhere)
so only one term survives: −log p̂true.
class MLP(nn.Module): # __init__: fc1 = Linear(784, 128), fc2 = Linear(128, 10) def forward(self, x): h = torch.tanh(self.fc1(x)) p_hat = torch.softmax(self.fc2(h), dim=1) return p_hat.clamp(1e-7, 1.0) # avoid log(0) p_hat = model(x) loss = nn.NLLLoss()(torch.log(p_hat), labels)
class MLP(nn.Module): # __init__: fc1 = Linear(784, 128), fc2 = Linear(128, 10) def forward(self, x): h = torch.tanh(self.fc1(x)) return self.fc2(h) # raw logits def predict(self, x): return torch.softmax(self.forward(x), dim=1) # p̂ loss = nn.CrossEntropyLoss()(model(x), labels)
import torch import torch.nn as nn import torch.optim as optim import torch.optim.lr_scheduler as lr_scheduler # 1. Define your model, loss, and optimizer model = nn.Linear(10, 2) criterion = nn.MSELoss() optimizer = optim.SGD(model.parameters(), lr=0.01, momentum=0.9) optimizer = optim.Adagrad(model.parameters(), lr=0.01) optimizer = optim.RMSprop(model.parameters(), lr=0.01, alpha=0.99, momentum=0.9) optimizer = torch.optim.AdamW( model.parameters(), lr=1e-3, weight_decay=0.01 ) optimizer = optim.LBFGS(model.parameters(), lr=1.0)
import torch import torch.nn as nn import torch.optim as optim import torch.optim.lr_scheduler as lr_scheduler # 1. Define your model, loss, and optimizer model = nn.Linear(10, 2) criterion = nn.MSELoss() optimizer = optim.SGD(model.parameters(), lr=0.1) # 2. Define the scheduler (e.g., cut LR in half every 10 epochs) # https://docs.pytorch.org/docs/stable/optim.html scheduler = lr_scheduler.StepLR(optimizer, step_size=10, gamma=0.5) # 3. Training Loop for epoch in range(100): for inputs, targets in dataloader: optimizer.zero_grad() outputs = model(inputs) loss = criterion(outputs, targets) loss.backward() # Update model weights optimizer.step() # Update the learning rate AFTER the optimizer step at the end of the epoch scheduler.step()
Dropout destabilizes variance
Helps with generalization when models are big and data is relatively small
import torch import torch.nn as nn class ExampleNetwork(nn.Module): def __init__(self): super(ExampleNetwork, self): # 1. Define your layers self.fc1 = nn.Linear(784, 256) self.fc2 = nn.Linear(256, 10) # 2. Define the dropout layer (p = probability of zeroing an element) # Standard default is 0.5 if not specified self.dropout = nn.Dropout(p=0.3) self.relu = nn.ReLU() def forward(self, x): # 3. Pass the activation output through the dropout layer x = self.fc1(x) x = self.relu(x) x = self.dropout(x) # Applied after the activation function x = self.fc2(x) return x
We know much about this model.
Other things we don’t know.
import torch.nn as nn import torch class MLP_Multiclass(nn.Module): def __init__(self, features, hidden_size): super().__init__() self.fc1 = nn.Linear(features, hidden_size) self.fc2 = nn.Linear(hidden_size, 10) def forward(self, x): h = self.fc1(x) h = torch.tanh(h) return self.fc2(h) def predict(self, x): return torch.softmax(self.forward(x), dim=1).argmax(dim=1)