Lecture 12: Neural Networks (1) – CS 189, Fall 2026
Logistic regression fits a linear boundary on whatever features $\phi(x)$ we give it. Today the model learns the features. We will:
- Watch logistic regression fail on XOR, then solve XOR with a two-layer network whose weights we set by hand.
- Train a deep network with no activation function and check that it is still a linear model.
- Add a non-linearity and watch the boundary bend.
- Compare sigmoid, tanh, ReLU, and GELU by where their derivatives vanish, and watch a ReLU unit die.
- Match the output layer and loss to the target, compute softmax cross-entropy from logits, and see that the loss is no longer convex.
Reference: Bishop & Bishop, Deep Learning: Foundations and Concepts, Ch. 6, §6.1, §6.2.1-6.2.3, §6.3.1, §6.4.
Imports¶
import numpy as np
import matplotlib.pyplot as plt
from matplotlib.colors import LinearSegmentedColormap
from sklearn.datasets import make_moons
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import log_loss
import torch
import torch.nn as nn
import torch.nn.functional as F
import warnings
warnings.filterwarnings("ignore")
import ipywidgets as widgets
%config InlineBackend.figure_format = "retina"
C_BLUE, C_GOLD, C_RED, C_GREEN = "#003262", "#FDB515", "#D55E00", "#00553A"
CLASS_COLORS = [C_BLUE, C_GOLD, C_RED]
plt.rcParams.update({"figure.figsize": (8, 5), "axes.grid": True, "grid.alpha": 0.3,
"font.size": 13, "axes.titlesize": 15, "axes.labelsize": 14})
Helper Code¶
Run these two cells before the rest. You do not need to read them today: Step 8 opens up MLP and train, and Lecture 13 explains loss.backward().
def make_xor(n=400, noise=0.15, seed=189):
"""n points split evenly over the corners of the unit square, plus Gaussian noise.
The label is x1 XOR x2 of the corner."""
rng = np.random.default_rng(seed)
corners = np.repeat(np.array([[0, 0], [0, 1], [1, 0], [1, 1]]), n // 4, axis=0)
X = corners + noise * rng.standard_normal(corners.shape)
return X, corners[:, 0] ^ corners[:, 1]
def make_spiral(n_per_class=150, K=3, noise=0.2, seed=189):
"""K interleaved spiral arms, one class per arm."""
rng = np.random.default_rng(seed)
r = np.linspace(0.2, 1, n_per_class)
X, t = [], []
for k in range(K):
theta = np.linspace(0, 4, n_per_class) + 2 * np.pi * k / K + noise * rng.standard_normal(n_per_class)
X.append(np.c_[r * np.cos(theta), r * np.sin(theta)])
t.append(np.full(n_per_class, k))
return np.vstack(X), np.concatenate(t)
PROB_CMAP = LinearSegmentedColormap.from_list("blue_white_gold", ["#B9CCE2", "#FFFFFF", "#FEE3A0"])
def plot_data(X, t, ax=None, title=None):
ax = ax or plt.gca()
for k in np.unique(t):
ax.scatter(*X[t == k].T, s=16, color=CLASS_COLORS[k], edgecolor="k", linewidth=0.3, label=f"class {k}")
ax.set(xlabel="$x_1$", ylabel="$x_2$", title=title, aspect="equal")
return ax
def plot_boundary(predict, X, t, ax=None, title=None, pad=0.5, n=300):
"""Shade the plane by the model's prediction, then overlay the data.
predict maps an (m, 2) array to P(class 1) with shape (m,), or to class probabilities (m, K)."""
ax = ax or plt.gca()
lo, hi = X.min(0) - pad, X.max(0) + pad
g1, g2 = np.meshgrid(np.linspace(lo[0], hi[0], n), np.linspace(lo[1], hi[1], n))
P = predict(np.c_[g1.ravel(), g2.ravel()])
if P.ndim == 1:
ax.contourf(g1, g2, P.reshape(g1.shape), levels=np.linspace(0, 1, 21), cmap=PROB_CMAP)
ax.contour(g1, g2, P.reshape(g1.shape), levels=[0.5], colors="k", linewidths=1.5)
else:
K = P.shape[1]
ax.contourf(g1, g2, P.argmax(1).reshape(g1.shape), levels=np.arange(K + 1) - 0.5,
colors=[c + "40" for c in CLASS_COLORS[:K]])
plot_data(X, t, ax, title)
ax.grid(False)
return ax
ACTIVATIONS = {"identity": lambda a: a, "relu": torch.relu, "tanh": torch.tanh,
"sigmoid": torch.sigmoid, "gelu": F.gelu}
class MLP(nn.Module):
"""Fully connected network. dims = [inputs, hidden_1, ..., hidden_L, outputs].
Every hidden layer applies the activation. The output layer returns logits (no activation)."""
def __init__(self, dims, activation="relu"):
super().__init__()
self.layers = nn.ModuleList(nn.Linear(d_in, d_out) for d_in, d_out in zip(dims[:-1], dims[1:]))
self.act = ACTIVATIONS[activation]
def forward(self, x):
for layer in self.layers[:-1]:
x = self.act(layer(x))
return self.layers[-1](x)
def train(model, X, t, epochs=500, lr=0.01, loss_fn=None):
"""Full-batch Adam (Lecture 11). Unless a loss is given, it follows the output layer:
1 output -> binary cross-entropy on logits; K outputs -> softmax cross-entropy on logits."""
X = torch.as_tensor(X, dtype=torch.float32)
K = model.layers[-1].out_features
if loss_fn is None:
loss_fn = nn.BCEWithLogitsLoss() if K == 1 else nn.CrossEntropyLoss()
if isinstance(loss_fn, nn.CrossEntropyLoss):
T = torch.as_tensor(t, dtype=torch.long)
else:
T = torch.as_tensor(t, dtype=torch.float32).reshape(-1, K)
optimizer = torch.optim.Adam(model.parameters(), lr=lr)
history = []
for epoch in range(epochs):
optimizer.zero_grad()
loss = loss_fn(model(X), T)
loss.backward() # gradients: Lecture 13
optimizer.step()
history.append(loss.item())
return history
def predict_proba(model):
"""Wrap a trained classifier as a NumPy function: P(class 1), or class probabilities."""
def f(X):
with torch.no_grad():
z = model(torch.as_tensor(X, dtype=torch.float32))
return torch.sigmoid(z)[:, 0].numpy() if z.shape[1] == 1 else torch.softmax(z, dim=1).numpy()
return f
def accuracy(model, X, t):
P = predict_proba(model)(X)
pred = (P > 0.5).astype(int) if P.ndim == 1 else P.argmax(1)
return (pred == t).mean()
The Data: XOR¶
Four clusters at the corners of the unit square. The label is $x_1 \text{ XOR } x_2$ of the corner: class 1 when exactly one coordinate is 1.
X, t = make_xor()
plot_data(X, t, title="XOR: 400 points, 100 per corner")
plt.legend(loc="center")
plt.show()
Step 1: Logistic Regression on XOR (Demo 1)¶
Logistic regression from Lectures 8-9, with the raw coordinates as features, $\phi(x) = (x_1, x_2)$, and no regularization.
logreg = LogisticRegression(C=np.inf).fit(X, t)
p_lr = logreg.predict_proba(X)[:, 1]
print(f"weights {logreg.coef_.ravel().round(3)}, bias {logreg.intercept_.round(3)}")
print(f"accuracy {logreg.score(X, t):.3f}, cross-entropy {log_loss(t, p_lr):.4f} (ln 2 = {np.log(2):.4f})")
print(f"predicted P(class 1) ranges from {p_lr.min():.3f} to {p_lr.max():.3f}")
weights [0.044 0.083], bias [-0.063] accuracy 0.497, cross-entropy 0.6928 (ln 2 = 0.6931) predicted P(class 1) ranges from 0.475 to 0.524
The failure: the best line is no line¶
Any line leaves at least one cluster on the wrong side. Cross-entropy punishes confident mistakes, so the best fit shrinks the weights toward zero and predicts about 0.5 everywhere. Its loss is $\ln 2$, the loss of a coin flip.
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_data(X, t, axes[0], "The data")
plot_boundary(lambda Z: logreg.predict_proba(Z)[:, 1], X, t, axes[1],
f"Logistic regression: accuracy {logreg.score(X, t):.0%}")
plt.tight_layout()
plt.show()
Step 2: One Neuron, Wired by Hand¶
A neuron takes a weighted sum of its inputs plus a bias, then applies an activation: $z = h(w_1 x_1 + w_2 x_2 + w_0)$. With the step activation, $h(a) = 1$ if $a > 0$ and $0$ otherwise, a neuron draws one line and reports which side of it you are on.
def step(a):
return (a > 0).astype(int)
def neuron(X, w, w0, h=step):
return h(X @ np.asarray(w) + w0)
corners = np.array([[0, 0], [0, 1], [1, 0], [1, 1]])
print("x1 x2 | AND OR")
for x, a, o in zip(corners, neuron(corners, [1, 1], -1.5), neuron(corners, [1, 1], -0.5)):
print(f" {x[0]} {x[1]} | {a} {o}")
x1 x2 | AND OR 0 0 | 0 0 0 1 | 0 1 1 0 | 0 1 1 1 | 1 1
AND and OR each need one line. XOR needs class 1 at $(0, 1)$ and $(1, 0)$ and class 0 at $(0, 0)$ and $(1, 1)$. No single line does that, so no single neuron does.
Step 3: Two Layers Solve XOR¶
Write XOR as $(x_1 \text{ AND NOT } x_2) \text{ OR } (\text{NOT } x_1 \text{ AND } x_2)$. Each piece is one neuron:
- hidden unit 1: $z_1 = h(x_1 - x_2 - 0.5)$, on only at $(1, 0)$
- hidden unit 2: $z_2 = h(-x_1 + x_2 - 0.5)$, on only at $(0, 1)$
- output unit: $y = h(z_1 + z_2 - 0.5)$, the OR of the two
Three neurons and nine weights, all set by hand.
W1 = np.array([[1, -1], [-1, 1]]); b1 = np.array([-0.5, -0.5]) # hidden layer: 2 units
w2 = np.array([1, 1]); b2 = -0.5 # output unit: OR
def xor_net(X, h=step):
Z = h(X @ W1.T + b1) # hidden layer
return step(Z @ w2 + b2), Z # output, hidden activations
y, Z = xor_net(corners)
print("x1 x2 | z1 z2 | y")
for x, z, v in zip(corners, Z, y):
print(f" {x[0]} {x[1]} | {z[0]} {z[1]} | {v}")
print(f"\naccuracy on the 400 noisy points: {(xor_net(X)[0] == t).mean():.4f}")
x1 x2 | z1 z2 | y 0 0 | 0 0 | 0 0 1 | 0 1 | 1 1 0 | 1 0 | 1 1 1 | 0 0 | 0 accuracy on the 400 noisy points: 0.9775
plot_boundary(lambda Z: xor_net(Z)[0].astype(float), X, t,
title=f"Hand-wired network: accuracy {(xor_net(X)[0] == t).mean():.1%}")
plt.show()
The hidden layer is a new $\phi(x)$¶
Plot each point by its hidden activations $(z_1, z_2)$ instead of its inputs. Here the hidden step is replaced by a steep sigmoid, $\sigma(5a)$, so the points spread out instead of piling onto three corners. In the new coordinates one line separates the classes: the output unit is logistic regression on features that the hidden layer computed. (The points fall on a single curve because both hidden units read the same quantity, $x_1 - x_2$.)
steep = lambda a: 1 / (1 + np.exp(-5 * a))
_, Z_smooth = xor_net(X, h=steep)
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_data(X, t, axes[0], "Input space $(x_1, x_2)$")
plot_data(Z_smooth, t, axes[1], "Hidden space $(z_1, z_2)$")
axes[1].plot([-0.1, 0.6], [0.6, -0.1], color="k", lw=1.5) # the output unit's line: z1 + z2 = 0.5
axes[1].set(xlabel="$z_1$", ylabel="$z_2$")
plt.tight_layout()
plt.show()
Step 4: Depth Without a Non-Linearity (Demo 1, continued)¶
Poll 1 comes before this cell. Now we train a network instead of wiring one: three hidden layers of 100 units each, with the identity as the activation, $h(a) = a$.
torch.manual_seed(189)
deep_linear = MLP([2, 100, 100, 100, 1], activation="identity")
history = train(deep_linear, X, t, epochs=500, lr=0.01)
n_params = sum(p.numel() for p in deep_linear.parameters())
print(f"deep linear network ({n_params:,} parameters): accuracy {accuracy(deep_linear, X, t):.3f}, loss {history[-1]:.4f}")
print(f"logistic regression (3 parameters): accuracy {logreg.score(X, t):.3f}, loss {log_loss(t, p_lr):.4f}")
deep linear network (20,601 parameters): accuracy 0.497, loss 0.6928 logistic regression (3 parameters): accuracy 0.497, loss 0.6928
The failure: three hidden layers, one straight line¶
The same loss as logistic regression, to four decimals. Multiply the layers out and the whole network is a single linear map $Wx + b$:
with torch.no_grad():
W, b = torch.eye(2), torch.zeros(2)
for layer in deep_linear.layers: # x -> W_l (W x + b) + b_l
W, b = layer.weight @ W, layer.weight @ b + layer.bias
Xt = torch.as_tensor(X, dtype=torch.float32)
gap = (deep_linear(Xt) - (Xt @ W.T + b)).abs().max().item()
print(f"collapsed weights {W.numpy().round(3)}, bias {b.numpy().round(3)}")
print(f"largest difference from the full network's output: {gap:.1e}")
collapsed weights [[0.044 0.083]], bias [-0.063] largest difference from the full network's output: 4.8e-08
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_boundary(predict_proba(deep_linear), X, t, axes[0],
f"3 hidden layers of 100, identity: {accuracy(deep_linear, X, t):.0%}")
plot_boundary(lambda Z: xor_net(Z)[0].astype(float), X, t, axes[1],
f"2 step units, set by hand: {(xor_net(X)[0] == t).mean():.0%}")
plt.tight_layout()
plt.show()
Step 5: Train a ReLU Network¶
Same training code as Step 4, with ReLU, $h(a) = \max(0, a)$, in the hidden layer. One hidden layer of 8 units is enough.
torch.manual_seed(189)
relu_net = MLP([2, 8, 1], activation="relu")
history = train(relu_net, X, t, epochs=500, lr=0.05)
n_params = sum(p.numel() for p in relu_net.parameters())
print(f"ReLU network ({n_params} parameters): accuracy {accuracy(relu_net, X, t):.3f}, loss {history[-1]:.4f}")
ReLU network (33 parameters): accuracy 1.000, loss 0.0040
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_boundary(lambda Z: xor_net(Z)[0].astype(float), X, t, axes[0],
f"2 step units, set by hand: {(xor_net(X)[0] == t).mean():.0%}")
plot_boundary(predict_proba(relu_net), X, t, axes[1],
f"8 ReLU units, trained: {accuracy(relu_net, X, t):.0%}")
plt.tight_layout()
plt.show()
Step 6: Activation Functions and Their Derivatives¶
Any non-linear $h$ prevents the collapse. Which one we pick matters for training. Gradient descent (Lectures 10-11) moves each weight in proportion to its gradient, and the gradient that reaches a unit's incoming weights is multiplied by $h'(a)$ at that unit (Lecture 13 makes this precise). Where $h'(a) \approx 0$, those weights barely move.
names, colors = ["sigmoid", "tanh", "relu", "gelu"], [C_BLUE, C_GOLD, C_RED, C_GREEN]
a = torch.linspace(-6, 6, 1201)
eps = 1e-3
fig, axes = plt.subplots(1, 2, figsize=(12, 4.5))
for name, color in zip(names, colors):
h = ACTIVATIONS[name]
axes[0].plot(a, h(a), color=color, lw=2.5, label=name)
axes[1].plot(a, (h(a + eps) - h(a - eps)) / (2 * eps), color=color, lw=2.5, label=name) # numerical derivative
axes[0].set(title="$h(a)$", xlabel="$a$", ylim=(-1.5, 3))
axes[1].set(title="$h'(a)$", xlabel="$a$", ylim=(-0.2, 1.2))
axes[0].legend(loc="upper left")
plt.tight_layout()
plt.show()
pts = torch.tensor([-5.0, -1.0, 1.0, 5.0])
print(" h'(-5) h'(-1) h'(1) h'(5)")
for name in names:
h = ACTIVATIONS[name]
d = (h(pts + eps) - h(pts - eps)) / (2 * eps)
print(f"{name:8s}" + "".join(f"{v:9.4f}" for v in d.tolist()))
h'(-5) h'(-1) h'(1) h'(5) sigmoid 0.0066 0.1966 0.1966 0.0066 tanh 0.0002 0.4200 0.4200 0.0002 relu 0.0000 0.0000 1.0000 0.9999 gelu 0.0000 -0.0833 1.0833 0.9999
Sigmoid's derivative is at most 0.25, and both sigmoid and tanh flatten out for large $|a|$: they saturate. ReLU's derivative is exactly 1 for $a > 0$, one reason it became the default hidden activation. GELU, used in transformers later in the course, is a smooth version of ReLU. For $a < 0$ ReLU's derivative is exactly 0, which has its own failure.
The failure: a dead ReLU unit¶
Poll 2 comes before this cell. Train an 8-unit ReLU network on the moons data, but first set unit 0's bias to $-10$ so its pre-activation is negative for every training point. Then measure how far each unit's incoming weights move.
Xm, tm = make_moons(400, noise=0.2, random_state=189)
torch.manual_seed(189)
net = MLP([2, 8, 1], activation="relu")
with torch.no_grad():
net.layers[0].bias[0] = -10.0
pre = torch.as_tensor(Xm, dtype=torch.float32) @ net.layers[0].weight.T + net.layers[0].bias
print(f"unit 0 pre-activation, largest value over the data: {pre[:, 0].max():.2f}")
W_before = net.layers[0].weight.detach().clone()
_ = train(net, Xm, tm, epochs=500, lr=0.01)
moved = (net.layers[0].weight.detach() - W_before).norm(dim=1)
print("distance each unit's incoming weights moved:", moved.numpy().round(3))
unit 0 pre-activation, largest value over the data: -9.42 distance each unit's incoming weights moved: [0. 0.898 1.454 0.687 2.146 2.21 1.158 0.252]
fig, ax = plt.subplots(figsize=(9, 4.5))
ax.bar(range(8), moved.numpy(), color=C_BLUE)
ax.text(0, 0.05, "0.000", ha="center", va="bottom", color=C_RED, fontsize=14, fontweight="bold")
ax.set(xlabel="hidden unit", ylabel="distance moved", xticks=range(8),
title="How far each unit's incoming weights moved in training")
ax.get_xticklabels()[0].set_color(C_RED)
plt.show()
Unit 0 outputs 0 for every input, its derivative is 0 at every input, so its incoming weights get zero gradient and never move. It stays dead for the rest of training. Lecture 14 covers initialization, which makes this less likely.
Step 7: Output Layers and Loss¶
The hidden layers compute features $z = \phi(x;\, W^{(1)}, \dots, W^{(L-1)})$. The output layer is a model from Lectures 8-9 applied to those features, so the output activation and the loss come from the same maximum-likelihood argument:
| Target | Output activation | Likelihood | Loss | PyTorch loss |
|---|---|---|---|---|
| a real number | identity | Gaussian | squared error | nn.MSELoss |
| one of 2 classes | sigmoid | Bernoulli | binary cross-entropy | nn.BCEWithLogitsLoss |
| one of $K$ classes | softmax | categorical | cross-entropy | nn.CrossEntropyLoss |
| any subset of $K$ labels | $K$ sigmoids | $K$ Bernoullis | sum of $K$ binary cross-entropies | nn.BCEWithLogitsLoss |
The two cross-entropy losses take logits and apply the sigmoid or softmax themselves, so our MLP returns logits.
Same hidden layers, different output layers¶
Regression: a noisy sine curve, identity output, squared error.
rng = np.random.default_rng(189)
x_reg = np.sort(rng.uniform(-3, 3, 200))[:, None]
t_reg = np.sin(2 * x_reg[:, 0]) + 0.2 * rng.standard_normal(200)
torch.manual_seed(189)
reg_net = MLP([1, 32, 32, 1], activation="tanh")
history = train(reg_net, x_reg, t_reg, epochs=1000, lr=0.01, loss_fn=nn.MSELoss())
print(f"final mean squared error {history[-1]:.4f} (noise variance 0.04)")
final mean squared error 0.0426 (noise variance 0.04)
xg = np.linspace(-3.5, 3.5, 300)[:, None]
with torch.no_grad():
yg = reg_net(torch.as_tensor(xg, dtype=torch.float32)).numpy()
plt.scatter(x_reg, t_reg, s=14, color=C_BLUE, alpha=0.6, label="data")
plt.plot(xg, yg, color=C_GOLD, lw=3, label="network")
plt.plot(xg, np.sin(2 * xg), "k--", lw=1, label=r"$\sin(2x)$")
plt.xlabel("$x$")
plt.legend()
plt.show()
Multi-class: three interleaved spirals, softmax output, cross-entropy. Compare with softmax regression on the raw inputs (no hidden layer).
Xs, ts = make_spiral()
torch.manual_seed(189)
softmax_reg = MLP([2, 3]) # no hidden layer
_ = train(softmax_reg, Xs, ts, epochs=500, lr=0.01)
torch.manual_seed(189)
spiral_net = MLP([2, 32, 32, 3], activation="relu")
_ = train(spiral_net, Xs, ts, epochs=500, lr=0.01)
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_boundary(predict_proba(softmax_reg), Xs, ts, axes[0], f"Softmax regression: {accuracy(softmax_reg, Xs, ts):.0%}")
plot_boundary(predict_proba(spiral_net), Xs, ts, axes[1], f"2 hidden layers + softmax: {accuracy(spiral_net, Xs, ts):.0%}")
plt.tight_layout()
plt.show()
The failure: softmax overflows¶
Softmax exponentiates the logits, and large logits overflow:
logits = np.array([1000.0, 1001.0, 1002.0])
naive = np.exp(logits) / np.exp(logits).sum()
print("naive softmax: ", naive)
shifted = np.exp(logits - logits.max()) # softmax(a) = softmax(a - c) for any constant c
print("shifted softmax:", (shifted / shifted.sum()).round(4))
naive softmax: [nan nan nan] shifted softmax: [0.09 0.2447 0.6652]
PyTorch's cross-entropy takes logits and does this shift internally (the log-sum-exp trick), so it never forms the overflowing exponentials:
z = torch.tensor([[1000.0, 1001.0, 1002.0]])
print("F.cross_entropy: ", round(F.cross_entropy(z, torch.tensor([2])).item(), 4))
print("-log softmax, by hand:", round(-F.log_softmax(z, dim=1)[0, 2].item(), 4))
F.cross_entropy: 0.4076 -log softmax, by hand: 0.4076
Pass logits, not probabilities¶
A common bug: apply softmax inside the model, then use nn.CrossEntropyLoss, which applies softmax again.
z = torch.tensor([[10.0, 0.0, 0.0]]) # confident and correct: the target is class 0
target = torch.tensor([0])
print("logits in: ", round(F.cross_entropy(z, target).item(), 5))
print("probabilities in:", round(F.cross_entropy(F.softmax(z, dim=1), target).item(), 5))
logits in: 9e-05 probabilities in: 0.5515
Step 8: The PyTorch Pieces¶
Everything above used two helpers, MLP and train. This section opens them up. Nothing after it depends on it, so it can be skipped.
Four ideas:
torch.tensor: an array like a NumPy array, which can live on a GPU and records how it was computed.nn.Module: a model.__init__creates the layers (and their parameters);forwardcomputes the output.- Autograd:
loss.backward()computes the gradient of the loss with respect to every parameter. How it does that is Lecture 13. - Optimizers:
torch.optim.SGDandtorch.optim.Adamapply the update rules from Lecture 11.
x = torch.tensor([[1.0, 2.0], [3.0, 4.0]])
print(x.shape, x.dtype)
print(x @ x.T) # the same operators as NumPy
print(torch.from_numpy(X[:2]).float()) # NumPy -> torch
print(x.numpy()) # torch -> NumPy
torch.Size([2, 2]) torch.float32
tensor([[ 5., 11.],
[11., 25.]])
tensor([[ 0.1006, 0.0177],
[ 0.1533, -0.0952]])
[[1. 2.]
[3. 4.]]
nn.Linear(d_in, d_out) holds a weight matrix of shape (d_out, d_in) and a bias of length d_out. Counting the parameters of a network with layer sizes 784, 128, 64, and 10 (slides 18-19):
model = MLP([784, 128, 64, 10])
print(model)
print(f"{sum(p.numel() for p in model.parameters()):,} parameters")
print(f"{(784*128 + 128) + (128*64 + 64) + (64*10 + 10):,} by hand: weights + biases, layer by layer")
MLP(
(layers): ModuleList(
(0): Linear(in_features=784, out_features=128, bias=True)
(1): Linear(in_features=128, out_features=64, bias=True)
(2): Linear(in_features=64, out_features=10, bias=True)
)
)
109,386 parameters
109,386 by hand: weights + biases, layer by layer
One training step, written out. train repeats this epochs times:
torch.manual_seed(189)
model = MLP([2, 8, 1], activation="relu")
optimizer = torch.optim.Adam(model.parameters(), lr=0.05)
loss_fn = nn.BCEWithLogitsLoss()
X_torch = torch.as_tensor(X, dtype=torch.float32)
t_torch = torch.as_tensor(t, dtype=torch.float32)[:, None]
optimizer.zero_grad() # clear gradients left from the previous step
loss = loss_fn(model(X_torch), t_torch) # forward pass: inputs -> logits -> loss
loss.backward() # gradient of the loss w.r.t. every parameter (Lecture 13)
optimizer.step() # Adam update (Lecture 11)
print(f"loss {loss.item():.4f}")
print("gradient of the first layer's weights:\n", model.layers[0].weight.grad)
loss 0.6830
gradient of the first layer's weights:
tensor([[ 0.0000e+00, 0.0000e+00],
[ 1.1526e-03, 1.1808e-03],
[ 1.3112e-03, 1.3434e-03],
[-4.1051e-02, -1.6660e-03],
[ 9.5359e-04, 7.0638e-04],
[-1.8018e-03, -2.4530e-02],
[ 1.9409e-03, 3.5108e-05],
[ 0.0000e+00, 0.0000e+00]])
A row of exact zeros is a unit whose pre-activation is negative on every training point at this initialization: the dead-unit case from Step 6, this time by chance.
Step 9: Width, Depth, and Activation (Demo 2)¶
Pick a dataset and an architecture, press Run Interact, and look at the boundary and the loss curve. Things to try:
- moons, depth 1, width 1, then 8, then 64: what happens to the boundary as units are added?
- spiral, identity vs. ReLU at the same size
- depth 4, sigmoid vs. ReLU (Step 6 says why sigmoid is slower)
DATASETS = {"xor": make_xor(), "moons": make_moons(400, noise=0.2, random_state=189), "spiral": make_spiral()}
def explore(dataset="moons", depth=1, width=8, activation="relu", epochs=500, lr=0.01):
Xd, td = DATASETS[dataset]
K = len(np.unique(td))
torch.manual_seed(189)
model = MLP([2] + [width] * depth + [1 if K == 2 else K], activation)
history = train(model, Xd, td, epochs=epochs, lr=lr)
fig, axes = plt.subplots(1, 2, figsize=(12, 5))
plot_boundary(predict_proba(model), Xd, td, axes[0], f"accuracy {accuracy(model, Xd, td):.1%}")
axes[1].plot(history, color=C_BLUE, lw=2)
axes[1].set(xlabel="epoch", ylabel="training loss", yscale="log")
plt.tight_layout()
plt.show()
widgets.interact_manual(
explore,
dataset=["xor", "moons", "spiral"],
depth=widgets.IntSlider(value=1, min=1, max=5),
width=widgets.IntSlider(value=8, min=1, max=64),
activation=["identity", "relu", "tanh", "sigmoid", "gelu"],
epochs=widgets.IntSlider(value=500, min=50, max=2000, step=50),
lr=widgets.FloatLogSlider(value=0.01, base=10, min=-3, max=-1),
);
The same comparison as a fixed figure: moons with one hidden layer of 1, 8, and 64 ReLU units, then two layers of 32.
Xd, td = DATASETS["moons"]
fig, axes = plt.subplots(1, 4, figsize=(18, 4.8))
for ax, hidden in zip(axes, [[1], [8], [64], [32, 32]]):
torch.manual_seed(189)
m = MLP([2] + hidden + [1], activation="relu")
_ = train(m, Xd, td, epochs=500, lr=0.01)
plot_boundary(predict_proba(m), Xd, td, ax, f"hidden {hidden}: {accuracy(m, Xd, td):.0%}")
plt.tight_layout()
plt.show()