Lecture 12: Neural Networks (1) – CS 189, Fall 2026

Logistic regression fits a linear boundary on whatever features $\phi(x)$ we give it. Today the model learns the features. We will:

  • Watch logistic regression fail on XOR, then solve XOR with a two-layer network whose weights we set by hand.
  • Train a deep network with no activation function and check that it is still a linear model.
  • Add a non-linearity and watch the boundary bend.
  • Compare sigmoid, tanh, ReLU, and GELU by where their derivatives vanish, and watch a ReLU unit die.
  • Match the output layer and loss to the target, compute softmax cross-entropy from logits, and see that the loss is no longer convex.

Reference: Bishop & Bishop, Deep Learning: Foundations and Concepts, Ch. 6, §6.1, §6.2.1-6.2.3, §6.3.1, §6.4.

Imports¶

In [1]:
import numpy as np
import matplotlib.pyplot as plt
from matplotlib.colors import LinearSegmentedColormap
from sklearn.datasets import make_moons
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import log_loss

import torch
import torch.nn as nn
import torch.nn.functional as F

import warnings
warnings.filterwarnings("ignore")

import ipywidgets as widgets

%config InlineBackend.figure_format = "retina"

C_BLUE, C_GOLD, C_RED, C_GREEN = "#003262", "#FDB515", "#D55E00", "#00553A"
CLASS_COLORS = [C_BLUE, C_GOLD, C_RED]
plt.rcParams.update({"figure.figsize": (8, 5), "axes.grid": True, "grid.alpha": 0.3,
                     "font.size": 13, "axes.titlesize": 15, "axes.labelsize": 14})

Helper Code¶

Run these two cells before the rest. You do not need to read them today: Step 8 opens up MLP and train, and Lecture 13 explains loss.backward().

In [2]:
def make_xor(n=400, noise=0.15, seed=189):
    """n points split evenly over the corners of the unit square, plus Gaussian noise.
    The label is x1 XOR x2 of the corner."""
    rng = np.random.default_rng(seed)
    corners = np.repeat(np.array([[0, 0], [0, 1], [1, 0], [1, 1]]), n // 4, axis=0)
    X = corners + noise * rng.standard_normal(corners.shape)
    return X, corners[:, 0] ^ corners[:, 1]


def make_spiral(n_per_class=150, K=3, noise=0.2, seed=189):
    """K interleaved spiral arms, one class per arm."""
    rng = np.random.default_rng(seed)
    r = np.linspace(0.2, 1, n_per_class)
    X, t = [], []
    for k in range(K):
        theta = np.linspace(0, 4, n_per_class) + 2 * np.pi * k / K + noise * rng.standard_normal(n_per_class)
        X.append(np.c_[r * np.cos(theta), r * np.sin(theta)])
        t.append(np.full(n_per_class, k))
    return np.vstack(X), np.concatenate(t)


PROB_CMAP = LinearSegmentedColormap.from_list("blue_white_gold", ["#B9CCE2", "#FFFFFF", "#FEE3A0"])


def plot_data(X, t, ax=None, title=None):
    ax = ax or plt.gca()
    for k in np.unique(t):
        ax.scatter(*X[t == k].T, s=16, color=CLASS_COLORS[k], edgecolor="k", linewidth=0.3, label=f"class {k}")
    ax.set(xlabel="$x_1$", ylabel="$x_2$", title=title, aspect="equal")
    return ax


def plot_boundary(predict, X, t, ax=None, title=None, pad=0.5, n=300):
    """Shade the plane by the model's prediction, then overlay the data.
    predict maps an (m, 2) array to P(class 1) with shape (m,), or to class probabilities (m, K)."""
    ax = ax or plt.gca()
    lo, hi = X.min(0) - pad, X.max(0) + pad
    g1, g2 = np.meshgrid(np.linspace(lo[0], hi[0], n), np.linspace(lo[1], hi[1], n))
    P = predict(np.c_[g1.ravel(), g2.ravel()])
    if P.ndim == 1:
        ax.contourf(g1, g2, P.reshape(g1.shape), levels=np.linspace(0, 1, 21), cmap=PROB_CMAP)
        ax.contour(g1, g2, P.reshape(g1.shape), levels=[0.5], colors="k", linewidths=1.5)
    else:
        K = P.shape[1]
        ax.contourf(g1, g2, P.argmax(1).reshape(g1.shape), levels=np.arange(K + 1) - 0.5,
                    colors=[c + "40" for c in CLASS_COLORS[:K]])
    plot_data(X, t, ax, title)
    ax.grid(False)
    return ax
In [3]:
ACTIVATIONS = {"identity": lambda a: a, "relu": torch.relu, "tanh": torch.tanh,
               "sigmoid": torch.sigmoid, "gelu": F.gelu}


class MLP(nn.Module):
    """Fully connected network. dims = [inputs, hidden_1, ..., hidden_L, outputs].
    Every hidden layer applies the activation. The output layer returns logits (no activation)."""

    def __init__(self, dims, activation="relu"):
        super().__init__()
        self.layers = nn.ModuleList(nn.Linear(d_in, d_out) for d_in, d_out in zip(dims[:-1], dims[1:]))
        self.act = ACTIVATIONS[activation]

    def forward(self, x):
        for layer in self.layers[:-1]:
            x = self.act(layer(x))
        return self.layers[-1](x)


def train(model, X, t, epochs=500, lr=0.01, loss_fn=None):
    """Full-batch Adam (Lecture 11). Unless a loss is given, it follows the output layer:
    1 output -> binary cross-entropy on logits; K outputs -> softmax cross-entropy on logits."""
    X = torch.as_tensor(X, dtype=torch.float32)
    K = model.layers[-1].out_features
    if loss_fn is None:
        loss_fn = nn.BCEWithLogitsLoss() if K == 1 else nn.CrossEntropyLoss()
    if isinstance(loss_fn, nn.CrossEntropyLoss):
        T = torch.as_tensor(t, dtype=torch.long)
    else:
        T = torch.as_tensor(t, dtype=torch.float32).reshape(-1, K)
    optimizer = torch.optim.Adam(model.parameters(), lr=lr)
    history = []
    for epoch in range(epochs):
        optimizer.zero_grad()
        loss = loss_fn(model(X), T)
        loss.backward()                      # gradients: Lecture 13
        optimizer.step()
        history.append(loss.item())
    return history


def predict_proba(model):
    """Wrap a trained classifier as a NumPy function: P(class 1), or class probabilities."""
    def f(X):
        with torch.no_grad():
            z = model(torch.as_tensor(X, dtype=torch.float32))
        return torch.sigmoid(z)[:, 0].numpy() if z.shape[1] == 1 else torch.softmax(z, dim=1).numpy()
    return f


def accuracy(model, X, t):
    P = predict_proba(model)(X)
    pred = (P > 0.5).astype(int) if P.ndim == 1 else P.argmax(1)
    return (pred == t).mean()

The Data: XOR¶

Four clusters at the corners of the unit square. The label is $x_1 \text{ XOR } x_2$ of the corner: class 1 when exactly one coordinate is 1.

In [4]:
X, t = make_xor()
plot_data(X, t, title="XOR: 400 points, 100 per corner")
plt.legend(loc="center")
plt.show()
No description has been provided for this image

Step 1: Logistic Regression on XOR (Demo 1)¶

Logistic regression from Lectures 8-9, with the raw coordinates as features, $\phi(x) = (x_1, x_2)$, and no regularization.

In [5]:
logreg = LogisticRegression(C=np.inf).fit(X, t)
p_lr = logreg.predict_proba(X)[:, 1]
print(f"weights {logreg.coef_.ravel().round(3)}, bias {logreg.intercept_.round(3)}")
print(f"accuracy {logreg.score(X, t):.3f}, cross-entropy {log_loss(t, p_lr):.4f}  (ln 2 = {np.log(2):.4f})")
print(f"predicted P(class 1) ranges from {p_lr.min():.3f} to {p_lr.max():.3f}")
weights [0.044 0.083], bias [-0.063]
accuracy 0.497, cross-entropy 0.6928  (ln 2 = 0.6931)
predicted P(class 1) ranges from 0.475 to 0.524

The failure: the best line is no line¶

Any line leaves at least one cluster on the wrong side. Cross-entropy punishes confident mistakes, so the best fit shrinks the weights toward zero and predicts about 0.5 everywhere. Its loss is $\ln 2$, the loss of a coin flip.

In [6]:
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_data(X, t, axes[0], "The data")
plot_boundary(lambda Z: logreg.predict_proba(Z)[:, 1], X, t, axes[1],
              f"Logistic regression: accuracy {logreg.score(X, t):.0%}")
plt.tight_layout()
plt.show()
No description has been provided for this image

Step 2: One Neuron, Wired by Hand¶

A neuron takes a weighted sum of its inputs plus a bias, then applies an activation: $z = h(w_1 x_1 + w_2 x_2 + w_0)$. With the step activation, $h(a) = 1$ if $a > 0$ and $0$ otherwise, a neuron draws one line and reports which side of it you are on.

In [7]:
def step(a):
    return (a > 0).astype(int)


def neuron(X, w, w0, h=step):
    return h(X @ np.asarray(w) + w0)


corners = np.array([[0, 0], [0, 1], [1, 0], [1, 1]])
print("x1 x2 | AND OR")
for x, a, o in zip(corners, neuron(corners, [1, 1], -1.5), neuron(corners, [1, 1], -0.5)):
    print(f" {x[0]}  {x[1]} |  {a}   {o}")
x1 x2 | AND OR
 0  0 |  0   0
 0  1 |  0   1
 1  0 |  0   1
 1  1 |  1   1

AND and OR each need one line. XOR needs class 1 at $(0, 1)$ and $(1, 0)$ and class 0 at $(0, 0)$ and $(1, 1)$. No single line does that, so no single neuron does.


Step 3: Two Layers Solve XOR¶

Write XOR as $(x_1 \text{ AND NOT } x_2) \text{ OR } (\text{NOT } x_1 \text{ AND } x_2)$. Each piece is one neuron:

  • hidden unit 1: $z_1 = h(x_1 - x_2 - 0.5)$, on only at $(1, 0)$
  • hidden unit 2: $z_2 = h(-x_1 + x_2 - 0.5)$, on only at $(0, 1)$
  • output unit: $y = h(z_1 + z_2 - 0.5)$, the OR of the two

Three neurons and nine weights, all set by hand.

In [8]:
W1 = np.array([[1, -1], [-1, 1]]);  b1 = np.array([-0.5, -0.5])   # hidden layer: 2 units
w2 = np.array([1, 1]);              b2 = -0.5                       # output unit: OR


def xor_net(X, h=step):
    Z = h(X @ W1.T + b1)             # hidden layer
    return step(Z @ w2 + b2), Z      # output, hidden activations


y, Z = xor_net(corners)
print("x1 x2 | z1 z2 | y")
for x, z, v in zip(corners, Z, y):
    print(f" {x[0]}  {x[1]} |  {z[0]}  {z[1]} | {v}")
print(f"\naccuracy on the 400 noisy points: {(xor_net(X)[0] == t).mean():.4f}")
x1 x2 | z1 z2 | y
 0  0 |  0  0 | 0
 0  1 |  0  1 | 1
 1  0 |  1  0 | 1
 1  1 |  0  0 | 0

accuracy on the 400 noisy points: 0.9775
In [9]:
plot_boundary(lambda Z: xor_net(Z)[0].astype(float), X, t,
              title=f"Hand-wired network: accuracy {(xor_net(X)[0] == t).mean():.1%}")
plt.show()
No description has been provided for this image

The hidden layer is a new $\phi(x)$¶

Plot each point by its hidden activations $(z_1, z_2)$ instead of its inputs. Here the hidden step is replaced by a steep sigmoid, $\sigma(5a)$, so the points spread out instead of piling onto three corners. In the new coordinates one line separates the classes: the output unit is logistic regression on features that the hidden layer computed. (The points fall on a single curve because both hidden units read the same quantity, $x_1 - x_2$.)

In [10]:
steep = lambda a: 1 / (1 + np.exp(-5 * a))
_, Z_smooth = xor_net(X, h=steep)

fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_data(X, t, axes[0], "Input space $(x_1, x_2)$")
plot_data(Z_smooth, t, axes[1], "Hidden space $(z_1, z_2)$")
axes[1].plot([-0.1, 0.6], [0.6, -0.1], color="k", lw=1.5)      # the output unit's line: z1 + z2 = 0.5
axes[1].set(xlabel="$z_1$", ylabel="$z_2$")
plt.tight_layout()
plt.show()
No description has been provided for this image

Step 4: Depth Without a Non-Linearity (Demo 1, continued)¶

Poll 1 comes before this cell. Now we train a network instead of wiring one: three hidden layers of 100 units each, with the identity as the activation, $h(a) = a$.

In [11]:
torch.manual_seed(189)
deep_linear = MLP([2, 100, 100, 100, 1], activation="identity")
history = train(deep_linear, X, t, epochs=500, lr=0.01)
n_params = sum(p.numel() for p in deep_linear.parameters())
print(f"deep linear network ({n_params:,} parameters): accuracy {accuracy(deep_linear, X, t):.3f}, loss {history[-1]:.4f}")
print(f"logistic regression (3 parameters):         accuracy {logreg.score(X, t):.3f}, loss {log_loss(t, p_lr):.4f}")
deep linear network (20,601 parameters): accuracy 0.497, loss 0.6928
logistic regression (3 parameters):         accuracy 0.497, loss 0.6928

The failure: three hidden layers, one straight line¶

The same loss as logistic regression, to four decimals. Multiply the layers out and the whole network is a single linear map $Wx + b$:

In [12]:
with torch.no_grad():
    W, b = torch.eye(2), torch.zeros(2)
    for layer in deep_linear.layers:                     # x -> W_l (W x + b) + b_l
        W, b = layer.weight @ W, layer.weight @ b + layer.bias
    Xt = torch.as_tensor(X, dtype=torch.float32)
    gap = (deep_linear(Xt) - (Xt @ W.T + b)).abs().max().item()
print(f"collapsed weights {W.numpy().round(3)}, bias {b.numpy().round(3)}")
print(f"largest difference from the full network's output: {gap:.1e}")
collapsed weights [[0.044 0.083]], bias [-0.063]
largest difference from the full network's output: 4.8e-08
In [13]:
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_boundary(predict_proba(deep_linear), X, t, axes[0],
              f"3 hidden layers of 100, identity: {accuracy(deep_linear, X, t):.0%}")
plot_boundary(lambda Z: xor_net(Z)[0].astype(float), X, t, axes[1],
              f"2 step units, set by hand: {(xor_net(X)[0] == t).mean():.0%}")
plt.tight_layout()
plt.show()
No description has been provided for this image

Step 5: Train a ReLU Network¶

Same training code as Step 4, with ReLU, $h(a) = \max(0, a)$, in the hidden layer. One hidden layer of 8 units is enough.

In [14]:
torch.manual_seed(189)
relu_net = MLP([2, 8, 1], activation="relu")
history = train(relu_net, X, t, epochs=500, lr=0.05)
n_params = sum(p.numel() for p in relu_net.parameters())
print(f"ReLU network ({n_params} parameters): accuracy {accuracy(relu_net, X, t):.3f}, loss {history[-1]:.4f}")
ReLU network (33 parameters): accuracy 1.000, loss 0.0040
In [15]:
fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_boundary(lambda Z: xor_net(Z)[0].astype(float), X, t, axes[0],
              f"2 step units, set by hand: {(xor_net(X)[0] == t).mean():.0%}")
plot_boundary(predict_proba(relu_net), X, t, axes[1],
              f"8 ReLU units, trained: {accuracy(relu_net, X, t):.0%}")
plt.tight_layout()
plt.show()
No description has been provided for this image

Step 6: Activation Functions and Their Derivatives¶

Any non-linear $h$ prevents the collapse. Which one we pick matters for training. Gradient descent (Lectures 10-11) moves each weight in proportion to its gradient, and the gradient that reaches a unit's incoming weights is multiplied by $h'(a)$ at that unit (Lecture 13 makes this precise). Where $h'(a) \approx 0$, those weights barely move.

In [16]:
names, colors = ["sigmoid", "tanh", "relu", "gelu"], [C_BLUE, C_GOLD, C_RED, C_GREEN]
a = torch.linspace(-6, 6, 1201)
eps = 1e-3

fig, axes = plt.subplots(1, 2, figsize=(12, 4.5))
for name, color in zip(names, colors):
    h = ACTIVATIONS[name]
    axes[0].plot(a, h(a), color=color, lw=2.5, label=name)
    axes[1].plot(a, (h(a + eps) - h(a - eps)) / (2 * eps), color=color, lw=2.5, label=name)   # numerical derivative
axes[0].set(title="$h(a)$", xlabel="$a$", ylim=(-1.5, 3))
axes[1].set(title="$h'(a)$", xlabel="$a$", ylim=(-0.2, 1.2))
axes[0].legend(loc="upper left")
plt.tight_layout()
plt.show()
No description has been provided for this image
In [17]:
pts = torch.tensor([-5.0, -1.0, 1.0, 5.0])
print("          h'(-5)   h'(-1)    h'(1)    h'(5)")
for name in names:
    h = ACTIVATIONS[name]
    d = (h(pts + eps) - h(pts - eps)) / (2 * eps)
    print(f"{name:8s}" + "".join(f"{v:9.4f}" for v in d.tolist()))
          h'(-5)   h'(-1)    h'(1)    h'(5)
sigmoid    0.0066   0.1966   0.1966   0.0066
tanh       0.0002   0.4200   0.4200   0.0002
relu       0.0000   0.0000   1.0000   0.9999
gelu       0.0000  -0.0833   1.0833   0.9999

Sigmoid's derivative is at most 0.25, and both sigmoid and tanh flatten out for large $|a|$: they saturate. ReLU's derivative is exactly 1 for $a > 0$, one reason it became the default hidden activation. GELU, used in transformers later in the course, is a smooth version of ReLU. For $a < 0$ ReLU's derivative is exactly 0, which has its own failure.

The failure: a dead ReLU unit¶

Poll 2 comes before this cell. Train an 8-unit ReLU network on the moons data, but first set unit 0's bias to $-10$ so its pre-activation is negative for every training point. Then measure how far each unit's incoming weights move.

In [18]:
Xm, tm = make_moons(400, noise=0.2, random_state=189)
torch.manual_seed(189)
net = MLP([2, 8, 1], activation="relu")
with torch.no_grad():
    net.layers[0].bias[0] = -10.0
    pre = torch.as_tensor(Xm, dtype=torch.float32) @ net.layers[0].weight.T + net.layers[0].bias
print(f"unit 0 pre-activation, largest value over the data: {pre[:, 0].max():.2f}")

W_before = net.layers[0].weight.detach().clone()
_ = train(net, Xm, tm, epochs=500, lr=0.01)
moved = (net.layers[0].weight.detach() - W_before).norm(dim=1)
print("distance each unit's incoming weights moved:", moved.numpy().round(3))
unit 0 pre-activation, largest value over the data: -9.42
distance each unit's incoming weights moved: [0.    0.898 1.454 0.687 2.146 2.21  1.158 0.252]
In [19]:
fig, ax = plt.subplots(figsize=(9, 4.5))
ax.bar(range(8), moved.numpy(), color=C_BLUE)
ax.text(0, 0.05, "0.000", ha="center", va="bottom", color=C_RED, fontsize=14, fontweight="bold")
ax.set(xlabel="hidden unit", ylabel="distance moved", xticks=range(8),
       title="How far each unit's incoming weights moved in training")
ax.get_xticklabels()[0].set_color(C_RED)
plt.show()
No description has been provided for this image

Unit 0 outputs 0 for every input, its derivative is 0 at every input, so its incoming weights get zero gradient and never move. It stays dead for the rest of training. Lecture 14 covers initialization, which makes this less likely.


Step 7: Output Layers and Loss¶

The hidden layers compute features $z = \phi(x;\, W^{(1)}, \dots, W^{(L-1)})$. The output layer is a model from Lectures 8-9 applied to those features, so the output activation and the loss come from the same maximum-likelihood argument:

Target Output activation Likelihood Loss PyTorch loss
a real number identity Gaussian squared error nn.MSELoss
one of 2 classes sigmoid Bernoulli binary cross-entropy nn.BCEWithLogitsLoss
one of $K$ classes softmax categorical cross-entropy nn.CrossEntropyLoss
any subset of $K$ labels $K$ sigmoids $K$ Bernoullis sum of $K$ binary cross-entropies nn.BCEWithLogitsLoss

The two cross-entropy losses take logits and apply the sigmoid or softmax themselves, so our MLP returns logits.

Same hidden layers, different output layers¶

Regression: a noisy sine curve, identity output, squared error.

In [20]:
rng = np.random.default_rng(189)
x_reg = np.sort(rng.uniform(-3, 3, 200))[:, None]
t_reg = np.sin(2 * x_reg[:, 0]) + 0.2 * rng.standard_normal(200)

torch.manual_seed(189)
reg_net = MLP([1, 32, 32, 1], activation="tanh")
history = train(reg_net, x_reg, t_reg, epochs=1000, lr=0.01, loss_fn=nn.MSELoss())
print(f"final mean squared error {history[-1]:.4f}  (noise variance 0.04)")
final mean squared error 0.0426  (noise variance 0.04)
In [21]:
xg = np.linspace(-3.5, 3.5, 300)[:, None]
with torch.no_grad():
    yg = reg_net(torch.as_tensor(xg, dtype=torch.float32)).numpy()
plt.scatter(x_reg, t_reg, s=14, color=C_BLUE, alpha=0.6, label="data")
plt.plot(xg, yg, color=C_GOLD, lw=3, label="network")
plt.plot(xg, np.sin(2 * xg), "k--", lw=1, label=r"$\sin(2x)$")
plt.xlabel("$x$")
plt.legend()
plt.show()
No description has been provided for this image

Multi-class: three interleaved spirals, softmax output, cross-entropy. Compare with softmax regression on the raw inputs (no hidden layer).

In [22]:
Xs, ts = make_spiral()
torch.manual_seed(189)
softmax_reg = MLP([2, 3])                                   # no hidden layer
_ = train(softmax_reg, Xs, ts, epochs=500, lr=0.01)
torch.manual_seed(189)
spiral_net = MLP([2, 32, 32, 3], activation="relu")
_ = train(spiral_net, Xs, ts, epochs=500, lr=0.01)

fig, axes = plt.subplots(1, 2, figsize=(11, 5))
plot_boundary(predict_proba(softmax_reg), Xs, ts, axes[0], f"Softmax regression: {accuracy(softmax_reg, Xs, ts):.0%}")
plot_boundary(predict_proba(spiral_net), Xs, ts, axes[1], f"2 hidden layers + softmax: {accuracy(spiral_net, Xs, ts):.0%}")
plt.tight_layout()
plt.show()
No description has been provided for this image

The failure: softmax overflows¶

Softmax exponentiates the logits, and large logits overflow:

In [23]:
logits = np.array([1000.0, 1001.0, 1002.0])
naive = np.exp(logits) / np.exp(logits).sum()
print("naive softmax:  ", naive)
shifted = np.exp(logits - logits.max())          # softmax(a) = softmax(a - c) for any constant c
print("shifted softmax:", (shifted / shifted.sum()).round(4))
naive softmax:   [nan nan nan]
shifted softmax: [0.09   0.2447 0.6652]

PyTorch's cross-entropy takes logits and does this shift internally (the log-sum-exp trick), so it never forms the overflowing exponentials:

In [24]:
z = torch.tensor([[1000.0, 1001.0, 1002.0]])
print("F.cross_entropy:      ", round(F.cross_entropy(z, torch.tensor([2])).item(), 4))
print("-log softmax, by hand:", round(-F.log_softmax(z, dim=1)[0, 2].item(), 4))
F.cross_entropy:       0.4076
-log softmax, by hand: 0.4076

Pass logits, not probabilities¶

A common bug: apply softmax inside the model, then use nn.CrossEntropyLoss, which applies softmax again.

In [25]:
z = torch.tensor([[10.0, 0.0, 0.0]])        # confident and correct: the target is class 0
target = torch.tensor([0])
print("logits in:       ", round(F.cross_entropy(z, target).item(), 5))
print("probabilities in:", round(F.cross_entropy(F.softmax(z, dim=1), target).item(), 5))
logits in:        9e-05
probabilities in: 0.5515

Step 8: The PyTorch Pieces¶

Everything above used two helpers, MLP and train. This section opens them up. Nothing after it depends on it, so it can be skipped.

Four ideas:

  1. torch.tensor: an array like a NumPy array, which can live on a GPU and records how it was computed.
  2. nn.Module: a model. __init__ creates the layers (and their parameters); forward computes the output.
  3. Autograd: loss.backward() computes the gradient of the loss with respect to every parameter. How it does that is Lecture 13.
  4. Optimizers: torch.optim.SGD and torch.optim.Adam apply the update rules from Lecture 11.
In [26]:
x = torch.tensor([[1.0, 2.0], [3.0, 4.0]])
print(x.shape, x.dtype)
print(x @ x.T)                               # the same operators as NumPy
print(torch.from_numpy(X[:2]).float())       # NumPy -> torch
print(x.numpy())                             # torch -> NumPy
torch.Size([2, 2]) torch.float32
tensor([[ 5., 11.],
        [11., 25.]])
tensor([[ 0.1006,  0.0177],
        [ 0.1533, -0.0952]])
[[1. 2.]
 [3. 4.]]

nn.Linear(d_in, d_out) holds a weight matrix of shape (d_out, d_in) and a bias of length d_out. Counting the parameters of a network with layer sizes 784, 128, 64, and 10 (slides 18-19):

In [27]:
model = MLP([784, 128, 64, 10])
print(model)
print(f"{sum(p.numel() for p in model.parameters()):,} parameters")
print(f"{(784*128 + 128) + (128*64 + 64) + (64*10 + 10):,} by hand: weights + biases, layer by layer")
MLP(
  (layers): ModuleList(
    (0): Linear(in_features=784, out_features=128, bias=True)
    (1): Linear(in_features=128, out_features=64, bias=True)
    (2): Linear(in_features=64, out_features=10, bias=True)
  )
)
109,386 parameters
109,386 by hand: weights + biases, layer by layer

One training step, written out. train repeats this epochs times:

In [28]:
torch.manual_seed(189)
model = MLP([2, 8, 1], activation="relu")
optimizer = torch.optim.Adam(model.parameters(), lr=0.05)
loss_fn = nn.BCEWithLogitsLoss()
X_torch = torch.as_tensor(X, dtype=torch.float32)
t_torch = torch.as_tensor(t, dtype=torch.float32)[:, None]

optimizer.zero_grad()                   # clear gradients left from the previous step
loss = loss_fn(model(X_torch), t_torch) # forward pass: inputs -> logits -> loss
loss.backward()                         # gradient of the loss w.r.t. every parameter (Lecture 13)
optimizer.step()                        # Adam update (Lecture 11)
print(f"loss {loss.item():.4f}")
print("gradient of the first layer's weights:\n", model.layers[0].weight.grad)
loss 0.6830
gradient of the first layer's weights:
 tensor([[ 0.0000e+00,  0.0000e+00],
        [ 1.1526e-03,  1.1808e-03],
        [ 1.3112e-03,  1.3434e-03],
        [-4.1051e-02, -1.6660e-03],
        [ 9.5359e-04,  7.0638e-04],
        [-1.8018e-03, -2.4530e-02],
        [ 1.9409e-03,  3.5108e-05],
        [ 0.0000e+00,  0.0000e+00]])

A row of exact zeros is a unit whose pre-activation is negative on every training point at this initialization: the dead-unit case from Step 6, this time by chance.


Step 9: Width, Depth, and Activation (Demo 2)¶

Pick a dataset and an architecture, press Run Interact, and look at the boundary and the loss curve. Things to try:

  • moons, depth 1, width 1, then 8, then 64: what happens to the boundary as units are added?
  • spiral, identity vs. ReLU at the same size
  • depth 4, sigmoid vs. ReLU (Step 6 says why sigmoid is slower)
In [29]:
DATASETS = {"xor": make_xor(), "moons": make_moons(400, noise=0.2, random_state=189), "spiral": make_spiral()}


def explore(dataset="moons", depth=1, width=8, activation="relu", epochs=500, lr=0.01):
    Xd, td = DATASETS[dataset]
    K = len(np.unique(td))
    torch.manual_seed(189)
    model = MLP([2] + [width] * depth + [1 if K == 2 else K], activation)
    history = train(model, Xd, td, epochs=epochs, lr=lr)
    fig, axes = plt.subplots(1, 2, figsize=(12, 5))
    plot_boundary(predict_proba(model), Xd, td, axes[0], f"accuracy {accuracy(model, Xd, td):.1%}")
    axes[1].plot(history, color=C_BLUE, lw=2)
    axes[1].set(xlabel="epoch", ylabel="training loss", yscale="log")
    plt.tight_layout()
    plt.show()
In [30]:
widgets.interact_manual(
    explore,
    dataset=["xor", "moons", "spiral"],
    depth=widgets.IntSlider(value=1, min=1, max=5),
    width=widgets.IntSlider(value=8, min=1, max=64),
    activation=["identity", "relu", "tanh", "sigmoid", "gelu"],
    epochs=widgets.IntSlider(value=500, min=50, max=2000, step=50),
    lr=widgets.FloatLogSlider(value=0.01, base=10, min=-3, max=-1),
);

The same comparison as a fixed figure: moons with one hidden layer of 1, 8, and 64 ReLU units, then two layers of 32.

In [31]:
Xd, td = DATASETS["moons"]
fig, axes = plt.subplots(1, 4, figsize=(18, 4.8))
for ax, hidden in zip(axes, [[1], [8], [64], [32, 32]]):
    torch.manual_seed(189)
    m = MLP([2] + hidden + [1], activation="relu")
    _ = train(m, Xd, td, epochs=500, lr=0.01)
    plot_boundary(predict_proba(m), Xd, td, ax, f"hidden {hidden}: {accuracy(m, Xd, td):.0%}")
plt.tight_layout()
plt.show()
No description has been provided for this image
In [ ]: