# MLP backprop DryBean
course: Module 3 — Deep Learning & NLP
module: Module-3-Deep-Learning-NLP
type: notebook
source_url: https://personal-learn.armco.dev/files/Module-3-Deep-Learning-NLP/General/Lab_Materials-14-03-2026/MLP_backprop_DryBean.ipynb
---
[cell 1 markdown]
# MLP with PyTorch - Backpropagation on Dry Bean Dataset
This notebook trains a **Multi-Layer Perceptron (MLP)** in **PyTorch** to classify 7 types of dry beans from shape measurements.
| | |
|---|---|
| **Dataset** | Dry Bean Dataset (UCI) |
| **Samples** | 13,611 |
| **Features** | 16 geometric shape features |
| **Task** | Multi-class Classification — 7 bean types |
| **Architecture** | Input(16) → Hidden(8, ReLU) → Output(7, Softmax) |
| **Framework** | **PyTorch** |
**Learning goals:**
- Understand how PyTorch builds a **computation graph** during the forward pass
- See how `loss.backward()` runs **backpropagation automatically** via the chain rule
- Watch **weights and gradients evolve** every 10 epochs
- Compare manual backprop math to PyTorch autograd output
---
[cell 2 markdown]
## Step 1: Imports
[cell 3 code]
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import matplotlib.gridspec as gridspec
import torch
import torch.nn as nn
import torch.optim as optim
import torch.nn.functional as F
from torch.utils.data import DataLoader, TensorDataset
from sklearn.preprocessing import LabelEncoder, StandardScaler
from sklearn.model_selection import train_test_split
# Reproducibility
torch.manual_seed(42)
np.random.seed(42)
DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
print(f"PyTorch version : {torch.__version__}")
print(f" Device : {DEVICE}")
[cell 4 markdown]
## Step 2: Load and Explore the Dataset
[cell 5 code]
df = pd.read_excel("Dry_Bean_Dataset.xlsx")
print(f"Dataset Shape : {df.shape}")
print(f"Features (16) : {df.columns[:-1].tolist()}")
print(f"\nClass Distribution:")
vc = df['Class'].value_counts()
for label, count in vc.items():
print(f" {label:>10} : {count:>5} samples ({count/len(df)*100:.1f}%)")
df.head()
[cell 6 code]
# Visualise feature distributions per class
features_to_plot = ['Area', 'Perimeter', 'MajorAxisLength', 'MinorAxisLength',
'Eccentricity', 'roundness', 'Compactness', 'Solidity']
class_names_all = df['Class'].unique()
palette = plt.cm.Set2(np.linspace(0, 1, len(class_names_all)))
fig, axes = plt.subplots(4, 2, figsize=(14, 12))
for ax, feat in zip(axes.flat, features_to_plot):
for cls, col in zip(class_names_all, palette):
vals = df[df['Class'] == cls][feat].values
ax.hist(vals, bins=30, alpha=0.55, color=col, label=cls, density=True)
ax.set_title(feat, fontsize=10, fontweight='bold')
ax.set_xlabel('Value', fontsize=8)
ax.tick_params(labelsize=7)
ax.grid(True, alpha=0.2)
handles = [plt.Rectangle((0,0),1,1, fc=c) for c in palette]
fig.legend(handles, class_names_all, loc='lower center', ncol=7,
fontsize=10, bbox_to_anchor=(0.5, -0.04))
plt.suptitle('Feature Distributions by Bean Class',
fontsize=14, fontweight='bold')
plt.tight_layout()
plt.savefig('feature_distributions.png', dpi=500, bbox_inches='tight')
plt.show()
[cell 7 markdown]
## Step 3: Preprocess and Build PyTorch DataLoaders
[cell 8 code]
# ---------- Features & Labels ----------
X_raw = df.drop(columns=['Class']).values.astype(np.float32) # (13611, 16)
y_raw = df['Class'].values
# ---------- Encode string labels → integers 0..6 ----------
le = LabelEncoder()
y_int = le.fit_transform(y_raw).astype(np.int64) # required by CrossEntropyLoss
class_names = le.classes_
n_classes = len(class_names)
print(f"Classes ({n_classes}): {class_names}")
# ---------- Normalize features ----------
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X_raw).astype(np.float32)
# ---------- Train / Test split ----------
X_train_np, X_test_np, y_train_np, y_test_np = train_test_split(
X_scaled, y_int, test_size=0.2, random_state=42, stratify=y_int
)
# ---------- Convert to PyTorch tensors ----------
# NOTE: CrossEntropyLoss in PyTorch expects raw logits (no softmax)
# and integer class labels — NOT one-hot vectors.
X_train = torch.tensor(X_train_np).to(DEVICE) # (N, 16) float32
X_test = torch.tensor(X_test_np).to(DEVICE)
y_train = torch.tensor(y_train_np).to(DEVICE) # (N,) int64
y_test = torch.tensor(y_test_np).to(DEVICE)
# ---------- DataLoaders ----------
train_loader = DataLoader(TensorDataset(X_train, y_train),
batch_size=256, shuffle=True)
test_loader = DataLoader(TensorDataset(X_test, y_test),
batch_size=256, shuffle=False)
print(f"\nX_train : {X_train.shape} | X_test : {X_test.shape}")
print(f"y_train : {y_train.shape} | y_test : {y_test.shape}")
print(f"\nTrain batches : {len(train_loader)} (batch_size=256)")
print(f"Test batches : {len(test_loader)}")
[cell 9 markdown]
## Step 4: Define the MLP in PyTorch
```
Input Layer Hidden Layer Output Layer
(16 neurons) → (8 neurons, ReLU) → (7 neurons, Softmax)
```
**Key PyTorch design choice for multiclass:**
`nn.CrossEntropyLoss` = `LogSoftmax` + `NLLLoss` **combined**.
So the model outputs **raw logits** (no softmax in `forward`).
Softmax is only applied manually when we need probabilities (e.g., for display).
[cell 10 code]
class MLP(nn.Module):
"""
MLP: Input(16) → Hidden(8, ReLU) → Output(7, raw logits)
PyTorch's nn.Linear(in_features, out_features) creates:
W: (out, in) b: (out,)
and computes: z = x @ W.T + b
We return raw logits from the output layer.
CrossEntropyLoss applies log-softmax + NLL internally.
"""
def __init__(self, n_input=16, n_hidden=8, n_output=7):
super(MLP, self).__init__()
self.hidden = nn.Linear(n_input, n_hidden) # W1:(8,16), b1:(8,)
self.output = nn.Linear(n_hidden, n_output) # W2:(7,8), b2:(7,)
self.relu = nn.ReLU()
def forward(self, x):
"""
Every operation here is recorded by PyTorch's autograd engine
into a dynamic computation graph.
Calling loss.backward() later traverses this graph in reverse.
"""
z1 = self.hidden(x) # z1 = x @ W1.T + b1
a1 = self.relu(z1) # a1 = ReLU(z1)
logits = self.output(a1) # logits = a1 @ W2.T + b2 (raw scores)
return logits # CrossEntropyLoss handles softmax internally
def predict_proba(self, x):
"""Returns softmax probabilities (for evaluation/display only)."""
return F.softmax(self.forward(x), dim=1)
# Instantiate
model = MLP(n_input=16, n_hidden=8, n_output=n_classes).to(DEVICE)
print("=" * 55)
print(" MLP Architecture")
print("=" * 55)
print(model)
print("=" * 55)
print(f"\nParameter breakdown:")
total = 0
for name, param in model.named_parameters():
print(f" {name:25s}: shape={str(param.shape):15s} count={param.numel()}")
total += param.numel()
print(f"\n Total trainable parameters: {total}")
[cell 11 markdown]
## Step 5: How PyTorch Autograd Works
Before training, let's see PyTorch building and traversing the computation graph on a single sample.
[cell 12 code]
print("── PyTorch Autograd Demo — One Sample, One Forward+Backward ──\n")
x_demo = X_train[:1] # shape (1, 16)
y_demo = y_train[:1] # shape (1,) — integer class label
model.zero_grad()
# ── Forward: PyTorch builds a computation graph ───────────────────────
logits_demo = model(x_demo) # (1, 7)
probs_demo = F.softmax(logits_demo, dim=1) # (1, 7)
loss_demo = nn.CrossEntropyLoss()(logits_demo, y_demo)
print(f"Input shape : {x_demo.shape}")
print(f"True class : {class_names[y_demo.item()]} (index {y_demo.item()})")
print(f"\nOutput logits : {logits_demo.detach().numpy()[0].round(4)}")
print(f"After Softmax :")
for cls, p in zip(class_names, probs_demo.detach().numpy()[0]):
marker = ' ← predicted' if p == probs_demo.max().item() else ''
marker += ' ← TRUE' if cls == class_names[y_demo.item()] else ''
print(f" {cls:>10}: {p:.4f}{marker}")
print(f"\nCross-Entropy Loss : {loss_demo.item():.4f}")
# ── Backward: one call — PyTorch applies the full chain rule ──────────
loss_demo.backward()
print("\n── Gradients computed by loss.backward() ──")
for name, param in model.named_parameters():
if param.grad is not None:
g = param.grad
print(f" ∂L/∂{name:20s}: shape={str(g.shape):12s} "
f"mean={g.mean().item():+.6f} norm={g.norm().item():.6f}")
print("\nPyTorch computed ALL gradients automatically — this is backpropagation!")
print(" It applied the chain rule layer by layer, from output back to input.")
[cell 13 markdown]
## Step 6: The PyTorch Training Loop
Every training step follows this exact 5-line recipe:
```python
optimizer.zero_grad() # 1. Clear gradients from previous step
logits = model(X_batch) # 2. Forward pass → builds computation graph
loss = criterion(logits, y_batch)# 3. Compute loss
loss.backward() # 4. BACKPROPAGATION → fills .grad on all params
optimizer.step() # 5. Gradient descent: w ← w − lr × ∂L/∂w
```
[cell 14 code]
def train_epoch(model, loader, criterion, optimizer):
"""Run one full training epoch. Returns avg loss and accuracy."""
model.train()
total_loss, correct, total = 0.0, 0, 0
for X_batch, y_batch in loader:
# 1. Clear stale gradients
optimizer.zero_grad()
# 2. Forward pass
logits = model(X_batch)
# 3. Compute Cross-Entropy loss
# (applies log-softmax + NLL internally)
loss = criterion(logits, y_batch)
# 4. Backpropagation — chain rule through all layers
loss.backward()
# 5. Gradient descent step
optimizer.step()
total_loss += loss.item() * X_batch.size(0)
preds = logits.argmax(dim=1)
correct += (preds == y_batch).sum().item()
total += X_batch.size(0)
return total_loss / total, correct / total
@torch.no_grad()
def evaluate(model, loader, criterion):
"""Evaluate on a DataLoader. Returns avg loss and accuracy."""
model.eval()
total_loss, correct, total = 0.0, 0, 0
for X_batch, y_batch in loader:
logits = model(X_batch)
loss = criterion(logits, y_batch)
total_loss += loss.item() * X_batch.size(0)
correct += (logits.argmax(dim=1) == y_batch).sum().item()
total += X_batch.size(0)
return total_loss / total, correct / total
print("Training and evaluation functions defined!")
[cell 15 markdown]
## Step 7: Train for 50 Epochs - Weight Updates Every 10 Epochs
[cell 16 code]
# ── Hyperparameters ──────────────────────────────────────────────────
LEARNING_RATE = 0.01
EPOCHS = 50
PRINT_EVERY = 10
# ── Fresh model ──────────────────────────────────────────────────────
model = MLP(n_input=16, n_hidden=8, n_output=n_classes).to(DEVICE)
criterion = nn.CrossEntropyLoss() # softmax + NLL combined
optimizer = optim.SGD(model.parameters(), lr=LEARNING_RATE)
# ── History ───────────────────────────────────────────────────────────
history = {
'train_loss': [], 'test_loss': [],
'train_acc' : [], 'test_acc' : [],
'W1_norm' : [], 'W2_norm' : [],
'dW1_norm' : [], 'dW2_norm' : [],
}
print("=" * 78)
print(f" Training | lr={LEARNING_RATE} | hidden=8 | epochs={EPOCHS} | SGD | 7 classes")
print("=" * 78)
hdr = (f"{'Epoch':>6} | {'Train Loss':>11} | {'Train Acc':>10} "
f"| {'Test Loss':>10} | {'Test Acc':>9} | {'|W1| mean':>9} | {'|W2| mean':>9}")
print(hdr)
print("-" * len(hdr))
for epoch in range(1, EPOCHS + 1):
train_loss, train_acc = train_epoch(model, train_loader, criterion, optimizer)
test_loss, test_acc = evaluate(model, test_loader, criterion)
# Capture fresh gradients for logging
model.train()
Xb, yb = next(iter(train_loader))
optimizer.zero_grad()
criterion(model(Xb), yb).backward()
W1 = model.hidden.weight.data
W2 = model.output.weight.data
dW1 = model.hidden.weight.grad
dW2 = model.output.weight.grad
history['train_loss'].append(train_loss)
history['test_loss'].append(test_loss)
history['train_acc'].append(train_acc)
history['test_acc'].append(test_acc)
history['W1_norm'].append(W1.abs().mean().item())
history['W2_norm'].append(W2.abs().mean().item())
history['dW1_norm'].append(dW1.abs().mean().item() if dW1 is not None else 0.0)
history['dW2_norm'].append(dW2.abs().mean().item() if dW2 is not None else 0.0)
if epoch % PRINT_EVERY == 0 or epoch == 1:
print(f"{epoch:>6} | {train_loss:>11.4f} | {train_acc*100:>9.2f}% "
f"| {test_loss:>10.4f} | {test_acc*100:>8.2f}% "
f"| {W1.abs().mean().item():>9.5f} | {W2.abs().mean().item():>9.5f}")
print("-" * len(hdr))
print(f"\n Final Test Accuracy : {test_acc*100:.2f}%")
[cell 17 markdown]
## Step 8: Detailed Weight & Gradient Snapshots Every 10 Epochs
[cell 18 code]
model2 = MLP(n_input=16, n_hidden=8, n_output=n_classes).to(DEVICE)
optimizer2 = optim.SGD(model2.parameters(), lr=LEARNING_RATE)
criterion2 = nn.CrossEntropyLoss()
feature_names = df.columns[:-1].tolist()
print("=" * 72)
print(" Detailed Weight & Gradient Snapshots (every 10 epochs)")
print("=" * 72)
for epoch in range(1, EPOCHS + 1):
train_loss, train_acc = train_epoch(model2, train_loader, criterion2, optimizer2)
# Capture gradients
model2.train()
Xb, yb = next(iter(train_loader))
optimizer2.zero_grad()
criterion2(model2(Xb), yb).backward()
if epoch % PRINT_EVERY == 0:
W1 = model2.hidden.weight.data.cpu().numpy() # (8, 16)
b1 = model2.hidden.bias.data.cpu().numpy() # (8,)
W2 = model2.output.weight.data.cpu().numpy() # (7, 8)
dW1 = model2.hidden.weight.grad.cpu().numpy() # (8, 16)
dW2 = model2.output.weight.grad.cpu().numpy() # (7, 8)
print(f"\n{'─'*72}")
print(f" EPOCH {epoch:>3} | Train Loss = {train_loss:.4f} "
f" Train Acc = {train_acc*100:.2f}%")
print(f"{'─'*72}")
print(" [W1] Input → Hidden (all 8 hidden neurons)")
for h in range(8):
w_row = W1[h] # weights from all 16 inputs into hidden neuron h
print(f" H{h+1}: mean={w_row.mean():+.5f} "
f"std={w_row.std():.5f} "
f"min={w_row.min():+.5f} "
f"max={w_row.max():+.5f} "
f"bias={b1[h]:+.5f}")
print("\n [W2] Hidden → Output (per class)")
for c in range(n_classes):
w_row = W2[c] # weights from all 8 hidden neurons into output class c
print(f" {class_names[c]:>10}: mean={w_row.mean():+.5f} "
f"std={w_row.std():.5f} "
f"min={w_row.min():+.5f} "
f"max={w_row.max():+.5f}")
print(f"\n Gradient norms → "
f"|∂L/∂W1| = {np.abs(dW1).mean():.7f} "
f"|∂L/∂W2| = {np.abs(dW2).mean():.7f}")
print(f" (shrinking gradient norms = network converging)")
print(f"\n{'='*72}")
print(" Weight inspection complete!")
[cell 19 markdown]
## Step 9: Convergence Plots
[cell 20 code]
epochs_x = np.arange(1, EPOCHS + 1)
check_epochs = list(range(10, EPOCHS + 1, 10))
check_idx = [e - 1 for e in check_epochs]
COLORS = {'train': '#e63946', 'test': '#457b9d',
'W1': '#2a9d8f', 'W2': '#e9c46a',
'dW1': '#f4a261', 'dW2': '#264653'}
fig = plt.figure(figsize=(16, 12))
gs = gridspec.GridSpec(2, 2, hspace=0.42, wspace=0.35)
# ── Plot 1: Loss ──────────────────────────────────────────────────────
ax1 = fig.add_subplot(gs[0, 0])
ax1.plot(epochs_x, history['train_loss'], color=COLORS['train'], lw=2.5, label='Train Loss')
ax1.plot(epochs_x, history['test_loss'], color=COLORS['test'], lw=2.5, label='Test Loss', ls='--')
ax1.scatter([e for e in check_epochs], [history['train_loss'][i] for i in check_idx],
color=COLORS['train'], s=80, zorder=5)
ax1.scatter([e for e in check_epochs], [history['test_loss'][i] for i in check_idx],
color=COLORS['test'], s=80, zorder=5, marker='D')
ax1.set_title('Loss Convergence (Cross-Entropy)', fontsize=13, fontweight='bold')
ax1.set_xlabel('Epoch'); ax1.set_ylabel('Cross-Entropy Loss')
ax1.legend(); ax1.grid(True, alpha=0.3)
ax1.set_xticks(range(0, EPOCHS + 1, 5))
# ── Plot 2: Accuracy ─────────────────────────────────────────────────
ax2 = fig.add_subplot(gs[0, 1])
ax2.plot(epochs_x, [a*100 for a in history['train_acc']], color=COLORS['train'], lw=2.5, label='Train Acc')
ax2.plot(epochs_x, [a*100 for a in history['test_acc']], color=COLORS['test'], lw=2.5, label='Test Acc', ls='--')
ax2.scatter([e for e in check_epochs], [history['train_acc'][i]*100 for i in check_idx],
color=COLORS['train'], s=80, zorder=5)
ax2.scatter([e for e in check_epochs], [history['test_acc'][i]*100 for i in check_idx],
color=COLORS['test'], s=80, zorder=5, marker='D')
ax2.set_title('Accuracy Convergence (7 Classes)', fontsize=13, fontweight='bold')
ax2.set_xlabel('Epoch'); ax2.set_ylabel('Accuracy (%)')
ax2.legend(); ax2.grid(True, alpha=0.3)
ax2.set_xticks(range(0, EPOCHS + 1, 5))
# ── Plot 3: Weight Norms ─────────────────────────────────────────────
ax3 = fig.add_subplot(gs[1, 0])
ax3.plot(epochs_x, history['W1_norm'], color=COLORS['W1'], lw=2.5, label='|W1| mean (input→hidden)')
ax3.plot(epochs_x, history['W2_norm'], color=COLORS['W2'], lw=2.5, label='|W2| mean (hidden→output)', ls='--')
for e in check_epochs:
ax3.axvline(e, color='gray', lw=0.8, ls=':', alpha=0.7)
ax3.text(e + 0.2, max(history['W1_norm']) * 0.97, f'e{e}', fontsize=7, color='gray')
ax3.set_title('Mean Absolute Weight Values', fontsize=13, fontweight='bold')
ax3.set_xlabel('Epoch'); ax3.set_ylabel('Mean |w|')
ax3.legend(); ax3.grid(True, alpha=0.3)
ax3.set_xticks(range(0, EPOCHS + 1, 5))
# ── Plot 4: Gradient Norms ───────────────────────────────────────────
ax4 = fig.add_subplot(gs[1, 1])
ax4.semilogy(epochs_x, history['dW1_norm'], color=COLORS['dW1'], lw=2.5, label='|∂L/∂W1| hidden grads')
ax4.semilogy(epochs_x, history['dW2_norm'], color=COLORS['dW2'], lw=2.5, label='|∂L/∂W2| output grads', ls='--')
for e in check_epochs:
ax4.axvline(e, color='gray', lw=0.8, ls=':', alpha=0.7)
ax4.set_title('Gradient Norms (log scale)', fontsize=13, fontweight='bold')
ax4.set_xlabel('Epoch'); ax4.set_ylabel('Mean |gradient| (log)')
ax4.legend(); ax4.grid(True, alpha=0.3)
ax4.set_xticks(range(0, EPOCHS + 1, 5))
fig.suptitle(
'PyTorch MLP — Backpropagation Convergence on Dry Bean Dataset\n'
'Architecture: 16 → 8 (ReLU) → 7 (Softmax) | SGD | lr=0.05',
fontsize=14, fontweight='bold', y=1.01
)
plt.savefig('convergence_plots.png', dpi=130, bbox_inches='tight')
plt.show()
[cell 21 markdown]
## Step 11: Step-by-Step Backprop Trace - One Sample
[cell 22 code]
sample_idx = 5
x_s = X_train[sample_idx:sample_idx+1] # (1, 16)
y_s = y_train[sample_idx:sample_idx+1] # (1,) integer label
true_class = class_names[y_s.item()]
print("=" * 68)
print(f" STEP-BY-STEP BACKPROPAGATION TRACE (sample #{sample_idx})")
print(f" True class : {true_class} (index {y_s.item()})")
print("=" * 68)
# Extract weights as numpy arrays
W1 = model.hidden.weight.data.cpu().numpy() # (8, 16)
b1 = model.hidden.bias.data.cpu().numpy() # (8,)
W2 = model.output.weight.data.cpu().numpy() # (7, 8)
b2 = model.output.bias.data.cpu().numpy() # (7,)
x_np = x_s.cpu().numpy() # (1, 16)
y_idx = y_s.item() # integer: 0..6
# ────────────────────────────────────────────────────────────────────
# FORWARD PASS
# ────────────────────────────────────────────────────────────────────
print("\n──── FORWARD PASS ────")
Z1 = x_np @ W1.T + b1 # (1, 8)
A1 = np.maximum(0, Z1) # ReLU → (1, 8)
print(f"Z1 (pre-ReLU) : {Z1[0].round(4)}")
print(f"A1 (post-ReLU) : {A1[0].round(4)}")
print(f" ({np.sum(Z1[0] <= 0)} of 8 neurons are dead/zero after ReLU)")
Z2 = A1 @ W2.T + b2 # (1, 7) raw logits
exp_z = np.exp(Z2 - Z2.max(axis=1, keepdims=True)) # numerically stable
A2 = exp_z / exp_z.sum(axis=1, keepdims=True) # softmax → (1, 7)
print(f"\nZ2 (logits) : {Z2[0].round(4)}")
print(f"\nA2 (softmax probabilities):")
for cls, p in zip(class_names, A2[0]):
marker = ' ← predicted' if p == A2[0].max() else ''
marker += ' ← TRUE' if cls == true_class else ''
print(f" {cls:>10}: {p:.4f}{marker}")
ce_loss = -np.log(A2[0, y_idx] + 1e-9)
print(f"\nCross-Entropy loss = -log(p_true) = -log({A2[0, y_idx]:.4f}) = {ce_loss:.4f}")
# ────────────────────────────────────────────────────────────────────
# BACKWARD PASS — manual chain rule
# ────────────────────────────────────────────────────────────────────
print("\n──── BACKWARD PASS (manual chain rule) ────")
# δ² = Softmax(Z2) − one_hot(y)
# This clean form comes from d(CrossEntropy)/dZ2 = A2 - Y
delta2 = A2.copy() # start from softmax output
delta2[0, y_idx] -= 1.0 # subtract 1 at the true class position (1, 7)
print(f"δ² = A2 − one_hot(y) [output layer error]:")
for cls, d in zip(class_names, delta2[0]):
marker = ' ← true class (subtract 1 here)' if cls == true_class else ''
print(f" {cls:>10}: {d:+.4f}{marker}")
dW2 = delta2.T @ A1 # (7, 8)
db2 = delta2 # (1, 7)
print(f"\n∂L/∂W2 shape={dW2.shape} norm={np.abs(dW2).mean():.6f}")
# Backprop through W2, then through ReLU
delta1 = (delta2 @ W2) * (Z1 > 0).astype(float) # (1, 8)
dW1 = delta1.T @ x_np # (8, 16)
db1 = delta1 # (1, 8)
print(f"\nδ¹ (hidden layer error via chain rule):")
print(f" {delta1[0].round(6)}")
print(f"\n∂L/∂W1 shape={dW1.shape} norm={np.abs(dW1).mean():.6f}")
[cell 23 markdown]
## Step 12: Final Evaluation — Per-Class Accuracy
[cell 24 code]
model.eval()
with torch.no_grad():
all_preds = model(X_test).argmax(dim=1).cpu().numpy()
all_true = y_test.cpu().numpy()
final_acc = np.mean(all_preds == all_true)
print("=" * 55)
print(" Final Evaluation on Test Set")
print("=" * 55)
print(f" Overall Accuracy : {final_acc * 100:.2f}%\n")
print(f" {'Class':>10} | {'Correct':>7} | {'Total':>7} | {'Accuracy':>9}")
print(f" {'─'*42}")
for i, cls in enumerate(class_names):
mask = all_true == i
n_total = mask.sum()
n_correct = (all_preds[mask] == i).sum()
acc_cls = n_correct / n_total if n_total > 0 else 0.0
print(f" {cls:>10} | {n_correct:>7} | {n_total:>7} | {acc_cls*100:>8.2f}%")
# Confusion matrix
print(f"\n Confusion Matrix (rows=true, cols=predicted):")
conf = np.zeros((n_classes, n_classes), dtype=int)
for t, p in zip(all_true, all_preds):
conf[t, p] += 1
header_str = f" {'':>10} " + " ".join(f"{c:>8}" for c in class_names)
print(header_str)
for i, cls in enumerate(class_names):
row = " ".join(f"{conf[i,j]:>8}" for j in range(n_classes))
print(f" {cls:>10} {row}")
print(f"\n{'='*55}")
print(" Architecture Summary")
print(f"{'='*55}")
print(f" Framework : PyTorch {torch.__version__}")
print(f" Input : 16 shape features")
print(f" Hidden : 8 neurons (ReLU)")
print(f" Output : 7 neurons (Softmax via CrossEntropyLoss)")
print(f" Params : {sum(p.numel() for p in model.parameters())}")
print(f" Optimizer : SGD | lr={LEARNING_RATE}")
print(f" Epochs : {EPOCHS}")
print(f" Final Test Accuracy : {final_acc*100:.2f}%")
print(f"{'='*55}")
[cell 25 markdown]
## Define MLP & Train with Full Weight History Recording
We record **every weight, gradient, and delta** at every epoch — giving us the full picture of how backpropagation updates each individual connection.
[cell 26 code]
class MLP(nn.Module):
def __init__(self, n_in=16, n_hidden=8, n_out=7):
super().__init__()
self.hidden = nn.Linear(n_in, n_hidden)
self.output = nn.Linear(n_hidden, n_out)
self.relu = nn.ReLU()
nn.init.zeros_(self.hidden.bias)
nn.init.zeros_(self.output.bias)
def forward(self, x):
return self.output(self.relu(self.hidden(x)))
EPOCHS = 50
LR = 0.05
model = MLP().to(DEVICE)
criterion = nn.CrossEntropyLoss()
optimizer = optim.SGD(model.parameters(), lr=LR)
# ── Storage: shape described at each key ────────────────────────────
# W1_hist[epoch, neuron, feature] — (50, 8, 16)
# W2_hist[epoch, class, neuron ] — (50, 7, 8)
W1_hist = np.zeros((EPOCHS, 8, 16))
W2_hist = np.zeros((EPOCHS, 7, 8))
b1_hist = np.zeros((EPOCHS, 8))
b2_hist = np.zeros((EPOCHS, 7))
dW1_hist = np.zeros((EPOCHS, 8, 16)) # gradients
dW2_hist = np.zeros((EPOCHS, 7, 8))
delta_W1 = np.zeros((EPOCHS, 8, 16)) # actual weight change Δw = -lr * grad
delta_W2 = np.zeros((EPOCHS, 7, 8))
loss_hist = []
acc_hist_train = []
acc_hist_test = []
print('Training and recording every weight at every epoch...')
print(f'{"Epoch":>6} | {"Train Loss":>11} | {"Train Acc":>10} | {"Test Acc":>9}')
print('-' * 46)
for epoch in range(EPOCHS):
# Save weights BEFORE update
W1_before = model.hidden.weight.data.cpu().numpy().copy()
# ── Training ────────────────────────────────────────────────────
model.train()
ep_loss, correct, total = 0.0, 0, 0
for Xb, yb in train_loader:
optimizer.zero_grad()
loss = criterion(model(Xb), yb)
loss.backward()
optimizer.step()
ep_loss += loss.item() * Xb.size(0)
correct += (model(Xb).argmax(1) == yb).sum().item()
total += Xb.size(0)
# Capture gradients via one dedicated backward on full train set
model.train()
optimizer.zero_grad()
criterion(model(X_train), y_train).backward()
# ── Record everything ────────────────────────────────────────────
W1_hist[epoch] = model.hidden.weight.data.cpu().numpy()
W2_hist[epoch] = model.output.weight.data.cpu().numpy()
b1_hist[epoch] = model.hidden.bias.data.cpu().numpy()
b2_hist[epoch] = model.output.bias.data.cpu().numpy()
dW1_hist[epoch] = model.hidden.weight.grad.cpu().numpy()
dW2_hist[epoch] = model.output.weight.grad.cpu().numpy()
delta_W1[epoch] = W1_hist[epoch] - W1_before # actual change this epoch
if epoch > 0:
delta_W2[epoch] = W2_hist[epoch] - W2_hist[epoch - 1]
# Test accuracy
model.eval()
with torch.no_grad():
test_acc = (model(X_test).argmax(1) == y_test).float().mean().item()
loss_hist.append(ep_loss / total)
acc_hist_train.append(correct / total)
acc_hist_test.append(test_acc)
if (epoch + 1) % 10 == 0 or epoch == 0:
print(f'{epoch+1:>6} | {ep_loss/total:>11.4f} | {correct/total*100:>9.2f}% | {test_acc*100:>8.2f}%')
print(f'\nRecorded W1_hist shape: {W1_hist.shape} (epochs × neurons × features)')
print(f' Recorded W2_hist shape: {W2_hist.shape} (epochs × classes × neurons)')
[cell 27 markdown]
## Weight Trajectory — Every Connection in Hidden Layer
Each line = one weight connecting an **input feature → hidden neuron**.
One panel per hidden neuron (H1–H8). Shows how backprop gradually adjusts each weight.
[cell 28 code]
epochs_x = np.arange(1, EPOCHS + 1)
FEAT_COLORS = ['#e6194b', '#3cb44b', '#0000ff', '#f58231', '#911eb4', '#00ced1', '#aa6e28', '#000075', '#e63900', '#008000', '#9400d3', '#ff8c00', '#dc143c', '#00868b', '#8b0000', '#006400']
feat_colors = FEAT_COLORS
fig, axes = plt.subplots(2, 4, figsize=(20, 9))
for h, ax in enumerate(axes.flat):
for f in range(16):
ax.plot(epochs_x, W1_hist[:, h, f],
color=feat_colors[f], lw=1.2, alpha=0.85,
label=feature_names[f])
ax.axhline(0, color='#555577', lw=0.7, ls='--')
ax.set_title(f'Hidden Neuron H{h+1}', fontsize=11, fontweight='bold', color='#222222')
ax.set_xlabel('Epoch', fontsize=9)
ax.set_ylabel('Weight value', fontsize=9)
ax.grid(True)
# Mark epoch-10 checkpoints
for ck in range(10, EPOCHS + 1, 10):
ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)
# Legend below the figure
handles = [plt.Line2D([0],[0], color=feat_colors[f], lw=2) for f in range(16)]
fig.legend(handles, feature_names,
loc='lower center', ncol=8, fontsize=8,
bbox_to_anchor=(0.5, -0.04), framealpha=0.3)
fig.suptitle('W1: Individual Weight Trajectories per Hidden Neuron\n'
'(Each line = one input feature → this neuron)',
fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w1_weight_trajectories.png', dpi=130, bbox_inches='tight')
plt.show()
[cell 29 markdown]
## Gradient Magnitude per Neuron — How Hard Backprop is Pushing Each Weight
[cell 30 code]
fig, axes = plt.subplots(2, 4, figsize=(20, 9))
for h, ax in enumerate(axes.flat):
for f in range(16):
ax.plot(epochs_x, np.abs(dW1_hist[:, h, f]),
color=feat_colors[f], lw=1.2, alpha=0.85)
ax.set_title(f'|∂L/∂W1| — Neuron H{h+1}', fontsize=11, fontweight='bold', color='#222222')
ax.set_xlabel('Epoch', fontsize=9)
ax.set_ylabel('|Gradient|', fontsize=9)
ax.grid(True)
for ck in range(10, EPOCHS + 1, 10):
ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)
handles = [plt.Line2D([0],[0], color=feat_colors[f], lw=2) for f in range(16)]
fig.legend(handles, feature_names,
loc='lower center', ncol=8, fontsize=8,
bbox_to_anchor=(0.5, -0.04), framealpha=0.3)
fig.suptitle('W1: Gradient Magnitudes |∂L/∂w| per Hidden Neuron\n'
'(Decreasing gradients = converging)',
fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w1_gradient_magnitudes.png', dpi=130, bbox_inches='tight')
plt.show()
[cell 31 markdown]
## Actual Weight Delta Δw per Neuron (What Backprop Actually Applies)
`Δw = w_after − w_before = −lr × ∂L/∂w`
This is the **signed update** — positive means the weight increased, negative means it decreased.
[cell 32 code]
fig, axes = plt.subplots(2, 4, figsize=(20, 9))
for h, ax in enumerate(axes.flat):
for f in range(16):
ax.plot(epochs_x, delta_W1[:, h, f],
color=feat_colors[f], lw=1.0, alpha=0.85)
ax.axhline(0, color='#ffffff', lw=0.8, ls='--', alpha=0.5)
ax.set_title(f'Δw = −lr·∂L/∂w — Neuron H{h+1}', fontsize=10,
fontweight='bold', color='#222222')
ax.set_xlabel('Epoch', fontsize=9)
ax.set_ylabel('Weight change Δw', fontsize=9)
ax.grid(True)
for ck in range(10, EPOCHS + 1, 10):
ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)
handles = [plt.Line2D([0],[0], color=feat_colors[f], lw=2) for f in range(16)]
fig.legend(handles, feature_names,
loc='lower center', ncol=8, fontsize=8,
bbox_to_anchor=(0.5, -0.04), framealpha=0.3)
fig.suptitle('W1: Weight Updates Δw Applied Each Epoch per Hidden Neuron\n'
'(+ve = weight increased, −ve = weight decreased)',
fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w1_weight_deltas.png', dpi=130, bbox_inches='tight')
plt.show()
[cell 33 markdown]
## Output Layer W2 — Weight Trajectory per Class Neuron
Each panel = one output class neuron. Each line = its connection to one hidden neuron.
[cell 34 code]
HIDDEN_COLORS = ['#e6194b', '#0000ff', '#3cb44b', '#ff8c00', '#9400d3', '#00ced1', '#dc143c', '#228b22']
hidden_colors = HIDDEN_COLORS
fig, axes = plt.subplots(2, 4, figsize=(20, 9))
# 7 class panels + 1 combined summary
for c in range(7):
ax = axes.flat[c]
for h in range(8):
ax.plot(epochs_x, W2_hist[:, c, h],
color=hidden_colors[h], lw=1.8, alpha=0.9,
label=f'H{h+1}')
ax.axhline(0, color='#555577', lw=0.7, ls='--')
ax.set_title(f'Output: {class_names[c]}', fontsize=11,
fontweight='bold', color='#222222')
ax.set_xlabel('Epoch', fontsize=9)
ax.set_ylabel('Weight value', fontsize=9)
ax.grid(True)
for ck in range(10, EPOCHS + 1, 10):
ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)
ax.legend(fontsize=7, ncol=4, loc='upper right')
# Panel 8: mean absolute W2 per class
ax8 = axes.flat[7]
CLASS_COLORS = ['#e6194b', '#0000ff', '#3cb44b', '#ff8c00', '#9400d3', '#00ced1', '#dc143c']
class_palette = CLASS_COLORS
for c in range(7):
mean_abs = np.abs(W2_hist[:, c, :]).mean(axis=1)
ax8.plot(epochs_x, mean_abs, color=class_palette[c],
lw=2, label=class_names[c])
ax8.set_title('Mean |W2| per Output Class', fontsize=11,
fontweight='bold', color='#222222')
ax8.set_xlabel('Epoch', fontsize=9)
ax8.set_ylabel('Mean |weight|', fontsize=9)
ax8.legend(fontsize=8)
ax8.grid(True)
fig.suptitle('W2: Output Layer Weight Trajectories per Class\n'
'(Each line = hidden neuron → this class output)',
fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w2_weight_trajectories.png', dpi=130, bbox_inches='tight')
plt.show()
[cell 35 markdown]
## W2 Gradient Magnitudes per Output Neuron
[cell 36 code]
fig, axes = plt.subplots(2, 4, figsize=(20, 9))
for c in range(7):
ax = axes.flat[c]
for h in range(8):
ax.plot(epochs_x, np.abs(dW2_hist[:, c, h]),
color=hidden_colors[h], lw=1.8, alpha=0.9, label=f'H{h+1}')
ax.set_title(f'|∂L/∂W2| — {class_names[c]}', fontsize=11,
fontweight='bold', color='#222222')
ax.set_xlabel('Epoch', fontsize=9)
ax.set_ylabel('|Gradient|', fontsize=9)
ax.legend(fontsize=7, ncol=4, loc='upper right')
ax.grid(True)
for ck in range(10, EPOCHS + 1, 10):
ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)
# Panel 8: all gradient norms overlaid
ax8 = axes.flat[7]
ax8.semilogy(epochs_x, np.abs(dW1_hist).mean(axis=(1,2)),
color='#a8dadc', lw=2.5, label='|∂L/∂W1| (hidden)')
ax8.semilogy(epochs_x, np.abs(dW2_hist).mean(axis=(1,2)),
color='#e63946', lw=2.5, ls='--', label='|∂L/∂W2| (output)')
ax8.set_title('Overall Gradient Norms (log)', fontsize=11,
fontweight='bold', color='#222222')
ax8.set_xlabel('Epoch', fontsize=9)
ax8.set_ylabel('Mean |gradient| (log)', fontsize=9)
ax8.legend(fontsize=9)
ax8.grid(True)
fig.suptitle('W2: Gradient Magnitudes |∂L/∂w| per Output Class\n'
'(Each line = gradient for the hidden→class connection)',
fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w2_gradient_magnitudes.png', dpi=130, bbox_inches='tight')
plt.show()
[cell 37 markdown]
**We record both W1 and W2 after every epoch.**
[cell 38 code]
torch.manual_seed(42)
np.random.seed(42)
DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
# ── Load data ────────────────────────────────────────────────────────
df = pd.read_excel('Dry_Bean_Dataset.xlsx')
FEATURE_NAMES = df.columns[:-1].tolist()
X_raw = df.drop(columns=['Class']).values.astype(np.float32)
y_raw = df['Class'].values
le = LabelEncoder()
y_int = le.fit_transform(y_raw).astype(np.int64)
CLASS_NAMES = le.classes_
X_scaled = StandardScaler().fit_transform(X_raw).astype(np.float32)
X_tr, X_te, y_tr, y_te = train_test_split(
X_scaled, y_int, test_size=0.2, random_state=42, stratify=y_int
)
X_train = torch.tensor(X_tr).to(DEVICE)
y_train = torch.tensor(y_tr).to(DEVICE)
train_loader = DataLoader(TensorDataset(X_train, y_train), batch_size=256, shuffle=True)
# ── Model ────────────────────────────────────────────────────────────
class MLP(nn.Module):
def __init__(self):
super().__init__()
self.hidden = nn.Linear(16, 8)
self.output = nn.Linear(8, 7)
self.relu = nn.ReLU()
nn.init.zeros_(self.hidden.bias)
nn.init.zeros_(self.output.bias)
def forward(self, x):
return self.output(self.relu(self.hidden(x)))
model = MLP().to(DEVICE)
criterion = nn.CrossEntropyLoss()
optimizer = optim.SGD(model.parameters(), lr=0.05)
# ── Train 50 epochs — record W1 and W2 after every epoch ─────────────
EPOCHS = 50
# W1: PyTorch shape (8 neurons, 16 features) → W1_hist[epoch, neuron, feature]
# W2: PyTorch shape (7 classes, 8 neurons ) → W2_hist[epoch, class, neuron ]
W1_hist = np.zeros((EPOCHS, 8, 16))
W2_hist = np.zeros((EPOCHS, 7, 8))
loss_hist = []
for epoch in range(EPOCHS):
model.train()
ep_loss = 0.0
for Xb, yb in train_loader:
optimizer.zero_grad()
loss = criterion(model(Xb), yb)
loss.backward()
optimizer.step()
ep_loss += loss.item()
W1_hist[epoch] = model.hidden.weight.data.cpu().numpy() # (8, 16)
W2_hist[epoch] = model.output.weight.data.cpu().numpy() # (7, 8)
loss_hist.append(ep_loss / len(train_loader))
print('Training complete!')
print(f'W1_hist : {W1_hist.shape} (epochs x neurons x features)')
print(f'W2_hist : {W2_hist.shape} (epochs x classes x neurons)')
print(f'Final loss : {loss_hist[-1]:.4f}')
[cell 39 markdown]
## Choose Features & Classes to Inspect
[cell 40 code]
# ── W1: 3 input features to inspect ────────────────────────────────
SELECTED_FEATURES = ['Area', 'Perimeter', 'Eccentricity']
FEAT_IDX = [FEATURE_NAMES.index(f) for f in SELECTED_FEATURES]
# ── W2: 3 output classes to inspect ─────────────────────────────────
SELECTED_CLASSES = ['BOMBAY', 'SEKER', 'DERMASON']
CLASS_IDX = [list(CLASS_NAMES).index(c) for c in SELECTED_CLASSES]
# ── Same high-contrast colour per hidden neuron in every plot ────────
NEURON_COLORS = [
'#e6194b', # H1 vivid red
'#0000ff', # H2 pure blue
'#3cb44b', # H3 vivid green
'#ff8c00', # H4 dark orange
'#9400d3', # H5 vivid purple
'#00ced1', # H6 dark turquoise
'#dc143c', # H7 crimson
'#228b22', # H8 forest green
]
epochs_x = np.arange(1, EPOCHS + 1)
print('W1 features :', SELECTED_FEATURES, ' indices:', FEAT_IDX)
print('W2 classes :', SELECTED_CLASSES, ' indices:', CLASS_IDX)
[cell 41 markdown]
## Helper — Reusable Plot Function
[cell 42 code]
def plot_weight_grid(series_8, neuron_colors, epochs_x, ylabel_fn, suptitle, savename):
"""
series_8 : list of 8 arrays shape (EPOCHS,) — one per hidden neuron
ylabel_fn : function of h (int) -> y-axis label string
"""
fig, axes = plt.subplots(2, 4, figsize=(18, 7), sharey=False)
fig.patch.set_facecolor('white')
for h, ax in enumerate(axes.flat):
w = series_8[h]
color = neuron_colors[h]
ax.fill_between(epochs_x, w, alpha=0.15, color=color)
ax.plot(epochs_x, w, color=color, lw=2.2, zorder=3)
# Dot + annotated value at every-10-epoch checkpoint
for ck in range(0, len(epochs_x), 10):
ax.scatter(ck + 1, w[ck], color=color, s=60,
zorder=5, edgecolors='black', linewidths=0.8)
ax.annotate(f'{w[ck]:.3f}',
xy=(ck + 1, w[ck]),
xytext=(0, 9), textcoords='offset points',
ha='center', fontsize=7.5, color='#111111',
bbox=dict(boxstyle='round,pad=0.15', fc='white',
ec=color, lw=0.8, alpha=0.85))
# Dashed init line, dotted final line
ax.axhline(w[0], color='gray', lw=0.9, ls='--', alpha=0.6,
label=f'Init {w[0]:+.3f}')
ax.axhline(w[-1], color='#111111', lw=0.9, ls=':', alpha=0.85,
label=f'Final {w[-1]:+.3f}')
ax.set_title(f'H{h+1} Dw = {w[-1]-w[0]:+.3f}',
fontsize=11, fontweight='bold', color=color, pad=5)
ax.set_xlabel('Epoch', fontsize=9)
ax.set_ylabel(ylabel_fn(h), fontsize=8)
ax.legend(fontsize=7, loc='best')
ax.grid(True, alpha=0.35)
ax.set_xlim(1, len(epochs_x))
ax.xaxis.set_major_locator(ticker.MultipleLocator(10))
fig.suptitle(suptitle, fontsize=13, fontweight='bold', color='#111111')
plt.tight_layout()
plt.savefig(savename, dpi=130, bbox_inches='tight')
plt.show()
print('plot_weight_grid() ready')
[cell 43 markdown]
---
## W1 — Input → Hidden Layer
`W1[neuron, feature]` — weight connecting an input feature to a hidden neuron.
**Backprop update rule:** `W1 -= lr * dL/dW1` where `dL/dW1 = delta1.T @ X`
One cell per feature. 8 panels per cell (one per hidden neuron).
[cell 44 markdown]
### W1 — Feature: `Area`
[cell 45 code]
import matplotlib.ticker as ticker
fname = 'Area'
feat_idx = FEAT_IDX[0]
plot_weight_grid(
series_8 = [W1_hist[:, h, feat_idx] for h in range(8)],
neuron_colors= NEURON_COLORS,
epochs_x = epochs_x,
ylabel_fn = lambda h: f'w(Area -> H{h+1})',
suptitle = (
'W1 | Feature: "Area"\n'
'Each panel = one hidden neuron y = weight value after every epoch'
),
savename = 'W1_Area.png'
)
[cell 46 markdown]
### W1 — Feature: `Perimeter`
[cell 47 code]
fname = 'Perimeter'
feat_idx = FEAT_IDX[1]
plot_weight_grid(
series_8 = [W1_hist[:, h, feat_idx] for h in range(8)],
neuron_colors= NEURON_COLORS,
epochs_x = epochs_x,
ylabel_fn = lambda h: f'w(Perimeter -> H{h+1})',
suptitle = (
'W1 | Feature: "Perimeter"\n'
'Each panel = one hidden neuron y = weight value after every epoch'
),
savename = 'W1_Perimeter.png'
)
[cell 48 markdown]
### W1 — Feature: `Eccentricity`
[cell 49 code]
fname = 'Eccentricity'
feat_idx = FEAT_IDX[2]
plot_weight_grid(
series_8 = [W1_hist[:, h, feat_idx] for h in range(8)],
neuron_colors= NEURON_COLORS,
epochs_x = epochs_x,
ylabel_fn = lambda h: f'w(Eccentricity -> H{h+1})',
suptitle = (
'W1 | Feature: "Eccentricity"\n'
'Each panel = one hidden neuron y = weight value after every epoch'
),
savename = 'W1_Eccentricity.png'
)
[cell 50 markdown]
---
## W2 — Hidden → Output Layer
`W2[class, neuron]` — weight connecting a hidden neuron to an output class neuron.
**Backprop update rule:** `W2 -= lr * dL/dW2` where `dL/dW2 = delta2.T @ A1`
One cell per class. 8 panels per cell (one per hidden neuron).
Each panel shows how strongly that hidden neuron 'votes' for this class over training.
[cell 51 markdown]
### W2 — Output Class: `BOMBAY`
[cell 52 code]
cname = 'BOMBAY'
class_idx = CLASS_IDX[0]
plot_weight_grid(
series_8 = [W2_hist[:, class_idx, h] for h in range(8)],
neuron_colors= NEURON_COLORS,
epochs_x = epochs_x,
ylabel_fn = lambda h: f'w(H{h+1} -> BOMBAY)',
suptitle = (
'W2 | Output Class: "BOMBAY"\n'
'Each panel = one hidden neuron y = weight value after every epoch'
),
savename = 'W2_BOMBAY.png'
)
[cell 53 markdown]
### W2 — Output Class: `SEKER`
[cell 54 code]
cname = 'SEKER'
class_idx = CLASS_IDX[1]
plot_weight_grid(
series_8 = [W2_hist[:, class_idx, h] for h in range(8)],
neuron_colors= NEURON_COLORS,
epochs_x = epochs_x,
ylabel_fn = lambda h: f'w(H{h+1} -> SEKER)',
suptitle = (
'W2 | Output Class: "SEKER"\n'
'Each panel = one hidden neuron y = weight value after every epoch'
),
savename = 'W2_SEKER.png'
)
[cell 55 markdown]
### W2 — Output Class: `DERMASON`
[cell 56 code]
cname = 'DERMASON'
class_idx = CLASS_IDX[2]
plot_weight_grid(
series_8 = [W2_hist[:, class_idx, h] for h in range(8)],
neuron_colors= NEURON_COLORS,
epochs_x = epochs_x,
ylabel_fn = lambda h: f'w(H{h+1} -> DERMASON)',
suptitle = (
'W2 | Output Class: "DERMASON"\n'
'Each panel = one hidden neuron y = weight value after every epoch'
),
savename = 'W2_DERMASON.png'
)
[cell 57 markdown]
---
## One Neuron, Both Layers
Pick any hidden neuron and see:
- **Top row (W1):** its weights from the 3 selected features
- **Bottom row (W2):** its weights to the 3 selected classes
Change `NEURON_TO_INSPECT` to 0–7 to explore different neurons.
[cell 58 code]
NEURON_TO_INSPECT = 0 # change to 0-7
FEAT_COLORS_3 = ['#e6194b', '#0000ff', '#3cb44b']
CLASS_COLORS_3 = ['#ff8c00', '#9400d3', '#00ced1']
fig, axes = plt.subplots(2, 3, figsize=(18, 8))
fig.patch.set_facecolor('white')
# ── Top row: W1 — how this neuron receives from each feature ─────────
for col, (fname, fidx, fc) in enumerate(
zip(SELECTED_FEATURES, FEAT_IDX, FEAT_COLORS_3)):
ax = axes[0, col]
w = W1_hist[:, NEURON_TO_INSPECT, fidx]
ax.fill_between(epochs_x, w, alpha=0.18, color=fc)
ax.plot(epochs_x, w, color=fc, lw=2.4)
ax.scatter(epochs_x, w, color=fc, s=20,
edgecolors='black', linewidths=0.4, zorder=4)
for ck in range(0, EPOCHS, 10):
ax.annotate(f'{w[ck]:.3f}', xy=(ck+1, w[ck]),
xytext=(0, 10), textcoords='offset points',
ha='center', fontsize=8.5, fontweight='bold', color='#111111',
bbox=dict(boxstyle='round,pad=0.2', fc='white', ec=fc, lw=1.1, alpha=0.9))
ax.axhline(w[0], color='gray', lw=0.8, ls='--', alpha=0.6)
ax.axhline(w[-1], color='#111111', lw=0.8, ls=':', alpha=0.8)
ax.set_title(
f'W1: {fname} -> H{NEURON_TO_INSPECT+1}\n'
f'Dw = {w[-1]-w[0]:+.4f}',
fontsize=11, fontweight='bold', color=fc)
ax.set_xlabel('Epoch'); ax.grid(True, alpha=0.35)
ax.set_ylabel(f'w({fname} -> H{NEURON_TO_INSPECT+1})', fontsize=9)
ax.xaxis.set_major_locator(ticker.MultipleLocator(10))
# ── Bottom row: W2 — how this neuron votes for each class ────────────
for col, (cname, cidx, cc) in enumerate(
zip(SELECTED_CLASSES, CLASS_IDX, CLASS_COLORS_3)):
ax = axes[1, col]
w = W2_hist[:, cidx, NEURON_TO_INSPECT]
ax.fill_between(epochs_x, w, alpha=0.18, color=cc)
ax.plot(epochs_x, w, color=cc, lw=2.4)
ax.scatter(epochs_x, w, color=cc, s=20,
edgecolors='black', linewidths=0.4, zorder=4)
for ck in range(0, EPOCHS, 10):
ax.annotate(f'{w[ck]:.3f}', xy=(ck+1, w[ck]),
xytext=(0, 10), textcoords='offset points',
ha='center', fontsize=8.5, fontweight='bold', color='#111111',
bbox=dict(boxstyle='round,pad=0.2', fc='white', ec=cc, lw=1.1, alpha=0.9))
ax.axhline(w[0], color='gray', lw=0.8, ls='--', alpha=0.6)
ax.axhline(w[-1], color='#111111', lw=0.8, ls=':', alpha=0.8)
ax.set_title(
f'W2: H{NEURON_TO_INSPECT+1} -> {cname}\n'
f'Dw = {w[-1]-w[0]:+.4f}',
fontsize=11, fontweight='bold', color=cc)
ax.set_xlabel('Epoch'); ax.grid(True, alpha=0.35)
ax.set_ylabel(f'w(H{NEURON_TO_INSPECT+1} -> {cname})', fontsize=9)
ax.xaxis.set_major_locator(ticker.MultipleLocator(10))
fig.suptitle(
f'Hidden Neuron H{NEURON_TO_INSPECT+1} — Both Layers\n'
f'Top row: W1 (what it receives from features) | '
f'Bottom row: W2 (what it sends to output classes)',
fontsize=13, fontweight='bold', color='#111111'
)
plt.tight_layout()
plt.savefig(f'both_layers_H{NEURON_TO_INSPECT+1}.png', dpi=130, bbox_inches='tight')
plt.show()