# MLP backprop DryBean

course: Module 3 — Deep Learning & NLP
module: Module-3-Deep-Learning-NLP
type: notebook
source_url: https://personal-learn.armco.dev/files/Module-3-Deep-Learning-NLP/General/Lab_Materials-14-03-2026/MLP_backprop_DryBean.ipynb

---
[cell 1 markdown]
#  MLP with PyTorch - Backpropagation on Dry Bean Dataset

This notebook trains a **Multi-Layer Perceptron (MLP)** in **PyTorch** to classify 7 types of dry beans from shape measurements.

| | |
|---|---|
| **Dataset** | Dry Bean Dataset (UCI) |
| **Samples** | 13,611 |
| **Features** | 16 geometric shape features |
| **Task** | Multi-class Classification — 7 bean types |
| **Architecture** | Input(16) → Hidden(8, ReLU) → Output(7, Softmax) |
| **Framework** | **PyTorch** |

**Learning goals:**
- Understand how PyTorch builds a **computation graph** during the forward pass
- See how `loss.backward()` runs **backpropagation automatically** via the chain rule
- Watch **weights and gradients evolve** every 10 epochs
- Compare manual backprop math to PyTorch autograd output

---

[cell 2 markdown]
##  Step 1: Imports

[cell 3 code]
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import matplotlib.gridspec as gridspec

import torch
import torch.nn as nn
import torch.optim as optim
import torch.nn.functional as F
from torch.utils.data import DataLoader, TensorDataset

from sklearn.preprocessing import LabelEncoder, StandardScaler
from sklearn.model_selection import train_test_split

# Reproducibility
torch.manual_seed(42)
np.random.seed(42)

DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')
print(f"PyTorch version : {torch.__version__}")
print(f"   Device          : {DEVICE}")

[cell 4 markdown]
##  Step 2: Load and Explore the Dataset

[cell 5 code]
df = pd.read_excel("Dry_Bean_Dataset.xlsx")

print(f"Dataset Shape : {df.shape}")
print(f"Features (16) : {df.columns[:-1].tolist()}")
print(f"\nClass Distribution:")
vc = df['Class'].value_counts()
for label, count in vc.items():
    print(f"  {label:>10} : {count:>5} samples ({count/len(df)*100:.1f}%)")

df.head()

[cell 6 code]
# Visualise feature distributions per class
features_to_plot = ['Area', 'Perimeter', 'MajorAxisLength', 'MinorAxisLength',
                    'Eccentricity', 'roundness', 'Compactness', 'Solidity']
class_names_all  = df['Class'].unique()
palette = plt.cm.Set2(np.linspace(0, 1, len(class_names_all)))

fig, axes = plt.subplots(4, 2, figsize=(14, 12))
for ax, feat in zip(axes.flat, features_to_plot):
    for cls, col in zip(class_names_all, palette):
        vals = df[df['Class'] == cls][feat].values
        ax.hist(vals, bins=30, alpha=0.55, color=col, label=cls, density=True)
    ax.set_title(feat, fontsize=10, fontweight='bold')
    ax.set_xlabel('Value', fontsize=8)
    ax.tick_params(labelsize=7)
    ax.grid(True, alpha=0.2)

handles = [plt.Rectangle((0,0),1,1, fc=c) for c in palette]
fig.legend(handles, class_names_all, loc='lower center', ncol=7,
           fontsize=10, bbox_to_anchor=(0.5, -0.04))
plt.suptitle('Feature Distributions by Bean Class',
             fontsize=14, fontweight='bold')
plt.tight_layout()
plt.savefig('feature_distributions.png', dpi=500, bbox_inches='tight')
plt.show()

[cell 7 markdown]
##  Step 3: Preprocess and Build PyTorch DataLoaders

[cell 8 code]
# ---------- Features & Labels ----------
X_raw = df.drop(columns=['Class']).values.astype(np.float32)  # (13611, 16)
y_raw = df['Class'].values

# ---------- Encode string labels → integers 0..6 ----------
le = LabelEncoder()
y_int = le.fit_transform(y_raw).astype(np.int64)              # required by CrossEntropyLoss
class_names = le.classes_
n_classes   = len(class_names)
print(f"Classes ({n_classes}): {class_names}")

# ---------- Normalize features ----------
scaler   = StandardScaler()
X_scaled = scaler.fit_transform(X_raw).astype(np.float32)

# ---------- Train / Test split ----------
X_train_np, X_test_np, y_train_np, y_test_np = train_test_split(
    X_scaled, y_int, test_size=0.2, random_state=42, stratify=y_int
)

# ---------- Convert to PyTorch tensors ----------
# NOTE: CrossEntropyLoss in PyTorch expects raw logits (no softmax)
# and integer class labels — NOT one-hot vectors.
X_train = torch.tensor(X_train_np).to(DEVICE)          # (N, 16)  float32
X_test  = torch.tensor(X_test_np).to(DEVICE)
y_train = torch.tensor(y_train_np).to(DEVICE)          # (N,)     int64
y_test  = torch.tensor(y_test_np).to(DEVICE)

# ---------- DataLoaders ----------
train_loader = DataLoader(TensorDataset(X_train, y_train),
                          batch_size=256, shuffle=True)
test_loader  = DataLoader(TensorDataset(X_test,  y_test),
                          batch_size=256, shuffle=False)

print(f"\nX_train : {X_train.shape}  |  X_test : {X_test.shape}")
print(f"y_train : {y_train.shape}  |  y_test : {y_test.shape}")
print(f"\nTrain batches : {len(train_loader)}  (batch_size=256)")
print(f"Test  batches : {len(test_loader)}")

[cell 9 markdown]
##  Step 4: Define the MLP in PyTorch

```
Input Layer       Hidden Layer          Output Layer
(16 neurons)  →  (8 neurons, ReLU)  →  (7 neurons, Softmax)
```

**Key PyTorch design choice for multiclass:**  
`nn.CrossEntropyLoss` = `LogSoftmax` + `NLLLoss` **combined**.  
So the model outputs **raw logits** (no softmax in `forward`).  
Softmax is only applied manually when we need probabilities (e.g., for display).

[cell 10 code]
class MLP(nn.Module):
    """
    MLP: Input(16) → Hidden(8, ReLU) → Output(7, raw logits)

    PyTorch's nn.Linear(in_features, out_features) creates:
      W: (out, in)   b: (out,)
    and computes:  z = x @ W.T + b

    We return raw logits from the output layer.
    CrossEntropyLoss applies log-softmax + NLL internally.
    """
    def __init__(self, n_input=16, n_hidden=8, n_output=7):
        super(MLP, self).__init__()

        self.hidden = nn.Linear(n_input,  n_hidden)  # W1:(8,16), b1:(8,)
        self.output = nn.Linear(n_hidden, n_output)  # W2:(7,8),  b2:(7,)
        self.relu   = nn.ReLU()

    def forward(self, x):
        """
        Every operation here is recorded by PyTorch's autograd engine
        into a dynamic computation graph.
        Calling loss.backward() later traverses this graph in reverse.
        """
        z1     = self.hidden(x)    # z1 = x @ W1.T + b1
        a1     = self.relu(z1)     # a1 = ReLU(z1)
        logits = self.output(a1)   # logits = a1 @ W2.T + b2  (raw scores)
        return logits              # CrossEntropyLoss handles softmax internally

    def predict_proba(self, x):
        """Returns softmax probabilities (for evaluation/display only)."""
        return F.softmax(self.forward(x), dim=1)


# Instantiate
model = MLP(n_input=16, n_hidden=8, n_output=n_classes).to(DEVICE)

print("=" * 55)
print("   MLP Architecture")
print("=" * 55)
print(model)
print("=" * 55)

print(f"\nParameter breakdown:")
total = 0
for name, param in model.named_parameters():
    print(f"  {name:25s}: shape={str(param.shape):15s}  count={param.numel()}")
    total += param.numel()
print(f"\n  Total trainable parameters: {total}")

[cell 11 markdown]
## Step 5: How PyTorch Autograd Works

Before training, let's see PyTorch building and traversing the computation graph on a single sample.

[cell 12 code]
print("── PyTorch Autograd Demo — One Sample, One Forward+Backward ──\n")

x_demo = X_train[:1]          # shape (1, 16)
y_demo = y_train[:1]          # shape (1,) — integer class label

model.zero_grad()

# ── Forward: PyTorch builds a computation graph ───────────────────────
logits_demo = model(x_demo)                         # (1, 7)
probs_demo  = F.softmax(logits_demo, dim=1)         # (1, 7)
loss_demo   = nn.CrossEntropyLoss()(logits_demo, y_demo)

print(f"Input shape     : {x_demo.shape}")
print(f"True class      : {class_names[y_demo.item()]}  (index {y_demo.item()})")
print(f"\nOutput logits   : {logits_demo.detach().numpy()[0].round(4)}")
print(f"After Softmax   :")
for cls, p in zip(class_names, probs_demo.detach().numpy()[0]):
    marker = ' ← predicted' if p == probs_demo.max().item() else ''
    marker += ' ← TRUE' if cls == class_names[y_demo.item()] else ''
    print(f"  {cls:>10}: {p:.4f}{marker}")
print(f"\nCross-Entropy Loss : {loss_demo.item():.4f}")

# ── Backward: one call — PyTorch applies the full chain rule ──────────
loss_demo.backward()

print("\n── Gradients computed by loss.backward() ──")
for name, param in model.named_parameters():
    if param.grad is not None:
        g = param.grad
        print(f"  ∂L/∂{name:20s}: shape={str(g.shape):12s} "
              f"mean={g.mean().item():+.6f}  norm={g.norm().item():.6f}")

print("\nPyTorch computed ALL gradients automatically — this is backpropagation!")
print("   It applied the chain rule layer by layer, from output back to input.")

[cell 13 markdown]
##  Step 6: The PyTorch Training Loop

Every training step follows this exact 5-line recipe:

```python
optimizer.zero_grad()              # 1. Clear gradients from previous step
logits = model(X_batch)            # 2. Forward pass  → builds computation graph
loss   = criterion(logits, y_batch)# 3. Compute loss
loss.backward()                    # 4. BACKPROPAGATION → fills .grad on all params
optimizer.step()                   # 5. Gradient descent: w ← w − lr × ∂L/∂w
```

[cell 14 code]
def train_epoch(model, loader, criterion, optimizer):
    """Run one full training epoch. Returns avg loss and accuracy."""
    model.train()
    total_loss, correct, total = 0.0, 0, 0

    for X_batch, y_batch in loader:
        # 1. Clear stale gradients
        optimizer.zero_grad()

        # 2. Forward pass
        logits = model(X_batch)

        # 3. Compute Cross-Entropy loss
        #    (applies log-softmax + NLL internally)
        loss = criterion(logits, y_batch)

        # 4. Backpropagation — chain rule through all layers
        loss.backward()

        # 5. Gradient descent step
        optimizer.step()

        total_loss += loss.item() * X_batch.size(0)
        preds   = logits.argmax(dim=1)
        correct += (preds == y_batch).sum().item()
        total   += X_batch.size(0)

    return total_loss / total, correct / total


@torch.no_grad()
def evaluate(model, loader, criterion):
    """Evaluate on a DataLoader. Returns avg loss and accuracy."""
    model.eval()
    total_loss, correct, total = 0.0, 0, 0

    for X_batch, y_batch in loader:
        logits = model(X_batch)
        loss   = criterion(logits, y_batch)
        total_loss += loss.item() * X_batch.size(0)
        correct += (logits.argmax(dim=1) == y_batch).sum().item()
        total   += X_batch.size(0)

    return total_loss / total, correct / total


print("Training and evaluation functions defined!")

[cell 15 markdown]
##  Step 7: Train for 50 Epochs - Weight Updates Every 10 Epochs

[cell 16 code]
# ── Hyperparameters ──────────────────────────────────────────────────
LEARNING_RATE = 0.01
EPOCHS        = 50
PRINT_EVERY   = 10

# ── Fresh model ──────────────────────────────────────────────────────
model     = MLP(n_input=16, n_hidden=8, n_output=n_classes).to(DEVICE)
criterion = nn.CrossEntropyLoss()   # softmax + NLL combined
optimizer = optim.SGD(model.parameters(), lr=LEARNING_RATE)

# ── History ───────────────────────────────────────────────────────────
history = {
    'train_loss': [], 'test_loss': [],
    'train_acc' : [], 'test_acc' : [],
    'W1_norm'   : [], 'W2_norm'  : [],
    'dW1_norm'  : [], 'dW2_norm' : [],
}

print("=" * 78)
print(f"  Training  |  lr={LEARNING_RATE}  |  hidden=8  |  epochs={EPOCHS}  |  SGD  |  7 classes")
print("=" * 78)

hdr = (f"{'Epoch':>6} | {'Train Loss':>11} | {'Train Acc':>10} "
       f"| {'Test Loss':>10} | {'Test Acc':>9} | {'|W1| mean':>9} | {'|W2| mean':>9}")
print(hdr)
print("-" * len(hdr))

for epoch in range(1, EPOCHS + 1):

    train_loss, train_acc = train_epoch(model, train_loader, criterion, optimizer)
    test_loss,  test_acc  = evaluate(model, test_loader, criterion)

    # Capture fresh gradients for logging
    model.train()
    Xb, yb = next(iter(train_loader))
    optimizer.zero_grad()
    criterion(model(Xb), yb).backward()

    W1  = model.hidden.weight.data
    W2  = model.output.weight.data
    dW1 = model.hidden.weight.grad
    dW2 = model.output.weight.grad

    history['train_loss'].append(train_loss)
    history['test_loss'].append(test_loss)
    history['train_acc'].append(train_acc)
    history['test_acc'].append(test_acc)
    history['W1_norm'].append(W1.abs().mean().item())
    history['W2_norm'].append(W2.abs().mean().item())
    history['dW1_norm'].append(dW1.abs().mean().item() if dW1 is not None else 0.0)
    history['dW2_norm'].append(dW2.abs().mean().item() if dW2 is not None else 0.0)

    if epoch % PRINT_EVERY == 0 or epoch == 1:
        print(f"{epoch:>6} | {train_loss:>11.4f} | {train_acc*100:>9.2f}% "
              f"| {test_loss:>10.4f} | {test_acc*100:>8.2f}% "
              f"| {W1.abs().mean().item():>9.5f} | {W2.abs().mean().item():>9.5f}")

print("-" * len(hdr))
print(f"\n Final Test Accuracy : {test_acc*100:.2f}%")

[cell 17 markdown]
## Step 8: Detailed Weight & Gradient Snapshots Every 10 Epochs

[cell 18 code]
model2     = MLP(n_input=16, n_hidden=8, n_output=n_classes).to(DEVICE)
optimizer2 = optim.SGD(model2.parameters(), lr=LEARNING_RATE)
criterion2 = nn.CrossEntropyLoss()
feature_names = df.columns[:-1].tolist()

print("=" * 72)
print("  Detailed Weight & Gradient Snapshots (every 10 epochs)")
print("=" * 72)

for epoch in range(1, EPOCHS + 1):
    train_loss, train_acc = train_epoch(model2, train_loader, criterion2, optimizer2)

    # Capture gradients
    model2.train()
    Xb, yb = next(iter(train_loader))
    optimizer2.zero_grad()
    criterion2(model2(Xb), yb).backward()

    if epoch % PRINT_EVERY == 0:
        W1  = model2.hidden.weight.data.cpu().numpy()   # (8, 16)
        b1  = model2.hidden.bias.data.cpu().numpy()     # (8,)
        W2  = model2.output.weight.data.cpu().numpy()   # (7, 8)
        dW1 = model2.hidden.weight.grad.cpu().numpy()   # (8, 16)
        dW2 = model2.output.weight.grad.cpu().numpy()   # (7, 8)

        print(f"\n{'─'*72}")
        print(f"  EPOCH {epoch:>3}  |  Train Loss = {train_loss:.4f}  "
              f" Train Acc = {train_acc*100:.2f}%")
        print(f"{'─'*72}")

        print("  [W1] Input → Hidden  (all 8 hidden neurons)")
        for h in range(8):
            w_row = W1[h]   # weights from all 16 inputs into hidden neuron h
            print(f"    H{h+1}: mean={w_row.mean():+.5f}  "
                  f"std={w_row.std():.5f}  "
                  f"min={w_row.min():+.5f}  "
                  f"max={w_row.max():+.5f}  "
                  f"bias={b1[h]:+.5f}")

        print("\n  [W2] Hidden → Output  (per class)")
        for c in range(n_classes):
            w_row = W2[c]   # weights from all 8 hidden neurons into output class c
            print(f"    {class_names[c]:>10}: mean={w_row.mean():+.5f}  "
                  f"std={w_row.std():.5f}  "
                  f"min={w_row.min():+.5f}  "
                  f"max={w_row.max():+.5f}")

        print(f"\n  Gradient norms →  "
              f"|∂L/∂W1| = {np.abs(dW1).mean():.7f}   "
              f"|∂L/∂W2| = {np.abs(dW2).mean():.7f}")
        print(f"  (shrinking gradient norms = network converging)")

print(f"\n{'='*72}")
print("  Weight inspection complete!")

[cell 19 markdown]
##  Step 9: Convergence Plots

[cell 20 code]
epochs_x     = np.arange(1, EPOCHS + 1)
check_epochs = list(range(10, EPOCHS + 1, 10))
check_idx    = [e - 1 for e in check_epochs]

COLORS = {'train': '#e63946', 'test': '#457b9d',
          'W1': '#2a9d8f',    'W2': '#e9c46a',
          'dW1': '#f4a261',   'dW2': '#264653'}

fig = plt.figure(figsize=(16, 12))
gs  = gridspec.GridSpec(2, 2, hspace=0.42, wspace=0.35)

# ── Plot 1: Loss ──────────────────────────────────────────────────────
ax1 = fig.add_subplot(gs[0, 0])
ax1.plot(epochs_x, history['train_loss'], color=COLORS['train'], lw=2.5, label='Train Loss')
ax1.plot(epochs_x, history['test_loss'],  color=COLORS['test'],  lw=2.5, label='Test Loss', ls='--')
ax1.scatter([e for e in check_epochs], [history['train_loss'][i] for i in check_idx],
            color=COLORS['train'], s=80, zorder=5)
ax1.scatter([e for e in check_epochs], [history['test_loss'][i]  for i in check_idx],
            color=COLORS['test'],  s=80, zorder=5, marker='D')
ax1.set_title('Loss Convergence (Cross-Entropy)', fontsize=13, fontweight='bold')
ax1.set_xlabel('Epoch'); ax1.set_ylabel('Cross-Entropy Loss')
ax1.legend(); ax1.grid(True, alpha=0.3)
ax1.set_xticks(range(0, EPOCHS + 1, 5))

# ── Plot 2: Accuracy ─────────────────────────────────────────────────
ax2 = fig.add_subplot(gs[0, 1])
ax2.plot(epochs_x, [a*100 for a in history['train_acc']], color=COLORS['train'], lw=2.5, label='Train Acc')
ax2.plot(epochs_x, [a*100 for a in history['test_acc']],  color=COLORS['test'],  lw=2.5, label='Test Acc', ls='--')
ax2.scatter([e for e in check_epochs], [history['train_acc'][i]*100 for i in check_idx],
            color=COLORS['train'], s=80, zorder=5)
ax2.scatter([e for e in check_epochs], [history['test_acc'][i]*100  for i in check_idx],
            color=COLORS['test'],  s=80, zorder=5, marker='D')
ax2.set_title('Accuracy Convergence (7 Classes)', fontsize=13, fontweight='bold')
ax2.set_xlabel('Epoch'); ax2.set_ylabel('Accuracy (%)')
ax2.legend(); ax2.grid(True, alpha=0.3)
ax2.set_xticks(range(0, EPOCHS + 1, 5))

# ── Plot 3: Weight Norms ─────────────────────────────────────────────
ax3 = fig.add_subplot(gs[1, 0])
ax3.plot(epochs_x, history['W1_norm'], color=COLORS['W1'], lw=2.5, label='|W1| mean  (input→hidden)')
ax3.plot(epochs_x, history['W2_norm'], color=COLORS['W2'], lw=2.5, label='|W2| mean  (hidden→output)', ls='--')
for e in check_epochs:
    ax3.axvline(e, color='gray', lw=0.8, ls=':', alpha=0.7)
    ax3.text(e + 0.2, max(history['W1_norm']) * 0.97, f'e{e}', fontsize=7, color='gray')
ax3.set_title('Mean Absolute Weight Values', fontsize=13, fontweight='bold')
ax3.set_xlabel('Epoch'); ax3.set_ylabel('Mean |w|')
ax3.legend(); ax3.grid(True, alpha=0.3)
ax3.set_xticks(range(0, EPOCHS + 1, 5))

# ── Plot 4: Gradient Norms ───────────────────────────────────────────
ax4 = fig.add_subplot(gs[1, 1])
ax4.semilogy(epochs_x, history['dW1_norm'], color=COLORS['dW1'], lw=2.5, label='|∂L/∂W1|  hidden grads')
ax4.semilogy(epochs_x, history['dW2_norm'], color=COLORS['dW2'], lw=2.5, label='|∂L/∂W2|  output grads', ls='--')
for e in check_epochs:
    ax4.axvline(e, color='gray', lw=0.8, ls=':', alpha=0.7)
ax4.set_title('Gradient Norms (log scale)', fontsize=13, fontweight='bold')
ax4.set_xlabel('Epoch'); ax4.set_ylabel('Mean |gradient| (log)')
ax4.legend(); ax4.grid(True, alpha=0.3)
ax4.set_xticks(range(0, EPOCHS + 1, 5))

fig.suptitle(
    'PyTorch MLP — Backpropagation Convergence on Dry Bean Dataset\n'
    'Architecture: 16 → 8 (ReLU) → 7 (Softmax)  |  SGD  |  lr=0.05',
    fontsize=14, fontweight='bold', y=1.01
)
plt.savefig('convergence_plots.png', dpi=130, bbox_inches='tight')
plt.show()

[cell 21 markdown]
## Step 11: Step-by-Step Backprop Trace - One Sample

[cell 22 code]
sample_idx = 5
x_s = X_train[sample_idx:sample_idx+1]   # (1, 16)
y_s = y_train[sample_idx:sample_idx+1]   # (1,)  integer label
true_class = class_names[y_s.item()]

print("=" * 68)
print(f"  STEP-BY-STEP BACKPROPAGATION TRACE (sample #{sample_idx})")
print(f"  True class : {true_class}  (index {y_s.item()})")
print("=" * 68)

# Extract weights as numpy arrays
W1 = model.hidden.weight.data.cpu().numpy()   # (8, 16)
b1 = model.hidden.bias.data.cpu().numpy()     # (8,)
W2 = model.output.weight.data.cpu().numpy()   # (7, 8)
b2 = model.output.bias.data.cpu().numpy()     # (7,)
x_np = x_s.cpu().numpy()                      # (1, 16)
y_idx = y_s.item()                            # integer: 0..6

# ────────────────────────────────────────────────────────────────────
# FORWARD PASS
# ────────────────────────────────────────────────────────────────────
print("\n──── FORWARD PASS ────")

Z1 = x_np @ W1.T + b1                       # (1, 8)
A1 = np.maximum(0, Z1)                       # ReLU  → (1, 8)
print(f"Z1 (pre-ReLU)  : {Z1[0].round(4)}")
print(f"A1 (post-ReLU) : {A1[0].round(4)}")
print(f"  ({np.sum(Z1[0] <= 0)} of 8 neurons are dead/zero after ReLU)")

Z2 = A1 @ W2.T + b2                          # (1, 7)  raw logits
exp_z = np.exp(Z2 - Z2.max(axis=1, keepdims=True))  # numerically stable
A2 = exp_z / exp_z.sum(axis=1, keepdims=True)        # softmax → (1, 7)

print(f"\nZ2 (logits)  : {Z2[0].round(4)}")
print(f"\nA2 (softmax probabilities):")
for cls, p in zip(class_names, A2[0]):
    marker  = ' ← predicted' if p == A2[0].max() else ''
    marker += ' ← TRUE'      if cls == true_class   else ''
    print(f"  {cls:>10}: {p:.4f}{marker}")

ce_loss = -np.log(A2[0, y_idx] + 1e-9)
print(f"\nCross-Entropy loss = -log(p_true) = -log({A2[0, y_idx]:.4f}) = {ce_loss:.4f}")

# ────────────────────────────────────────────────────────────────────
# BACKWARD PASS — manual chain rule
# ────────────────────────────────────────────────────────────────────
print("\n──── BACKWARD PASS (manual chain rule) ────")

# δ² = Softmax(Z2) − one_hot(y)
# This clean form comes from d(CrossEntropy)/dZ2 = A2 - Y
delta2 = A2.copy()              # start from softmax output
delta2[0, y_idx] -= 1.0         # subtract 1 at the true class position  (1, 7)
print(f"δ² = A2 − one_hot(y)  [output layer error]:")
for cls, d in zip(class_names, delta2[0]):
    marker = ' ← true class (subtract 1 here)' if cls == true_class else ''
    print(f"  {cls:>10}: {d:+.4f}{marker}")

dW2 = delta2.T @ A1              # (7, 8)
db2 = delta2                     # (1, 7)
print(f"\n∂L/∂W2 shape={dW2.shape}  norm={np.abs(dW2).mean():.6f}")

# Backprop through W2, then through ReLU
delta1 = (delta2 @ W2) * (Z1 > 0).astype(float)   # (1, 8)
dW1    = delta1.T @ x_np                            # (8, 16)
db1    = delta1                                     # (1, 8)

print(f"\nδ¹ (hidden layer error via chain rule):")
print(f"  {delta1[0].round(6)}")
print(f"\n∂L/∂W1 shape={dW1.shape}  norm={np.abs(dW1).mean():.6f}")

[cell 23 markdown]
## Step 12: Final Evaluation — Per-Class Accuracy

[cell 24 code]
model.eval()
with torch.no_grad():
    all_preds = model(X_test).argmax(dim=1).cpu().numpy()
    all_true  = y_test.cpu().numpy()

final_acc = np.mean(all_preds == all_true)

print("=" * 55)
print("  Final Evaluation on Test Set")
print("=" * 55)
print(f"  Overall Accuracy : {final_acc * 100:.2f}%\n")
print(f"  {'Class':>10} | {'Correct':>7} | {'Total':>7} | {'Accuracy':>9}")
print(f"  {'─'*42}")
for i, cls in enumerate(class_names):
    mask      = all_true == i
    n_total   = mask.sum()
    n_correct = (all_preds[mask] == i).sum()
    acc_cls   = n_correct / n_total if n_total > 0 else 0.0
    print(f"  {cls:>10} | {n_correct:>7} | {n_total:>7} | {acc_cls*100:>8.2f}%")

# Confusion matrix
print(f"\n  Confusion Matrix (rows=true, cols=predicted):")
conf = np.zeros((n_classes, n_classes), dtype=int)
for t, p in zip(all_true, all_preds):
    conf[t, p] += 1

header_str = f"  {'':>10}  " + "  ".join(f"{c:>8}" for c in class_names)
print(header_str)
for i, cls in enumerate(class_names):
    row = "  ".join(f"{conf[i,j]:>8}" for j in range(n_classes))
    print(f"  {cls:>10}  {row}")

print(f"\n{'='*55}")
print("  Architecture Summary")
print(f"{'='*55}")
print(f"  Framework  : PyTorch {torch.__version__}")
print(f"  Input      : 16 shape features")
print(f"  Hidden     : 8 neurons (ReLU)")
print(f"  Output     : 7 neurons (Softmax via CrossEntropyLoss)")
print(f"  Params     : {sum(p.numel() for p in model.parameters())}")
print(f"  Optimizer  : SGD  |  lr={LEARNING_RATE}")
print(f"  Epochs     : {EPOCHS}")
print(f"  Final Test Accuracy : {final_acc*100:.2f}%")
print(f"{'='*55}")

[cell 25 markdown]
## Define MLP & Train with Full Weight History Recording

We record **every weight, gradient, and delta** at every epoch — giving us the full picture of how backpropagation updates each individual connection.

[cell 26 code]
class MLP(nn.Module):
    def __init__(self, n_in=16, n_hidden=8, n_out=7):
        super().__init__()
        self.hidden = nn.Linear(n_in, n_hidden)
        self.output = nn.Linear(n_hidden, n_out)
        self.relu   = nn.ReLU()
        nn.init.zeros_(self.hidden.bias)
        nn.init.zeros_(self.output.bias)
    def forward(self, x):
        return self.output(self.relu(self.hidden(x)))


EPOCHS = 50
LR     = 0.05
model     = MLP().to(DEVICE)
criterion = nn.CrossEntropyLoss()
optimizer = optim.SGD(model.parameters(), lr=LR)

# ── Storage: shape described at each key ────────────────────────────
# W1_hist[epoch, neuron, feature]  — (50, 8, 16)
# W2_hist[epoch, class,  neuron ]  — (50, 7,  8)
W1_hist  = np.zeros((EPOCHS, 8, 16))
W2_hist  = np.zeros((EPOCHS, 7,  8))
b1_hist  = np.zeros((EPOCHS, 8))
b2_hist  = np.zeros((EPOCHS, 7))
dW1_hist = np.zeros((EPOCHS, 8, 16))   # gradients
dW2_hist = np.zeros((EPOCHS, 7,  8))
delta_W1 = np.zeros((EPOCHS, 8, 16))   # actual weight change Δw = -lr * grad
delta_W2 = np.zeros((EPOCHS, 7,  8))
loss_hist      = []
acc_hist_train = []
acc_hist_test  = []

print('Training and recording every weight at every epoch...')
print(f'{"Epoch":>6} | {"Train Loss":>11} | {"Train Acc":>10} | {"Test Acc":>9}')
print('-' * 46)

for epoch in range(EPOCHS):

    # Save weights BEFORE update
    W1_before = model.hidden.weight.data.cpu().numpy().copy()

    # ── Training ────────────────────────────────────────────────────
    model.train()
    ep_loss, correct, total = 0.0, 0, 0
    for Xb, yb in train_loader:
        optimizer.zero_grad()
        loss = criterion(model(Xb), yb)
        loss.backward()
        optimizer.step()
        ep_loss += loss.item() * Xb.size(0)
        correct += (model(Xb).argmax(1) == yb).sum().item()
        total   += Xb.size(0)

    # Capture gradients via one dedicated backward on full train set
    model.train()
    optimizer.zero_grad()
    criterion(model(X_train), y_train).backward()

    # ── Record everything ────────────────────────────────────────────
    W1_hist[epoch]  = model.hidden.weight.data.cpu().numpy()
    W2_hist[epoch]  = model.output.weight.data.cpu().numpy()
    b1_hist[epoch]  = model.hidden.bias.data.cpu().numpy()
    b2_hist[epoch]  = model.output.bias.data.cpu().numpy()
    dW1_hist[epoch] = model.hidden.weight.grad.cpu().numpy()
    dW2_hist[epoch] = model.output.weight.grad.cpu().numpy()
    delta_W1[epoch] = W1_hist[epoch] - W1_before   # actual change this epoch
    if epoch > 0:
        delta_W2[epoch] = W2_hist[epoch] - W2_hist[epoch - 1]

    # Test accuracy
    model.eval()
    with torch.no_grad():
        test_acc = (model(X_test).argmax(1) == y_test).float().mean().item()

    loss_hist.append(ep_loss / total)
    acc_hist_train.append(correct / total)
    acc_hist_test.append(test_acc)

    if (epoch + 1) % 10 == 0 or epoch == 0:
        print(f'{epoch+1:>6} | {ep_loss/total:>11.4f} | {correct/total*100:>9.2f}% | {test_acc*100:>8.2f}%')

print(f'\nRecorded W1_hist shape: {W1_hist.shape}  (epochs × neurons × features)')
print(f'   Recorded W2_hist shape: {W2_hist.shape}  (epochs × classes × neurons)')

[cell 27 markdown]
## Weight Trajectory — Every Connection in Hidden Layer

Each line = one weight connecting an **input feature → hidden neuron**.  
One panel per hidden neuron (H1–H8). Shows how backprop gradually adjusts each weight.

[cell 28 code]
epochs_x = np.arange(1, EPOCHS + 1)
FEAT_COLORS   = ['#e6194b', '#3cb44b', '#0000ff', '#f58231', '#911eb4', '#00ced1', '#aa6e28', '#000075', '#e63900', '#008000', '#9400d3', '#ff8c00', '#dc143c', '#00868b', '#8b0000', '#006400']
feat_colors   = FEAT_COLORS

fig, axes = plt.subplots(2, 4, figsize=(20, 9))

for h, ax in enumerate(axes.flat):
    for f in range(16):
        ax.plot(epochs_x, W1_hist[:, h, f],
                color=feat_colors[f], lw=1.2, alpha=0.85,
                label=feature_names[f])
    ax.axhline(0, color='#555577', lw=0.7, ls='--')
    ax.set_title(f'Hidden Neuron H{h+1}', fontsize=11, fontweight='bold', color='#222222')
    ax.set_xlabel('Epoch', fontsize=9)
    ax.set_ylabel('Weight value', fontsize=9)
    ax.grid(True)
    # Mark epoch-10 checkpoints
    for ck in range(10, EPOCHS + 1, 10):
        ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)

# Legend below the figure
handles = [plt.Line2D([0],[0], color=feat_colors[f], lw=2) for f in range(16)]
fig.legend(handles, feature_names,
           loc='lower center', ncol=8, fontsize=8,
           bbox_to_anchor=(0.5, -0.04), framealpha=0.3)

fig.suptitle('W1: Individual Weight Trajectories per Hidden Neuron\n'
             '(Each line = one input feature → this neuron)',
             fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w1_weight_trajectories.png', dpi=130, bbox_inches='tight')
plt.show()

[cell 29 markdown]
## Gradient Magnitude per Neuron — How Hard Backprop is Pushing Each Weight

[cell 30 code]
fig, axes = plt.subplots(2, 4, figsize=(20, 9))

for h, ax in enumerate(axes.flat):
    for f in range(16):
        ax.plot(epochs_x, np.abs(dW1_hist[:, h, f]),
                color=feat_colors[f], lw=1.2, alpha=0.85)
    ax.set_title(f'|∂L/∂W1| — Neuron H{h+1}', fontsize=11, fontweight='bold', color='#222222')
    ax.set_xlabel('Epoch', fontsize=9)
    ax.set_ylabel('|Gradient|', fontsize=9)
    ax.grid(True)
    for ck in range(10, EPOCHS + 1, 10):
        ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)

handles = [plt.Line2D([0],[0], color=feat_colors[f], lw=2) for f in range(16)]
fig.legend(handles, feature_names,
           loc='lower center', ncol=8, fontsize=8,
           bbox_to_anchor=(0.5, -0.04), framealpha=0.3)

fig.suptitle('W1: Gradient Magnitudes |∂L/∂w| per Hidden Neuron\n'
             '(Decreasing gradients = converging)',
             fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w1_gradient_magnitudes.png', dpi=130, bbox_inches='tight')
plt.show()

[cell 31 markdown]
## Actual Weight Delta Δw per Neuron (What Backprop Actually Applies)

`Δw = w_after − w_before = −lr × ∂L/∂w`  
This is the **signed update** — positive means the weight increased, negative means it decreased.

[cell 32 code]
fig, axes = plt.subplots(2, 4, figsize=(20, 9))

for h, ax in enumerate(axes.flat):
    for f in range(16):
        ax.plot(epochs_x, delta_W1[:, h, f],
                color=feat_colors[f], lw=1.0, alpha=0.85)
    ax.axhline(0, color='#ffffff', lw=0.8, ls='--', alpha=0.5)
    ax.set_title(f'Δw = −lr·∂L/∂w — Neuron H{h+1}', fontsize=10,
                 fontweight='bold', color='#222222')
    ax.set_xlabel('Epoch', fontsize=9)
    ax.set_ylabel('Weight change Δw', fontsize=9)
    ax.grid(True)
    for ck in range(10, EPOCHS + 1, 10):
        ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)

handles = [plt.Line2D([0],[0], color=feat_colors[f], lw=2) for f in range(16)]
fig.legend(handles, feature_names,
           loc='lower center', ncol=8, fontsize=8,
           bbox_to_anchor=(0.5, -0.04), framealpha=0.3)

fig.suptitle('W1: Weight Updates Δw Applied Each Epoch per Hidden Neuron\n'
             '(+ve = weight increased, −ve = weight decreased)',
             fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w1_weight_deltas.png', dpi=130, bbox_inches='tight')
plt.show()

[cell 33 markdown]
## Output Layer W2 — Weight Trajectory per Class Neuron

Each panel = one output class neuron. Each line = its connection to one hidden neuron.

[cell 34 code]
HIDDEN_COLORS = ['#e6194b', '#0000ff', '#3cb44b', '#ff8c00', '#9400d3', '#00ced1', '#dc143c', '#228b22']
hidden_colors = HIDDEN_COLORS

fig, axes = plt.subplots(2, 4, figsize=(20, 9))

# 7 class panels + 1 combined summary
for c in range(7):
    ax = axes.flat[c]
    for h in range(8):
        ax.plot(epochs_x, W2_hist[:, c, h],
                color=hidden_colors[h], lw=1.8, alpha=0.9,
                label=f'H{h+1}')
    ax.axhline(0, color='#555577', lw=0.7, ls='--')
    ax.set_title(f'Output: {class_names[c]}', fontsize=11,
                 fontweight='bold', color='#222222')
    ax.set_xlabel('Epoch', fontsize=9)
    ax.set_ylabel('Weight value', fontsize=9)
    ax.grid(True)
    for ck in range(10, EPOCHS + 1, 10):
        ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)
    ax.legend(fontsize=7, ncol=4, loc='upper right')

# Panel 8: mean absolute W2 per class
ax8 = axes.flat[7]
CLASS_COLORS  = ['#e6194b', '#0000ff', '#3cb44b', '#ff8c00', '#9400d3', '#00ced1', '#dc143c']
class_palette = CLASS_COLORS
for c in range(7):
    mean_abs = np.abs(W2_hist[:, c, :]).mean(axis=1)
    ax8.plot(epochs_x, mean_abs, color=class_palette[c],
             lw=2, label=class_names[c])
ax8.set_title('Mean |W2| per Output Class', fontsize=11,
              fontweight='bold', color='#222222')
ax8.set_xlabel('Epoch', fontsize=9)
ax8.set_ylabel('Mean |weight|', fontsize=9)
ax8.legend(fontsize=8)
ax8.grid(True)

fig.suptitle('W2: Output Layer Weight Trajectories per Class\n'
             '(Each line = hidden neuron → this class output)',
             fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w2_weight_trajectories.png', dpi=130, bbox_inches='tight')
plt.show()

[cell 35 markdown]
## W2 Gradient Magnitudes per Output Neuron

[cell 36 code]
fig, axes = plt.subplots(2, 4, figsize=(20, 9))

for c in range(7):
    ax = axes.flat[c]
    for h in range(8):
        ax.plot(epochs_x, np.abs(dW2_hist[:, c, h]),
                color=hidden_colors[h], lw=1.8, alpha=0.9, label=f'H{h+1}')
    ax.set_title(f'|∂L/∂W2| — {class_names[c]}', fontsize=11,
                 fontweight='bold', color='#222222')
    ax.set_xlabel('Epoch', fontsize=9)
    ax.set_ylabel('|Gradient|', fontsize=9)
    ax.legend(fontsize=7, ncol=4, loc='upper right')
    ax.grid(True)
    for ck in range(10, EPOCHS + 1, 10):
        ax.axvline(ck, color='#ff6b6b', lw=0.6, ls=':', alpha=0.6)

# Panel 8: all gradient norms overlaid
ax8 = axes.flat[7]
ax8.semilogy(epochs_x, np.abs(dW1_hist).mean(axis=(1,2)),
             color='#a8dadc', lw=2.5, label='|∂L/∂W1| (hidden)')
ax8.semilogy(epochs_x, np.abs(dW2_hist).mean(axis=(1,2)),
             color='#e63946', lw=2.5, ls='--', label='|∂L/∂W2| (output)')
ax8.set_title('Overall Gradient Norms (log)', fontsize=11,
              fontweight='bold', color='#222222')
ax8.set_xlabel('Epoch', fontsize=9)
ax8.set_ylabel('Mean |gradient| (log)', fontsize=9)
ax8.legend(fontsize=9)
ax8.grid(True)

fig.suptitle('W2: Gradient Magnitudes |∂L/∂w| per Output Class\n'
             '(Each line = gradient for the hidden→class connection)',
             fontsize=14, fontweight='bold', color='#222222', y=1.01)
plt.tight_layout()
plt.savefig('w2_gradient_magnitudes.png', dpi=130, bbox_inches='tight')
plt.show()

[cell 37 markdown]
**We record both W1 and W2 after every epoch.**

[cell 38 code]
torch.manual_seed(42)
np.random.seed(42)
DEVICE = torch.device('cuda' if torch.cuda.is_available() else 'cpu')

# ── Load data ────────────────────────────────────────────────────────
df            = pd.read_excel('Dry_Bean_Dataset.xlsx')
FEATURE_NAMES = df.columns[:-1].tolist()

X_raw = df.drop(columns=['Class']).values.astype(np.float32)
y_raw = df['Class'].values
le    = LabelEncoder()
y_int = le.fit_transform(y_raw).astype(np.int64)
CLASS_NAMES = le.classes_

X_scaled = StandardScaler().fit_transform(X_raw).astype(np.float32)
X_tr, X_te, y_tr, y_te = train_test_split(
    X_scaled, y_int, test_size=0.2, random_state=42, stratify=y_int
)

X_train = torch.tensor(X_tr).to(DEVICE)
y_train = torch.tensor(y_tr).to(DEVICE)
train_loader = DataLoader(TensorDataset(X_train, y_train), batch_size=256, shuffle=True)

# ── Model ────────────────────────────────────────────────────────────
class MLP(nn.Module):
    def __init__(self):
        super().__init__()
        self.hidden = nn.Linear(16, 8)
        self.output = nn.Linear(8,  7)
        self.relu   = nn.ReLU()
        nn.init.zeros_(self.hidden.bias)
        nn.init.zeros_(self.output.bias)
    def forward(self, x):
        return self.output(self.relu(self.hidden(x)))

model     = MLP().to(DEVICE)
criterion = nn.CrossEntropyLoss()
optimizer = optim.SGD(model.parameters(), lr=0.05)

# ── Train 50 epochs — record W1 and W2 after every epoch ─────────────
EPOCHS  = 50
# W1: PyTorch shape (8 neurons, 16 features) → W1_hist[epoch, neuron, feature]
# W2: PyTorch shape (7 classes,  8 neurons ) → W2_hist[epoch, class,  neuron ]
W1_hist   = np.zeros((EPOCHS, 8, 16))
W2_hist   = np.zeros((EPOCHS, 7,  8))
loss_hist = []

for epoch in range(EPOCHS):
    model.train()
    ep_loss = 0.0
    for Xb, yb in train_loader:
        optimizer.zero_grad()
        loss = criterion(model(Xb), yb)
        loss.backward()
        optimizer.step()
        ep_loss += loss.item()
    W1_hist[epoch] = model.hidden.weight.data.cpu().numpy()  # (8, 16)
    W2_hist[epoch] = model.output.weight.data.cpu().numpy()  # (7,  8)
    loss_hist.append(ep_loss / len(train_loader))

print('Training complete!')
print(f'W1_hist : {W1_hist.shape}  (epochs x neurons x features)')
print(f'W2_hist : {W2_hist.shape}  (epochs x classes x neurons)')
print(f'Final loss : {loss_hist[-1]:.4f}')

[cell 39 markdown]
## Choose Features & Classes to Inspect

[cell 40 code]
# ── W1: 3 input features to inspect ────────────────────────────────
SELECTED_FEATURES = ['Area', 'Perimeter', 'Eccentricity']
FEAT_IDX = [FEATURE_NAMES.index(f) for f in SELECTED_FEATURES]

# ── W2: 3 output classes to inspect ─────────────────────────────────
SELECTED_CLASSES = ['BOMBAY', 'SEKER', 'DERMASON']
CLASS_IDX = [list(CLASS_NAMES).index(c) for c in SELECTED_CLASSES]

# ── Same high-contrast colour per hidden neuron in every plot ────────
NEURON_COLORS = [
    '#e6194b',  # H1  vivid red
    '#0000ff',  # H2  pure blue
    '#3cb44b',  # H3  vivid green
    '#ff8c00',  # H4  dark orange
    '#9400d3',  # H5  vivid purple
    '#00ced1',  # H6  dark turquoise
    '#dc143c',  # H7  crimson
    '#228b22',  # H8  forest green
]

epochs_x = np.arange(1, EPOCHS + 1)

print('W1 features :', SELECTED_FEATURES, '  indices:', FEAT_IDX)
print('W2 classes  :', SELECTED_CLASSES,  '  indices:', CLASS_IDX)

[cell 41 markdown]
## Helper — Reusable Plot Function

[cell 42 code]
def plot_weight_grid(series_8, neuron_colors, epochs_x, ylabel_fn, suptitle, savename):
    """
    series_8  : list of 8 arrays shape (EPOCHS,) — one per hidden neuron
    ylabel_fn : function of h (int) -> y-axis label string
    """
    fig, axes = plt.subplots(2, 4, figsize=(18, 7), sharey=False)
    fig.patch.set_facecolor('white')

    for h, ax in enumerate(axes.flat):
        w     = series_8[h]
        color = neuron_colors[h]

        ax.fill_between(epochs_x, w, alpha=0.15, color=color)
        ax.plot(epochs_x, w, color=color, lw=2.2, zorder=3)

        # Dot + annotated value at every-10-epoch checkpoint
        for ck in range(0, len(epochs_x), 10):
            ax.scatter(ck + 1, w[ck], color=color, s=60,
                       zorder=5, edgecolors='black', linewidths=0.8)
            ax.annotate(f'{w[ck]:.3f}',
                        xy=(ck + 1, w[ck]),
                        xytext=(0, 9), textcoords='offset points',
                        ha='center', fontsize=7.5, color='#111111',
                        bbox=dict(boxstyle='round,pad=0.15', fc='white',
                                  ec=color, lw=0.8, alpha=0.85))

        # Dashed init line, dotted final line
        ax.axhline(w[0],  color='gray',    lw=0.9, ls='--', alpha=0.6,
                   label=f'Init   {w[0]:+.3f}')
        ax.axhline(w[-1], color='#111111', lw=0.9, ls=':',  alpha=0.85,
                   label=f'Final {w[-1]:+.3f}')

        ax.set_title(f'H{h+1}   Dw = {w[-1]-w[0]:+.3f}',
                     fontsize=11, fontweight='bold', color=color, pad=5)
        ax.set_xlabel('Epoch', fontsize=9)
        ax.set_ylabel(ylabel_fn(h), fontsize=8)
        ax.legend(fontsize=7, loc='best')
        ax.grid(True, alpha=0.35)
        ax.set_xlim(1, len(epochs_x))
        ax.xaxis.set_major_locator(ticker.MultipleLocator(10))

    fig.suptitle(suptitle, fontsize=13, fontweight='bold', color='#111111')
    plt.tight_layout()
    plt.savefig(savename, dpi=130, bbox_inches='tight')
    plt.show()

print('plot_weight_grid() ready')

[cell 43 markdown]
---
## W1 — Input → Hidden Layer
`W1[neuron, feature]` — weight connecting an input feature to a hidden neuron.

**Backprop update rule:** `W1 -= lr * dL/dW1`  where  `dL/dW1 = delta1.T @ X`

One cell per feature. 8 panels per cell (one per hidden neuron).

[cell 44 markdown]
### W1 — Feature: `Area`

[cell 45 code]
import matplotlib.ticker as ticker

fname    = 'Area'
feat_idx = FEAT_IDX[0]

plot_weight_grid(
    series_8     = [W1_hist[:, h, feat_idx] for h in range(8)],
    neuron_colors= NEURON_COLORS,
    epochs_x     = epochs_x,
    ylabel_fn    = lambda h: f'w(Area -> H{h+1})',
    suptitle     = (
        'W1 | Feature: "Area"\n'
        'Each panel = one hidden neuron   y = weight value after every epoch'
    ),
    savename     = 'W1_Area.png'
)

[cell 46 markdown]
### W1 — Feature: `Perimeter`

[cell 47 code]
fname    = 'Perimeter'
feat_idx = FEAT_IDX[1]

plot_weight_grid(
    series_8     = [W1_hist[:, h, feat_idx] for h in range(8)],
    neuron_colors= NEURON_COLORS,
    epochs_x     = epochs_x,
    ylabel_fn    = lambda h: f'w(Perimeter -> H{h+1})',
    suptitle     = (
        'W1 | Feature: "Perimeter"\n'
        'Each panel = one hidden neuron   y = weight value after every epoch'
    ),
    savename     = 'W1_Perimeter.png'
)

[cell 48 markdown]
### W1 — Feature: `Eccentricity`

[cell 49 code]
fname    = 'Eccentricity'
feat_idx = FEAT_IDX[2]

plot_weight_grid(
    series_8     = [W1_hist[:, h, feat_idx] for h in range(8)],
    neuron_colors= NEURON_COLORS,
    epochs_x     = epochs_x,
    ylabel_fn    = lambda h: f'w(Eccentricity -> H{h+1})',
    suptitle     = (
        'W1 | Feature: "Eccentricity"\n'
        'Each panel = one hidden neuron   y = weight value after every epoch'
    ),
    savename     = 'W1_Eccentricity.png'
)

[cell 50 markdown]
---
## W2 — Hidden → Output Layer
`W2[class, neuron]` — weight connecting a hidden neuron to an output class neuron.

**Backprop update rule:** `W2 -= lr * dL/dW2`  where  `dL/dW2 = delta2.T @ A1`

One cell per class. 8 panels per cell (one per hidden neuron).
Each panel shows how strongly that hidden neuron 'votes' for this class over training.

[cell 51 markdown]
### W2 — Output Class: `BOMBAY`

[cell 52 code]
cname     = 'BOMBAY'
class_idx = CLASS_IDX[0]

plot_weight_grid(
    series_8     = [W2_hist[:, class_idx, h] for h in range(8)],
    neuron_colors= NEURON_COLORS,
    epochs_x     = epochs_x,
    ylabel_fn    = lambda h: f'w(H{h+1} -> BOMBAY)',
    suptitle     = (
        'W2 | Output Class: "BOMBAY"\n'
        'Each panel = one hidden neuron   y = weight value after every epoch'
    ),
    savename     = 'W2_BOMBAY.png'
)

[cell 53 markdown]
### W2 — Output Class: `SEKER`

[cell 54 code]
cname     = 'SEKER'
class_idx = CLASS_IDX[1]

plot_weight_grid(
    series_8     = [W2_hist[:, class_idx, h] for h in range(8)],
    neuron_colors= NEURON_COLORS,
    epochs_x     = epochs_x,
    ylabel_fn    = lambda h: f'w(H{h+1} -> SEKER)',
    suptitle     = (
        'W2 | Output Class: "SEKER"\n'
        'Each panel = one hidden neuron   y = weight value after every epoch'
    ),
    savename     = 'W2_SEKER.png'
)

[cell 55 markdown]
### W2 — Output Class: `DERMASON`

[cell 56 code]
cname     = 'DERMASON'
class_idx = CLASS_IDX[2]

plot_weight_grid(
    series_8     = [W2_hist[:, class_idx, h] for h in range(8)],
    neuron_colors= NEURON_COLORS,
    epochs_x     = epochs_x,
    ylabel_fn    = lambda h: f'w(H{h+1} -> DERMASON)',
    suptitle     = (
        'W2 | Output Class: "DERMASON"\n'
        'Each panel = one hidden neuron   y = weight value after every epoch'
    ),
    savename     = 'W2_DERMASON.png'
)

[cell 57 markdown]
---
## One Neuron, Both Layers
Pick any hidden neuron and see:
- **Top row (W1):** its weights from the 3 selected features
- **Bottom row (W2):** its weights to the 3 selected classes

Change `NEURON_TO_INSPECT` to 0–7 to explore different neurons.

[cell 58 code]
NEURON_TO_INSPECT = 0   # change to 0-7

FEAT_COLORS_3  = ['#e6194b', '#0000ff', '#3cb44b']
CLASS_COLORS_3 = ['#ff8c00', '#9400d3', '#00ced1']

fig, axes = plt.subplots(2, 3, figsize=(18, 8))
fig.patch.set_facecolor('white')

# ── Top row: W1 — how this neuron receives from each feature ─────────
for col, (fname, fidx, fc) in enumerate(
        zip(SELECTED_FEATURES, FEAT_IDX, FEAT_COLORS_3)):
    ax = axes[0, col]
    w  = W1_hist[:, NEURON_TO_INSPECT, fidx]
    ax.fill_between(epochs_x, w, alpha=0.18, color=fc)
    ax.plot(epochs_x, w, color=fc, lw=2.4)
    ax.scatter(epochs_x, w, color=fc, s=20,
               edgecolors='black', linewidths=0.4, zorder=4)
    for ck in range(0, EPOCHS, 10):
        ax.annotate(f'{w[ck]:.3f}', xy=(ck+1, w[ck]),
                    xytext=(0, 10), textcoords='offset points',
                    ha='center', fontsize=8.5, fontweight='bold', color='#111111',
                    bbox=dict(boxstyle='round,pad=0.2', fc='white', ec=fc, lw=1.1, alpha=0.9))
    ax.axhline(w[0],  color='gray',    lw=0.8, ls='--', alpha=0.6)
    ax.axhline(w[-1], color='#111111', lw=0.8, ls=':',  alpha=0.8)
    ax.set_title(
        f'W1: {fname} -> H{NEURON_TO_INSPECT+1}\n'
        f'Dw = {w[-1]-w[0]:+.4f}',
        fontsize=11, fontweight='bold', color=fc)
    ax.set_xlabel('Epoch'); ax.grid(True, alpha=0.35)
    ax.set_ylabel(f'w({fname} -> H{NEURON_TO_INSPECT+1})', fontsize=9)
    ax.xaxis.set_major_locator(ticker.MultipleLocator(10))

# ── Bottom row: W2 — how this neuron votes for each class ────────────
for col, (cname, cidx, cc) in enumerate(
        zip(SELECTED_CLASSES, CLASS_IDX, CLASS_COLORS_3)):
    ax = axes[1, col]
    w  = W2_hist[:, cidx, NEURON_TO_INSPECT]
    ax.fill_between(epochs_x, w, alpha=0.18, color=cc)
    ax.plot(epochs_x, w, color=cc, lw=2.4)
    ax.scatter(epochs_x, w, color=cc, s=20,
               edgecolors='black', linewidths=0.4, zorder=4)
    for ck in range(0, EPOCHS, 10):
        ax.annotate(f'{w[ck]:.3f}', xy=(ck+1, w[ck]),
                    xytext=(0, 10), textcoords='offset points',
                    ha='center', fontsize=8.5, fontweight='bold', color='#111111',
                    bbox=dict(boxstyle='round,pad=0.2', fc='white', ec=cc, lw=1.1, alpha=0.9))
    ax.axhline(w[0],  color='gray',    lw=0.8, ls='--', alpha=0.6)
    ax.axhline(w[-1], color='#111111', lw=0.8, ls=':',  alpha=0.8)
    ax.set_title(
        f'W2: H{NEURON_TO_INSPECT+1} -> {cname}\n'
        f'Dw = {w[-1]-w[0]:+.4f}',
        fontsize=11, fontweight='bold', color=cc)
    ax.set_xlabel('Epoch'); ax.grid(True, alpha=0.35)
    ax.set_ylabel(f'w(H{NEURON_TO_INSPECT+1} -> {cname})', fontsize=9)
    ax.xaxis.set_major_locator(ticker.MultipleLocator(10))

fig.suptitle(
    f'Hidden Neuron H{NEURON_TO_INSPECT+1} — Both Layers\n'
    f'Top row: W1 (what it receives from features)   |   '
    f'Bottom row: W2 (what it sends to output classes)',
    fontsize=13, fontweight='bold', color='#111111'
)
plt.tight_layout()
plt.savefig(f'both_layers_H{NEURON_TO_INSPECT+1}.png', dpi=130, bbox_inches='tight')
plt.show()