{
  "version": "2.0-exploratory-cnn",
  "date": "2026-10-09",
  "dataset": "Fashion-MNIST, official checksummed files",
  "split_seed": 20261009,
  "training_examples": 10000,
  "validation_examples": 6000,
  "test_examples": 10000,
  "image_size": 14,
  "model": "plain CNN: conv(width 16 channels, padding same) -> optional LayerNorm (GroupNorm, one group) -> activation; 2x2 max-pool after conv layers 1 and 3; adaptive average pool to 3x3; linear classifier",
  "width": 16,
  "depths": [
    1,
    2,
    4,
    8
  ],
  "budgets": [
    1,
    2,
    4,
    8,
    16,
    32
  ],
  "search_repetitions": 3,
  "evaluation_seeds": 3,
  "epochs": 20,
  "lr": 0.001,
  "batch_size": 128,
  "optimizer": "Adam, PyTorch defaults, foreach=True",
  "initialization": "PyTorch default for conv and linear layers; GroupNorm affine scale=1 and offset=0",
  "device": "cpu",
  "threads": 1,
  "workers": 5,
  "per_layer_choices": "activation {ReLU, LeakyReLU(0.01), GELU} x normalization {none, LayerNorm} x kernel {3, 5}: 12 choices per layer",
  "selection_metric": "final-epoch training cross-entropy (eval mode) on the fixed 10,000-image training set; first occurrence wins ties",
  "primary_metric": "fresh-seed final training cross-entropy of the selected configuration (how well it fits the fixed training set)",
  "secondary_metric": "fresh-seed held-out test accuracy and cross-entropy (generalization diagnostic only)",
  "default": "LeakyReLU(0.01), LayerNorm, kernel 3 at every layer",
  "search": "uniform random without replacement, default included first; shared space (one choice repeated at every layer) is exhaustive after 12",
  "timing": "primary cost axis is cumulative process CPU seconds, including setup, training, and evaluation of the selection metric; measured elapsed durations are also retained",
  "lr_control_depths": [
    1,
    4,
    8
  ],
  "lr_control_values": [
    0.0003,
    0.001,
    0.003
  ],
  "horizon_control_depths": [
    1,
    4,
    8
  ],
  "horizon_control_epochs": 40,
  "controls": "LR-only selection on training loss at the fixed default architecture; 40-epoch reruns of default and B=32 selected architectures, without re-search",
  "uncertainty": "show all 3 search-repetition means and descriptive SD; average the 3 nested fresh training seeds within each repetition",
  "scope": "exploratory single-dataset, fixed-width plain CNN; objective is fitting a fixed training set, not generalization; no universal depth scaling claim",
  "calibration_decision": "An earlier MLP study (depth_hpo) was abandoned as unrealistic because depth hurt at width 16. For the CNN, two architecture choices (pooling after layers 1 and 3 and a 3x3 spatial head) were made from default-network pilot runs so that fitting improves with depth; pilots at 6,000/18,000 training images were reviewed on validation data, then the objective was changed to fitting a fixed 10,000-image set. Pilot on 10,000 images at 40 epochs: default training loss 0.436, 0.318, 0.139, 0.081, 0.094, 0.096 at depths 1, 2, 3, 4, 6, 8. 20 epochs retained for the main search to bound compute; 40 epochs is a control. Depths 3 and 6 were dropped before the main search (compute: about 75 CPU-seconds per fit under five-way contention) to keep the study to a few hours; depths 1, 2, 4, 8 retained."
}