{
  "version": "2.0-exploratory",
  "date": "2026-10-10",
  "dataset": "CIFAR-10 (official checksummed binary archive), 2 x 2 average pooled to 16 x 16",
  "split_seed": 20261010,
  "training_examples": 10000,
  "test_examples": 10000,
  "model": "residual CNN: fixed stem conv(3 -> 16 channels) + LeakyReLU + max-pool, then D residual blocks x -> x + conv3x3(16 channels) -> normalization -> activation, scaled by 1/sqrt(D); max-pool after block D/2; adaptive average pool to 2x2; linear classifier",
  "width": 16,
  "depths": [2, 4, 8, 16],
  "budgets": [1, 2, 4, 6, 8, 12, 16, 24],
  "eval_budgets": [1, 4, 8, 24],
  "search_repetitions": 4,
  "evaluation_seeds": 2,
  "epochs": 10,
  "lr": 0.003,
  "batch_size": 128,
  "optimizer": "Adam, PyTorch defaults, foreach=True",
  "initialization": "PyTorch defaults for conv and linear layers; norm layers unit scale, zero offset",
  "device": "cpu",
  "threads": 1,
  "workers": 6,
  "per_layer_choices": "activation {LeakyReLU(0.01), GELU, Tanh, ELU} x normalization {none, LayerNorm (one-group GroupNorm), BatchNorm} = 12 choices per block; kernel size fixed at 3, so all configurations have nearly identical parameter counts",
  "default": "LeakyReLU(0.01) and LayerNorm in every block",
  "arms": "norm_uniform (3 recipes), act_uniform (4), uniform (all 12 pairs), norm_mixed (per-block normalization, activation fixed to default, up to 8 candidates), act_mixed (per-block activation, normalization fixed to default, up to 8), joint_mixed (per-block activation and normalization, up to 24). Mixed arms contain only configurations with at least two distinct choices (plus the default as first element). Larger budgets are prefixes of the same pre-generated sequence.",
  "cap_joint": 24,
  "cap_part": 8,
  "selection_metric": "final-epoch training cross-entropy (eval mode) on the fixed 10,000-image training set, one training seed (1000 + repetition) shared by all candidates in a repetition; first occurrence wins ties",
  "primary_metric": "fresh-seed final training cross-entropy of the selected configuration (how well it fits the fixed training set)",
  "secondary_metric": "fresh-seed held-out test accuracy; test data are touched only in the evaluation stage",
  "timing": "cumulative process CPU seconds per fit, including setup, training and the training-set evaluation used for selection",
  "scope": "exploratory single-dataset, fixed-width residual CNN; objective is fitting a fixed training set, not generalization; no universal depth-scaling claim",
  "calibration_decision": "Pilot runs on the fixed training set only (4 recipes x depths 2, 4, 8, 16 x learning rates 0.001 and 0.003, 10 epochs, no failures; results/calibration_pilot.json). Learning rate 0.003 fitted better at nearly every depth and recipe, so it is used for everything. 10 epochs bounds compute. Pilot observation recorded before the main search: the default (LeakyReLU + LayerNorm) did not improve with depth, while Tanh + LayerNorm and GELU + BatchNorm did."
}
