From 0ff92e61dc49fa7dd6885814caed7a11bbb5ae0d Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 15:54:07 +0200 Subject: [PATCH 001/155] feat: add graph-based INT8 PTQ and QAT with verified x86 inference --- .github/workflows/ci.yml | 15 +- README.md | 8 +- docs/quantization.md | 126 ++++++++++++ docs/training-feature-validation.md | 4 + mini_trainer/modeling/quantization.py | 272 ++++++++++++++++++++++++++ pyproject.toml | 1 + tests/test_quantization.py | 170 ++++++++++++++++ uv.lock | 15 +- 8 files changed, 608 insertions(+), 3 deletions(-) create mode 100644 docs/quantization.md create mode 100644 mini_trainer/modeling/quantization.py create mode 100644 tests/test_quantization.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index eee7692..392a66b 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -2,7 +2,7 @@ name: CI on: push: - branches: ["master"] + branches: ["master", "quant"] pull_request: branches: ["master"] @@ -58,3 +58,16 @@ jobs: - run: uv python install ${{ matrix.python-version }} - name: Minimal installed-wheel checks run: bash dev/check-wheel.sh ${{ matrix.python-version }} + + quantization: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + - uses: astral-sh/setup-uv@v8.1.0 + with: + enable-cache: true + - run: uv sync --locked --extra cpu --extra quantization --python 3.13 + - name: QAT continuation and native INT8 inference + env: + OMP_NUM_THREADS: "1" + run: bash dev/check.sh test tests/test_quantization.py diff --git a/README.md b/README.md index de4dfac..5252748 100644 --- a/README.md +++ b/README.md @@ -132,4 +132,10 @@ with separate CPU and GPU profiles, visible summaries, and retained reproduction EMA (`--ema` / `ema=True`) is currently nonfunctional: classifier caches populated by evaluation can break later EMA updates. Leave it disabled. Enabling it emits a runtime warning; its API and checkpoint compatibility are retained, and repair is -deferred. See [known limitations](docs/roadmap.md). \ No newline at end of file +deferred. See [known limitations](docs/roadmap.md). + +## INT8 quantization + +An opt-in [PTQ and QAT Python API](docs/quantization.md) targets native x86 INT8 +inference. This is an initial backend increment; CPU float32 QAT, integer inference +and ordinary AMP are distinct capabilities. diff --git a/docs/quantization.md b/docs/quantization.md new file mode 100644 index 0000000..d1f8963 --- /dev/null +++ b/docs/quantization.md @@ -0,0 +1,126 @@ +# INT8 quantization: initial x86 backend + +This is an opt-in Python API for **static 8-bit weights and 8-bit activations**, +using TorchAO PT2E. It supports post-training calibration (PTQ) and +quantization-aware training (QAT). QAT uses fake quantization with float32 master +parameters/gradients; it does not promise integer backward computation or reduced +training memory. Converted inference executes native oneDNN integer Conv/Linear +kernels. This is separate from float16/bfloat16 AMP. + +Install the optional dependency in an explicitly selected backend environment: + +```bash +uv sync --extra cpu --extra quantization +# Existing CUDA environments: do not run a CPU sync; select their CUDA extra. +``` + +The first verified backend is x86 CPU with PyTorch 2.12 and TorchAO 0.17. +TorchAO is imported lazily. Ordinary training, prediction, checkpoint formats and +ONNX export are unchanged. The new API is in `mini_trainer.modeling.quantization`. + +## Calibration and inference + +```python +from mini_trainer.modeling.quantization import prepare_int8, load_int8 + +# model is a loaded floating-point mini_trainer model. All inputs below are +# float32 CPU batches AFTER the same preprocessing used for ordinary inference. +prepared = prepare_int8(model, example_batch) +with torch.no_grad(): + for images in training_calibration_batches: + prepared(images) +converted = prepared.convert() +converted.save( + "int8-model", + example_batch, + preprocessing={"recipe": "record the actual resize, scale and normalization"}, + calibration={"split": "train", "manifest_sha256": "record the actual manifest hash"}, +) +inference, coverage = load_int8("int8-model").lower(example_batch) +with torch.no_grad(): + scores = inference(example_batch) +``` + +Calibration must use training data, never held-out validation/test examples. +The caller supplies provenance; the API cannot infer the provenance of tensors. +The prepared PTQ graph is a calibration object, including when its mode is eval. +Convert it before evaluating held-out data. Conversion refuses unobserved or +nonfinite ranges, and does not modify the prepared model. + +Weights use symmetric per-channel int8; activations use affine per-tensor uint8. +Bias, normalization, score transforms and other non-linear operations may remain +floating point. The report includes the actual remaining operator inventory. +All captured Conv1d/Conv2d/Linear operations must receive weight and activation +annotations. Lowering fails if it cannot produce integer kernels or leaves +floating Conv/Linear kernels. It never labels a plain Q/DQ reference execution +as native integer inference. + +The bundle contains a reference `model.pt2` graph, checksum, input shape, class +metadata, structured output mapping, bit widths, dependency versions, +preprocessing/calibration provenance, and verified lowering coverage. Packing +is performed again on the deployment CPU. A reference graph alone is not an +accelerated runtime. Existing output directories are never overwritten. + +## Quantization-aware training + +```python +prepared = prepare_int8(model, example_batch, qat=True) +optimizer = torch.optim.AdamW(prepared.parameters(), lr=1e-4) +prepared.train() +for images, targets in training_batches: + optimizer.zero_grad() + loss = criterion(prepared(images), targets) + loss.backward() + optimizer.step() + +prepared.freeze_observers() # Optional: hold learned ranges fixed for later steps. +prepared.eval() # Evaluation does not update QAT ranges or BatchNorm. +with torch.no_grad(): + scores = prepared(validation_batch) +converted = prepared.convert() +``` + +Construct the optimizer **after** preparation. Save `prepared.state_dict()` and +the optimizer/scheduler/scaler states. Restore into an identically prepared model, +then restore optimizer state. Observer ranges, fake-quant flags and the explicit +freeze flag are part of the state. The recipe is checked on restoration. This is +not an ordinary `Classifier.build(weights=...)` checkpoint: automatic reconstruction +through `mt_train`/`mt_predict` remains a subsequent integration step. + +QAT runs can use `train_one_epoch` with an appropriate criterion, disabled EMA, +and no embedding-dependent regularizer. Captured graphs neither consume ambient +supervision nor populate `EmbeddingContext`. Train/eval switching covers dropout +and BatchNorm; arbitrary Python training branches are specialized by capture. +Autoregressive teacher-forcing/sampling requires a separate training integration +and is not currently a supported QAT claim. Capture failures propagate explicitly. + +## Coverage and next increments + +Inputs currently have a fixed captured batch and image shape. All batches must +match it; pad and slice final inference batches, or prepare a separate shape. +The original model's parameters, modes and caches are preserved by preparation. +Functional linears, weight parametrization, hierarchical aggregation and class +masks are included in capture; this does not rely on a backbone allowlist. + +Focused tests exercise flat, hierarchical, conditional and independent heads, +real gradients, exact controlled QAT/AdamW continuation, held-out-data isolation, +integer operator execution, checksums and artifact reload. A synthetic oracle +exercises QAT through the actual training loop and checks integer predictions. + +Still required: user-facing checkpoint/CLI integration, dynamic batch support, +MNIST/Blair reports, broader backbone/operator coverage, GPU quantization, +ONNX/runtime conversion, and lower-bit profiles. Model quality, artifact size, +memory and latency need measured comparisons; no speedup or quality benefit is +claimed by passing compatibility tests. EMA remains unsupported. + +Run focused checks without changing the installed environment: + +```bash +OMP_NUM_THREADS=1 bash dev/check.sh test tests/test_quantization.py +``` + +Backend references: [PT2E x86 quantization](https://docs.pytorch.org/ao/stable/pt2e_quantization/pt2e_quant_x86_inductor.html), +[QAT workflow](https://docs.pytorch.org/ao/stable/pt2e_quantization/pt2e_quant_qat.html). +Strict graph capture is intentional: the installed backend otherwise misses +functional-linears' source metadata. Explicit `lower_pt2e_quantized_to_x86` +provides native kernels without relying on a compiler silently optimizing Q/DQ. diff --git a/docs/training-feature-validation.md b/docs/training-feature-validation.md index 1d5314a..2985405 100644 --- a/docs/training-feature-validation.md +++ b/docs/training-feature-validation.md @@ -6,6 +6,10 @@ nonfunctional and excluded from these experiments; repair is deferred. ## Quantization is the primary implementation target +The initial [INT8 PTQ/QAT Python backend](quantization.md) is implemented on the +`quant` branch. Its CPU tests establish a training-to-integer-inference path; +user-facing checkpoint integration, other backends and quality studies remain open. + Deliver two distinct paths through the existing builders, checkpoint and export interfaces, with optional dependencies: diff --git a/mini_trainer/modeling/quantization.py b/mini_trainer/modeling/quantization.py new file mode 100644 index 0000000..d46902d --- /dev/null +++ b/mini_trainer/modeling/quantization.py @@ -0,0 +1,272 @@ +"""Opt-in PT2E INT8 calibration, QAT and x86 inference. + +Preprocessing stays outside the graph. Prepared graphs are training artifacts; +converted reference graphs are deployment artifacts and must be lowered before +claiming integer execution. Nothing in this module changes default training. +""" + +import copy +import hashlib +import json +import platform +from collections import Counter +from importlib.metadata import version +from pathlib import Path +from tempfile import TemporaryDirectory + +import torch +from torch import nn + +from .classifier import Classifier +from .context import EmbeddingContext, SupervisionContext +from .onnx import _flatten, _json_value, _structure + + +def _backend(): + try: + from torchao.quantization.pt2e import export_utils, quantize_pt2e + from torchao.quantization.pt2e.lowering import lower_pt2e_quantized_to_x86 + from torchao.quantization.pt2e.quantizer.x86_inductor_quantizer import ( + X86InductorQuantizer, + get_default_x86_inductor_quantization_config, + ) + except ImportError as error: + raise ImportError("INT8 quantization requires mini_trainer[quantization].") from error + return quantize_pt2e, export_utils, X86InductorQuantizer, get_default_x86_inductor_quantization_config, lower_pt2e_quantized_to_x86 + + +def _input(images): + if not isinstance(images, torch.Tensor) or images.ndim < 2 or images.shape[0] < 1: + raise ValueError("Supply a nonempty, preprocessed batch tensor.") + if images.device.type != "cpu" or images.dtype != torch.float32: + raise ValueError("The initial x86 INT8 profile requires CPU float32 inputs (no autocast).") + if not torch.isfinite(images).all(): + raise ValueError("Quantization inputs must be finite.") + + +def _weighted_nodes(graph): + return [ + n for n in graph.nodes if n.target in (torch.ops.aten.linear.default, torch.ops.aten.conv1d.default, torch.ops.aten.conv2d.default) + ] + + +def _graph_mode(graph, training): + utils = _backend()[1] + (utils._move_exported_model_to_train if training else utils._move_exported_model_to_eval)(graph) + # TorchAO switches BatchNorm math, but its exported counter increment stays + # specialized to training. Switch that increment too, at its graph source. + for node in graph.graph.nodes: + if node.target == torch.ops.aten.add_.Tensor: + target = node.args[0] + if isinstance(target, torch.fx.Node) and target.op == "get_attr" and str(target.target).endswith("num_batches_tracked"): + node.args = (target, int(training), *node.args[2:]) + graph.recompile() + + +class PreparedInt8(nn.Module): + """A private, captured model with observers or fake quantizers. + + Create optimizers AFTER preparation. Save this object's state_dict along with + optimizer/scheduler/scaler state; restore into the same preparation recipe. + QAT train/eval switching covers exported dropout and batchnorm only, not + arbitrary Python branches or ambient supervision/embedding contexts. + """ + + def __init__(self, graph, recipe): + super().__init__() + self.graph = graph + self.recipe = recipe + self.register_buffer("observers_frozen", torch.tensor(False)) + self.train(recipe["qat"]) + + def train(self, mode=True): + if not isinstance(mode, bool): + raise ValueError("training mode must be a bool") + if mode and not self.recipe["qat"]: + raise ValueError("PTQ preparation is for calibration; use qat=True for training.") + self.training = mode + if self.recipe["qat"]: + _graph_mode(self.graph, mode) + return self + + def freeze_observers(self): + """Keep learned ranges fixed during subsequent QAT steps.""" + self.observers_frozen.fill_(True) + + def forward(self, images): + _input(images) + if torch.is_autocast_enabled("cpu"): + raise ValueError("This QAT/calibration profile uses float32 without AMP.") + # FakeQuantize.eval() alone does not stop observers. Evaluation must not + # incorporate held-out examples into training/calibration ranges. + observers = [m for m in self.graph.modules() if hasattr(m, "observer_enabled")] + saved = [m.observer_enabled.clone() for m in observers] + if self.recipe["qat"] and (not self.training or self.observers_frozen): + for module in observers: + module.observer_enabled.zero_() + try: + return self.graph(images) + finally: + for module, enabled in zip(observers, saved): + module.observer_enabled.copy_(enabled) + + def get_extra_state(self): + return self.recipe + + def set_extra_state(self, state): + if state != self.recipe: + raise ValueError("Quantization checkpoint recipe differs; recreate the same model, shape and QAT configuration.") + + @torch.no_grad() + def convert(self): + """Convert an independent copy; keep the training model/optimizer usable.""" + observed = [] + for module in self.graph.modules(): + if hasattr(module, "min_val") and hasattr(module, "max_val"): + observed.append(module) + if not module.min_val.numel() or not torch.isfinite(module.min_val).all() or not torch.isfinite(module.max_val).all(): + raise ValueError("Every quantizer must observe finite training/calibration data before conversion.") + if not observed: + raise ValueError("No calibrated quantization observers found.") + backend, *_ = _backend() + graph = copy.deepcopy(self.graph) + if self.recipe["qat"]: + _graph_mode(graph, False) + converted = backend.convert_pt2e(graph) + return Int8Model(converted, copy.deepcopy(self.recipe)) + + +def prepare_int8(model: nn.Module, example_input: torch.Tensor, *, qat=False): + """Capture the actual model for static W8A8 PTQ or QAT, at a fixed input shape. + + Weights are symmetric per-channel int8; activations are affine per-tensor + uint8. Bias, normalization and unsupported non-linear operations stay float. + Graph capture/backend errors propagate. All captured Conv1d/Conv2d/Linear + operations must receive weight AND activation quantization annotations. + """ + _input(example_input) + if not isinstance(qat, bool): + raise TypeError("qat must be a bool") + if EmbeddingContext.active() or SupervisionContext.get() is not None: + raise RuntimeError("Prepare outside embedding/supervision contexts.") + backend, _, quantizer_cls, config_factory, _ = _backend() + while isinstance(model, (nn.DataParallel, nn.parallel.DistributedDataParallel)) or hasattr(model, "_orig_mod"): + model = model._orig_mod if hasattr(model, "_orig_mod") else model.module + with torch.random.fork_rng(devices=[]): + model = copy.deepcopy(model).cpu().float().eval() + for module in model.modules(): + if isinstance(module, Classifier): + module._dirty_cache.clear() + # Populate masks and immutable evaluation caches outside strict capture. + with torch.no_grad(): + outputs = model(example_input) + classifiers = [ + {"module": name, "metadata": _json_value(m.metadata)} for name, m in model.named_modules() if isinstance(m, Classifier) + ] + model.train(qat) + # Strict capture retains functional-op provenance needed by TorchAO's + # x86 quantizer. Non-strict export silently misses functional linears. + graph = torch.export.export(model, (example_input,), strict=True).module() + quantizer = quantizer_cls().set_global(config_factory(is_qat=qat)) + graph = (backend.prepare_qat_pt2e if qat else backend.prepare_pt2e)(graph, quantizer) + nodes = _weighted_nodes(graph.graph) + missing = [] + for node in nodes: + annotation = node.meta.get("quantization_annotation") + if annotation is None or sum(spec is not None for spec in annotation.input_qspec_map.values()) < 2: + missing.append(node.name) + if not nodes or missing: + raise ValueError( + f"INT8 requires quantized weights and activations for captured Conv/Linear operations; missing: {missing or 'all'}" + ) + recipe = { + "format_version": 1, + "backend": "x86_inductor", + "qat": qat, + "torch": str(torch.__version__), + "torchao": version("torchao"), + "input_shape": list(example_input.shape), + "weight_bits": 8, + "activation_bits": 8, + "weight_dtype": "int8", + "activation_dtype": "uint8", + "classifiers": classifiers, + "output_structure": _structure(outputs, iter(f"output_{i}" for i in range(len(_flatten(outputs))))), + "weighted_operations": [{"name": n.name, "operator": str(n.target)} for n in nodes], + } + return PreparedInt8(graph, recipe) + + +class Int8Model: + """Converted reference graph with explicit x86 lowering and portable storage.""" + + def __init__(self, graph, recipe): + self.graph = graph + self.recipe = recipe + + @torch.no_grad() + def lower(self, example_input): + """Return an inference graph with verified oneDNN INT8 Conv/Linear kernels. + + Reference Q/DQ execution is not integer arithmetic. Require the lowered + graph to contain integer kernels and no residual floating Conv/Linear. + """ + _input(example_input) + if platform.machine().lower() not in ("x86_64", "amd64"): + raise RuntimeError("This quantization backend requires x86 CPU hardware.") + lowered = _backend()[4](copy.deepcopy(self.graph), (example_input,)) + operators = Counter(str(n.target) for n in lowered.graph.nodes if n.op == "call_function") + integer = {op: count for op, count in operators.items() if op.startswith("onednn.q") and "pointwise" in op} + floating = [ + op + for op in operators + if op.startswith(("aten.linear.", "aten.convolution.", "aten.conv1d.", "aten.conv2d.", "aten.mm.", "aten.addmm.")) + ] + if not integer or floating: + raise RuntimeError(f"Incomplete INT8 lowering: integer kernels={integer}, remaining floating kernels={floating}") + actual = lowered(example_input) + expected = self.graph(example_input) + torch.testing.assert_close(actual, expected, rtol=1e-4, atol=1e-5) + return lowered, {"integer_kernels": integer, "operators": dict(operators)} + + @torch.no_grad() + def save(self, output_dir, example_input, *, preprocessing, calibration): + """Save a reference .pt2 graph and manifest after verifying native lowering. + + Supply JSON provenance for preprocessing and training-only calibration. + Destination must be new. Packed oneDNN weights are rebuilt on load. + """ + _input(example_input) + destination = Path(output_dir).absolute() + if destination.exists(): + raise FileExistsError(destination) + _, coverage = self.lower(example_input) + manifest = {"recipe": self.recipe, "preprocessing": preprocessing, "calibration": calibration, "lowering": coverage} + json.dumps(manifest, allow_nan=False) + destination.parent.mkdir(parents=True, exist_ok=True) + with TemporaryDirectory(prefix=".int8-", dir=destination.parent) as temporary: + bundle = Path(temporary) / "bundle" + bundle.mkdir() + path = bundle / "model.pt2" + program = torch.export.export(self.graph, (example_input,)) + torch.export.save(program, path) + restored = torch.export.load(path).module() + torch.testing.assert_close(restored(example_input), self.graph(example_input), rtol=0, atol=0) + manifest["sha256"] = hashlib.sha256(path.read_bytes()).hexdigest() + manifest["artifact_bytes"] = path.stat().st_size + (bundle / "manifest.json").write_text(json.dumps(manifest, indent=2, allow_nan=False) + "\n") + bundle.rename(destination) + return destination + + +def load_int8(output_dir): + """Load a saved reference graph; call .lower(example_input) for integer execution.""" + _backend() # Register decomposed quantization operators before torch.export.load. + directory = Path(output_dir) + manifest = json.loads((directory / "manifest.json").read_text()) + if manifest["recipe"]["format_version"] != 1: + raise ValueError("Unsupported INT8 bundle version") + path = directory / "model.pt2" + if hashlib.sha256(path.read_bytes()).hexdigest() != manifest["sha256"]: + raise ValueError("INT8 artifact checksum mismatch") + return Int8Model(torch.export.load(path).module(), manifest["recipe"]) diff --git a/pyproject.toml b/pyproject.toml index 1e56bfd..af5464e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -94,6 +94,7 @@ all = [ "mini_trainer[transformers]", "mini_trainer[export]", ] +quantization = ["torchao>=0.17,<0.18"] export = ["onnx>=1.17", "onnxscript>=0.3", "onnxruntime>=1.20"] minimal = [] notebook = [ diff --git a/tests/test_quantization.py b/tests/test_quantization.py new file mode 100644 index 0000000..c44a4e7 --- /dev/null +++ b/tests/test_quantization.py @@ -0,0 +1,170 @@ +"""Real W8A8 training, continuation and native integer inference regressions.""" + +import copy +import importlib.util +import json +from unittest.mock import Mock + +import pytest +import torch +from torch.utils.data import DataLoader, TensorDataset + +from mini_trainer.hierarchical.model import ConditionalClassifier, HierarchicalClassifier, IndependentClassifier +from mini_trainer.modeling import Classifier +from mini_trainer.modeling.quantization import load_int8, prepare_int8 +from mini_trainer.trainer import train_one_epoch +from tests.test_checkpoint_contract import assert_state_equal + +pytestmark = pytest.mark.skipif(importlib.util.find_spec("torchao") is None, reason="Install mini_trainer[quantization]") + + +def model(head=Classifier, normalized=True): + kwargs = {} + if head != Classifier: + kwargs["sparse_masks"] = [torch.tensor([0, 0, 1, 1])] + return head(in_features=8, out_features=4, hidden=False, normalized=normalized, **kwargs) + + +@pytest.mark.parametrize("qat", [False, True]) +@pytest.mark.parametrize("head", [Classifier, HierarchicalClassifier, ConditionalClassifier, IndependentClassifier]) +def test_heads_quantize_functional_parametrized_and_masked_linears(tmp_path, qat, head): + torch.manual_seed(42) + original = model(head) + original.set_active_features([0, 2, 3]) + original.train() + x = torch.randn(4, 8) + before = copy.deepcopy(original.state_dict()) + prepared = prepare_int8(original, x, qat=qat) + assert original.training + assert_state_equal(original.state_dict(), before) + output = prepared(x) + if qat: + sum(t.square().mean() for t in (output if isinstance(output, list) else [output])).backward() + assert any(p.grad is not None and p.grad.abs().sum() > 0 for p in prepared.parameters()) + converted = prepared.convert() + lowered, coverage = converted.lower(x) + assert coverage["integer_kernels"] + assert any(t.dtype == torch.int8 for t in converted.graph.buffers()) + with torch.no_grad(), torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CPU]) as profile: + expected = lowered(x) + assert "onednn::qlinear_pointwise" in {e.key for e in profile.key_averages()} + path = converted.save(tmp_path / "int8", x, preprocessing={"input": "embeddings"}, calibration={"split": "train", "seed": 42}) + reloaded, _ = load_int8(path).lower(x) + with torch.no_grad(): + torch.testing.assert_close(reloaded(x), expected, rtol=0, atol=0) + manifest = json.loads((path / "manifest.json").read_text()) + assert manifest["recipe"]["weight_bits"] == manifest["recipe"]["activation_bits"] == 8 + with pytest.raises(FileExistsError): + converted.save(path, x, preprocessing={}, calibration={}) + + +@pytest.mark.parametrize("normalized", [False, True]) +def test_qat_resume_and_evaluation_do_not_recalibrate(tmp_path, normalized): + torch.manual_seed(42) + original = model(normalized=normalized) + x = torch.randn(4, 8) + labels = torch.tensor([0, 1, 2, 3]) + prepared = prepare_int8(original, x, qat=True) + optimizer = torch.optim.AdamW(prepared.parameters(), lr=0.01) + + def step(p, opt): + p.train() + opt.zero_grad() + torch.nn.functional.cross_entropy(p(x), labels).backward() + opt.step() + + step(prepared, optimizer) + prepared.eval() + before = copy.deepcopy(prepared.state_dict()) + with torch.no_grad(): + prepared(x * 1000) # Held-out outliers must not influence ranges or BN. + assert_state_equal(prepared.state_dict(), before) + prepared.train() + prepared.freeze_observers() + checkpoint = tmp_path / "checkpoint.pt" + torch.save({"model": prepared.state_dict(), "optimizer": optimizer.state_dict()}, checkpoint) + step(prepared, optimizer) + restored = prepare_int8(original, x, qat=True) + restored_optimizer = torch.optim.AdamW(restored.parameters(), lr=0.01) + state = torch.load(checkpoint, weights_only=True) + restored.load_state_dict(state["model"]) + restored_optimizer.load_state_dict(state["optimizer"]) + step(restored, restored_optimizer) + assert_state_equal(restored.state_dict(), prepared.state_dict()) + assert_state_equal(restored_optimizer.state_dict(), optimizer.state_dict()) + expected, _ = prepared.convert().lower(x) + actual, _ = restored.convert().lower(x) + with torch.no_grad(): + torch.testing.assert_close(actual(x), expected(x), rtol=0, atol=0) + + +def test_synthetic_qat_uses_training_loop_and_integer_inference(): + torch.manual_seed(42) + + # Two independent factors give an exact oracle; disjoint nuisance samples. + def samples(seed): + generator = torch.Generator().manual_seed(seed) + labels = torch.arange(4).repeat(8) + values = torch.randn(32, 8, generator=generator) * 0.02 + values[:, 0] += ((labels // 2) * 2 - 1) * 2 + values[:, 1] += ((labels % 2) * 2 - 1) * 2 + return values.reshape(32, 8, 1, 1), labels + + images, labels = samples(1) + heldout, expected = samples(2) + prepared = prepare_int8(torch.nn.Sequential(torch.nn.Flatten(), model(normalized=False)), images[:8], qat=True) + optimizer = torch.optim.SGD(prepared.parameters(), lr=0.2, momentum=0.9) + scaler = torch.amp.GradScaler("cpu", enabled=False) + scheduler = torch.optim.lr_scheduler.StepLR(optimizer, 1000) + teacher = Mock() + teacher.teach.return_value = torch.tensor(0.0) + logger = Mock() + logger.status.return_value = "QAT oracle" + loader = DataLoader(TensorDataset(images, labels), batch_size=8) + for epoch in range(8): + train_one_epoch(prepared, teacher, torch.nn.CrossEntropyLoss(), optimizer, scaler, scheduler, loader, epoch, logger) + assert scheduler.last_epoch == 32 + inference, coverage = prepared.convert().lower(heldout[:8]) + with torch.no_grad(): + predictions = torch.cat([inference(batch).argmax(1) for batch in heldout.split(8)]) + assert torch.equal(predictions, expected) + assert coverage["integer_kernels"]["onednn.qlinear_pointwise.default"] == 1 + + +def test_refuse_uncalibrated_and_empty_quantization(): + x = torch.randn(4, 8) + prepared = prepare_int8(model(), x) + with pytest.raises(ValueError, match="observe finite"): + prepared.convert() + with pytest.raises(ValueError, match="missing"): + prepare_int8(torch.nn.Identity(), x) + with pytest.raises(ValueError, match="CPU float32"): + prepare_int8(model(), x.double()) + + +@pytest.mark.parametrize("qat", [False, True]) +def test_convolution_batchnorm_hidden_head_and_artifact_integrity(tmp_path, qat): + torch.manual_seed(42) + original = torch.nn.Sequential( + torch.nn.Conv2d(3, 8, 3, padding=1), + torch.nn.BatchNorm2d(8), + torch.nn.ReLU(), + torch.nn.AdaptiveAvgPool2d(1), + torch.nn.Flatten(), + Classifier(8, 4, hidden=6, droprate=0.1, normalized=True), + ) + x = torch.randn(4, 3, 8, 8) + prepared = prepare_int8(original, x, qat=qat) + if qat: + prepared(x).square().mean().backward() + else: + with torch.no_grad(): + prepared(x) + converted = prepared.convert() + _, coverage = converted.lower(x) + assert any("qconv_pointwise" in op for op in coverage["integer_kernels"]) + assert sum(count for op, count in coverage["integer_kernels"].items() if "qlinear" in op) == 2 + path = converted.save(tmp_path / "bundle", x, preprocessing={}, calibration={"split": "train"}) + (path / "model.pt2").write_bytes(b"corrupted") + with pytest.raises(ValueError, match="checksum"): + load_int8(path) diff --git a/uv.lock b/uv.lock index ed4114f..3d920fb 100644 --- a/uv.lock +++ b/uv.lock @@ -1490,6 +1490,9 @@ notebook = [ { name = "ipywidgets" }, { name = "pyremotedata" }, ] +quantization = [ + { name = "torchao" }, +] recommended = [ { name = "biopython" }, { name = "pyarrow" }, @@ -1546,6 +1549,7 @@ requires-dist = [ { name = "torch", marker = "extra == 'cu126'", specifier = ">=2.11", index = "https://download.pytorch.org/whl/cu126", conflict = { package = "mini-trainer", extra = "cu126" } }, { name = "torch", marker = "extra == 'cu130'", specifier = ">=2.11", index = "https://download.pytorch.org/whl/cu130", conflict = { package = "mini-trainer", extra = "cu130" } }, { name = "torch", marker = "extra == 'cu132'", specifier = ">=2.11", index = "https://download.pytorch.org/whl/cu132", conflict = { package = "mini-trainer", extra = "cu132" } }, + { name = "torchao", marker = "extra == 'quantization'", specifier = ">=0.17,<0.18" }, { name = "torchvision" }, { name = "torchvision", marker = "extra == 'cpu'", index = "https://download.pytorch.org/whl/cpu", conflict = { package = "mini-trainer", extra = "cpu" } }, { name = "torchvision", marker = "extra == 'cu126'", index = "https://download.pytorch.org/whl/cu126", conflict = { package = "mini-trainer", extra = "cu126" } }, @@ -1555,7 +1559,7 @@ requires-dist = [ { name = "transformers", marker = "extra == 'transformers'" }, { name = "wandb", marker = "extra == 'recommended'" }, ] -provides-extras = ["recommended", "all", "export", "minimal", "notebook", "bioclip", "transformers", "timm", "cpu", "cu126", "cu130", "cu132"] +provides-extras = ["recommended", "all", "quantization", "export", "minimal", "notebook", "bioclip", "transformers", "timm", "cpu", "cu126", "cu130", "cu132"] [package.metadata.requires-dev] dev = [ @@ -3619,6 +3623,15 @@ wheels = [ { url = "https://download-r2.pytorch.org/whl/cu132/torch-2.12.0%2Bcu132-cp314-cp314t-win_amd64.whl", hash = "sha256:b11be4c6b4f3621811152ee2dd211103c348f02ad4d76ab62b020efe1ca9f78c", upload-time = "2026-05-13T00:24:22Z" }, ] +[[package]] +name = "torchao" +version = "0.17.0" +source = { registry = "https://pypi.org/simple" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/32/fe/a4036a8e80fa800c92dbcbf75f541cd4c106248b6b579db6dab1800f616a/torchao-0.17.0-cp310-abi3-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:87a418ce0ec064a821ceab83c921b501acef0ce9a6ccd1be358fcd16c3ae8c58", size = 3206172, upload-time = "2026-03-30T22:25:52.974Z" }, + { url = "https://files.pythonhosted.org/packages/c9/37/ef37ca885265e5f79a168616767dd416a3cea1cc3b28bb6b503ce4a5b652/torchao-0.17.0-py3-none-any.whl", hash = "sha256:02eba449036715b9ae784fbaa1a6f97994bb7b0421ce92d1d5d1c08e5bd6d349", size = 1200680, upload-time = "2026-03-30T22:25:54.457Z" }, +] + [[package]] name = "torchvision" version = "0.27.0" From b5d6df202ee94017119b015e611912fa3be36bd9 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 21:33:50 +0200 Subject: [PATCH 002/155] fix: handle spatial INT8 layouts and omit calibration images from exports --- mini_trainer/modeling/quantization.py | 19 +++++++++++++++++++ tests/test_quantization.py | 5 +++-- 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/mini_trainer/modeling/quantization.py b/mini_trainer/modeling/quantization.py index d46902d..0bc9503 100644 --- a/mini_trainer/modeling/quantization.py +++ b/mini_trainer/modeling/quantization.py @@ -215,6 +215,13 @@ def lower(self, example_input): if platform.machine().lower() not in ("x86_64", "amd64"): raise RuntimeError("This quantization backend requires x86 CPU hardware.") lowered = _backend()[4](copy.deepcopy(self.graph), (example_input,)) + # oneDNN returns channels-last activations. PT2E lowering decomposes + # flatten to view before that layout change, which fails on spatial + # outputs. Match flatten's copy-if-needed semantics in the lowered graph. + for node in lowered.graph.nodes: + if node.target == torch.ops.aten.view.default: + node.target = torch.ops.aten.reshape.default + lowered.recompile() operators = Counter(str(n.target) for n in lowered.graph.nodes if n.op == "call_function") integer = {op: count for op, count in operators.items() if op.startswith("onednn.q") and "pointwise" in op} floating = [ @@ -249,6 +256,18 @@ def save(self, output_dir, example_input, *, preprocessing, calibration): bundle.mkdir() path = bundle / "model.pt2" program = torch.export.export(self.graph, (example_input,)) + # torch.export.save otherwise embeds the real calibration batch. + # Rebuild without example inputs: deployment needs shapes, not images. + program = torch.export.ExportedProgram( + root=program.graph_module, + graph=program.graph, + graph_signature=program.graph_signature, + state_dict=program.state_dict, + range_constraints=program.range_constraints, + module_call_graph=program.module_call_graph, + constants=program.constants, + verifiers=program.verifiers, + ) torch.export.save(program, path) restored = torch.export.load(path).module() torch.testing.assert_close(restored(example_input), self.graph(example_input), rtol=0, atol=0) diff --git a/tests/test_quantization.py b/tests/test_quantization.py index c44a4e7..0891d13 100644 --- a/tests/test_quantization.py +++ b/tests/test_quantization.py @@ -49,6 +49,7 @@ def test_heads_quantize_functional_parametrized_and_masked_linears(tmp_path, qat expected = lowered(x) assert "onednn::qlinear_pointwise" in {e.key for e in profile.key_averages()} path = converted.save(tmp_path / "int8", x, preprocessing={"input": "embeddings"}, calibration={"split": "train", "seed": 42}) + assert torch.export.load(path / "model.pt2").example_inputs is None reloaded, _ = load_int8(path).lower(x) with torch.no_grad(): torch.testing.assert_close(reloaded(x), expected, rtol=0, atol=0) @@ -149,9 +150,9 @@ def test_convolution_batchnorm_hidden_head_and_artifact_integrity(tmp_path, qat) torch.nn.Conv2d(3, 8, 3, padding=1), torch.nn.BatchNorm2d(8), torch.nn.ReLU(), - torch.nn.AdaptiveAvgPool2d(1), + torch.nn.AdaptiveAvgPool2d(2), torch.nn.Flatten(), - Classifier(8, 4, hidden=6, droprate=0.1, normalized=True), + Classifier(32, 4, hidden=6, droprate=0.1, normalized=True), ) x = torch.randn(4, 3, 8, 8) prepared = prepare_int8(original, x, qat=qat) From 9965e3d077ae4e04ebe6c6bf209789fdc3c32827 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 21:34:21 +0200 Subject: [PATCH 003/155] perf: batch cached reads and pin CUDA transfer batches --- dev/benchmarks/loader.py | 77 +++++++++++++++++++++++++++++++++++++ mini_trainer/data/io.py | 23 +++++++++++ mini_trainer/data/loader.py | 58 +++++++++++++++++++++++++--- tests/utils/test_loader.py | 50 ++++++++++++++++++++++++ 4 files changed, 203 insertions(+), 5 deletions(-) create mode 100644 dev/benchmarks/loader.py diff --git a/dev/benchmarks/loader.py b/dev/benchmarks/loader.py new file mode 100644 index 0000000..631debe --- /dev/null +++ b/dev/benchmarks/loader.py @@ -0,0 +1,77 @@ +"""Measure scalar versus batched cached loading without changing data or sampling.""" + +import json +import statistics +import time +from argparse import ArgumentParser + +import torch +from torch.utils.data import DataLoader, Dataset + +from mini_trainer.data.io import LazyDataset +from mini_trainer.data.loader import get_dataloader + + +class ScalarFetch(Dataset): + """The previous per-sample fetch contract, deliberately without __getitems__.""" + + def __init__(self, dataset): + self.dataset = dataset + + def __len__(self): + return len(self.dataset) + + def __getitem__(self, index): + return self.dataset[index] + + +def run(samples=2048, size=64, batch_size=64, repeats=5): + generator = torch.Generator().manual_seed(42) + images = torch.randint(0, 256, (samples, 3, size, size), dtype=torch.uint8, generator=generator) + dataset = LazyDataset(lambda item: (images[item[0]], torch.tensor(item[0])), (list(range(samples)),), cache="cpu") + loaders = { + "scalar": DataLoader(ScalarFetch(dataset), batch_size=batch_size, num_workers=0), + "batched": get_dataloader(dataset, "val", batch_size, 0, False, torch.device("cpu")), + } + for expected, actual in zip(loaders["scalar"], loaders["batched"], strict=True): + torch.testing.assert_close(actual, expected, rtol=0, atol=0) + timings = {name: [] for name in loaders} + for trial in range(repeats + 1): + # Alternate order to avoid consistently giving one variant warm caches. + for name in list(loaders) if trial % 2 else list(reversed(loaders)): + started = time.perf_counter() + count = sum(len(batch[0]) for batch in loaders[name]) + elapsed = time.perf_counter() - started + assert count == samples + if trial: + timings[name].append(elapsed) + return { + "samples": samples, + "shape": [3, size, size], + "dtype": "uint8", + "batch_size": batch_size, + "workers": 0, + "threads": torch.get_num_threads(), + "identical_batches": True, + "scope": "cached CPU loader iteration; excludes cache construction, preprocessing, H2D and model compute", + "seconds": timings, + "median_samples_per_second": {name: samples / statistics.median(values) for name, values in timings.items()}, + "speedup": statistics.median(timings["scalar"]) / statistics.median(timings["batched"]), + } + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--samples", type=int, default=2048) + parser.add_argument("--size", type=int, default=64) + parser.add_argument("--batch-size", type=int, default=64) + parser.add_argument("--repeats", type=int, default=5) + args = parser.parse_args() + if min(vars(args).values()) < 1: + parser.error("All sizes and repeat counts must be positive") + torch.set_num_threads(1) + print(json.dumps(run(**vars(args)), indent=2)) + + +if __name__ == "__main__": + main() diff --git a/mini_trainer/data/io.py b/mini_trainer/data/io.py index 1517822..5a582e5 100644 --- a/mini_trainer/data/io.py +++ b/mini_trainer/data/io.py @@ -339,6 +339,16 @@ def _infer_numeric_dtype(seq) -> Any: return object +class _FetchedBatch(list): + """Sample views for standard collators, with the already-stacked batch attached.""" + + def __init__(self, data): + self.data = data + # unbind produces views, not sample copies. An actual list preserves + # torch.stack compatibility in external DataLoaders' default collators. + super().__init__(data.unbind(0) if isinstance(data, torch.Tensor) else zip(*(value.unbind(0) for value in data))) + + class LazyDataset(torch.utils.data.Dataset): """A general lazy dataset which calls func on items to obtain the image (and label) when needed. @@ -501,6 +511,19 @@ def _write(): def __len__(self): return len(self.items[0]) + def __getitems__(self, indices): + # PyTorch's batched fetch protocol: one gather for cached tensors, or + # one stacking pass for decoded samples. Keep scalar indexing unchanged. + if self._cache_mode in (CACHE_MODE.CPU, CACHE_MODE.CUDA): + tensors = self._ram_cache.tensors + index = torch.as_tensor(indices, dtype=torch.long, device=tensors[0].device) + index = torch.where(index < 0, index + len(self), index) + # index_select copies whole rows; generic advanced indexing is much + # slower for uint8 image batches on CPU. Results own their storage. + data = tuple(tensor.index_select(0, index) for tensor in tensors) + return _FetchedBatch(data[0] if self._ram_was_single_tensor else data) + return _FetchedBatch(self[indices]) + def __getitem__(self, index): match self._cache_mode: case CACHE_MODE.NONE: diff --git a/mini_trainer/data/loader.py b/mini_trainer/data/loader.py index 806638f..6ce3018 100644 --- a/mini_trainer/data/loader.py +++ b/mini_trainer/data/loader.py @@ -2,7 +2,7 @@ import numpy as np import torch -from torch.utils.data import BatchSampler, DataLoader, RandomSampler, SequentialSampler +from torch.utils.data import BatchSampler, DataLoader, RandomSampler, SequentialSampler, default_collate from torch.utils.data.distributed import DistributedSampler from mini_trainer import get_logger @@ -12,6 +12,7 @@ from .io import ( CACHE_MODE, LazyDataset, + _FetchedBatch, guess_cache_mode, make_read_and_resize_fn, ) @@ -67,8 +68,23 @@ def __call__(self, x: str) -> torch.Tensor: return self.hook(self.reader(x)) +def _collate_batch(samples): + if isinstance(samples, _FetchedBatch): + # Match default_collate's tuple-to-list convention for (image, label). + return list(samples.data) if isinstance(samples.data, tuple) else samples.data + return default_collate(samples) + + def get_dataloader( # noqa: D103 - dataset: torch.utils.data.Dataset, mode: str, batch_size: int, num_workers: int, pin_memory: bool, device: torch.device + dataset: torch.utils.data.Dataset, + mode: str, + batch_size: int, + num_workers: int, + pin_memory: bool, + device: torch.device, + *, + prefetch_factor: int | None = None, + multiprocessing_context: str | None = None, ): assert isinstance(mode, str) if mode.strip().lower() == "train": @@ -84,15 +100,20 @@ def get_dataloader( # noqa: D103 base_sampler = RandomSampler(dataset) if shuffle else SequentialSampler(dataset) # type: ignore mp_context = None + if num_workers > 0 and multiprocessing_context is not None: + mp_context = multiprocessing_context + sampler = BatchSampler(base_sampler, batch_size=batch_size, drop_last=drop_last) return DataLoader( dataset, batch_sampler=sampler, + collate_fn=_collate_batch, num_workers=num_workers, pin_memory=pin_memory, persistent_workers=num_workers > 0, multiprocessing_context=mp_context, + prefetch_factor=prefetch_factor if num_workers > 0 else None, ) @@ -108,6 +129,8 @@ def get_dataset_dataloader( # noqa: D103 dtype: torch.dtype = torch.float32, cache: CACHE_MODE | str | int | None = None, multilabel: bool = False, + prefetch_factor: int | None = None, + multiprocessing_context: str | None = None, hook: Callable[[torch.Tensor], torch.Tensor] | None = None, ): resize_size = _normalize_resize_size(resize_size, error_suffix=".") @@ -144,8 +167,22 @@ def get_dataset_dataloader( # noqa: D103 elif num_workers is None: num_workers = _default_worker_count(16) - pin_memory = cache not in [CACHE_MODE.CUDA, CACHE_MODE.CPU] - loaders = [get_dataloader(dataset, mode, batch_size, num_workers, pin_memory, device) for mode, dataset in zip(modes, datasets)] + # A gather from a pinned cache allocates an unpinned result. Pin the actual + # CPU batch before asynchronous H2D, including for the CPU cache path. + pin_memory = device.type == "cuda" and cache is not CACHE_MODE.CUDA + loaders = [ + get_dataloader( + dataset, + mode, + batch_size, + num_workers, + pin_memory, + device, + prefetch_factor=prefetch_factor, + multiprocessing_context=multiprocessing_context, + ) + for mode, dataset in zip(modes, datasets) + ] return datasets, loaders @@ -159,6 +196,8 @@ def get_inference_dataloader( # noqa: D103 device: torch.device | str = torch.device("cpu"), dtype: torch.dtype = torch.float32, hook: Callable[[torch.Tensor], torch.Tensor] | None = None, + prefetch_factor: int | None = None, + multiprocessing_context: str | None = None, **kwargs, ): resize_size = _normalize_resize_size(resize_size) @@ -177,6 +216,15 @@ def get_inference_dataloader( # noqa: D103 if num_workers is None: num_workers = _default_worker_count(32) - loader = get_dataloader(dataset, "test", batch_size, num_workers, False, device) + loader = get_dataloader( + dataset, + "test", + batch_size, + num_workers, + device.type == "cuda", + device, + prefetch_factor=prefetch_factor, + multiprocessing_context=multiprocessing_context, + ) return dataset, loader diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index 8c937c1..715ce67 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -154,3 +154,53 @@ def test_distributed_loader_retains_spawn_and_sampler(monkeypatch): assert isinstance(loader.batch_sampler.sampler, torch.utils.data.DistributedSampler) assert loader.batch_sampler.sampler.num_replicas == 2 assert loader.batch_sampler.drop_last + + +@pytest.mark.parametrize("cache", ["none", "cpu"]) +@pytest.mark.parametrize("workers", [0, 1]) +def test_batched_fetch_matches_default_collation(metadata, cache, workers): + dataset = data_io.LazyDataset( + PathLabelProcessor(data_io.make_read_and_resize_fn((4, 4), torch.device("cpu"), torch.uint8), None, False), + (metadata["path"], metadata["class"]), + cache=cache, + ) + reference = torch.utils.data.DataLoader( + dataset, batch_size=2, num_workers=workers, multiprocessing_context="spawn" if workers else None + ) + optimized = data_loader.get_dataloader( + dataset, "val", 2, workers, False, torch.device("cpu"), multiprocessing_context="spawn", prefetch_factor=1 + ) + for expected, actual in zip(reference, optimized, strict=True): + assert isinstance(actual, list) and len(actual) == 2 + torch.testing.assert_close(actual, expected, rtol=0, atol=0) + if workers == 0: + batch = dataset.__getitems__([4, 1, 1]) + direct = data_loader._collate_batch(batch) + assert direct[0] is batch.data[0] + assert direct[1].tolist() == [4, 1, 1] + + +def test_inference_batched_fetch_keeps_tensor_output(metadata): + dataset, loader = get_inference_dataloader(metadata["path"], resize_size=4, batch_size=2, num_workers=0) + batches = list(loader) + assert all(isinstance(batch, torch.Tensor) for batch in batches) + torch.testing.assert_close(torch.cat(batches), dataset[list(range(5))], rtol=0, atol=0) + external = torch.utils.data.DataLoader(dataset, batch_size=2) + torch.testing.assert_close(torch.cat(list(external)), torch.cat(batches), rtol=0, atol=0) + + +@pytest.mark.parametrize("cache", ["none", "cpu"]) +def test_cuda_transfer_batches_are_pinned(metadata, cache): + import os + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to validate pinned CUDA transfer batches") + if not torch.cuda.is_available(): + pytest.fail("CUDA checks requested but no CUDA device is accessible") + device = torch.device("cuda:0") + _, loaders = get_dataset_dataloader(metadata, resize_size=4, modes=("val",), cache=cache, batch_size=2, num_workers=0, device=device) + images, labels = next(iter(loaders[0])) + assert images.is_pinned() and labels.is_pinned() + torch.testing.assert_close(images.to(device, non_blocking=True).cpu(), images) + _, inference = get_inference_dataloader(metadata["path"], resize_size=4, batch_size=2, num_workers=0, device=device) + assert next(iter(inference)).is_pinned() From d602725f27c9c848d44c2dc21fd1542aa6115113 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 21:37:51 +0200 Subject: [PATCH 004/155] bench: compare recorded float checkpoints with INT8 PTQ and QAT Verify baseline and dataset provenance, calibrate on training samples, and evaluate reloaded integer artifacts on held-out synthetic, MNIST, and Blair data. Preserve per-level predictions and coverage for review. --- dev/benchmarks/quantization.py | 210 +++++++++++++++++++++++++++++++++ 1 file changed, 210 insertions(+) create mode 100644 dev/benchmarks/quantization.py diff --git a/dev/benchmarks/quantization.py b/dev/benchmarks/quantization.py new file mode 100644 index 0000000..d05bac9 --- /dev/null +++ b/dev/benchmarks/quantization.py @@ -0,0 +1,210 @@ +"""Compare a recorded float checkpoint with PTQ and short QAT on training-only data.""" + +import hashlib +import json +import random +import subprocess +import time +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np +import torch +from torch.utils.data import DataLoader, TensorDataset + +from mini_trainer.data import get_inference_dataloader +from mini_trainer.hierarchical.loss import MultiLevelWeightedCrossEntropyLoss +from mini_trainer.logging import MultiLogger +from mini_trainer.modeling import Classifier, EMATeacher +from mini_trainer.modeling.quantization import load_int8, prepare_int8 +from mini_trainer.trainer import train_one_epoch + + +def digest(path): + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def run(baseline, output, *, data_root=None, calibration_samples=256, qat_epochs=2, batch_size=32): + baseline, output = Path(baseline), Path(output) + if output.exists(): + raise FileExistsError(output) + if batch_size < 2 or qat_epochs < 1 or calibration_samples < batch_size: + raise ValueError("Require batch_size >= 2, qat_epochs >= 1 and calibration_samples >= batch_size.") + baseline_report = json.loads((baseline / "report.json").read_text()) + dataset = baseline_report["dataset"] + synthetic = dataset == "synthetic" + root = baseline / "data" if synthetic else Path(data_root) if data_root else None + if root is None: + raise ValueError("Real datasets require --data-root.") + manifest_path = baseline / "data/manifest.json" if synthetic else baseline / "dataset_manifest.json" + weights = baseline / "training/weights/last.pt" + for path, key in ((manifest_path, "dataset_manifest_sha256"), (weights, "checkpoint_sha256")): + if digest(path) != baseline_report[key]: + raise ValueError(f"Baseline provenance mismatch: {path}") + manifest = json.loads(manifest_path.read_text()) + seed = baseline_report["seed"] + torch.manual_seed(seed) + train = [record for record in manifest["records"] if record["split"] == "train"] + random.Random(seed).shuffle(train) + # Keep full training batches: no duplicated or silently dropped examples. + count = min(calibration_samples, len(train)) // batch_size * batch_size + if count < batch_size: + raise ValueError("Not enough training samples for a full calibration batch.") + train = train[:count] + test = [record for record in manifest["records"] if record["split"] == "test"] + if not test: + raise ValueError("A held-out test split is required.") + for record in train + test: + if digest(root / record["path"]) != record["sha256"]: + raise ValueError(f"Dataset content changed: {record['path']}") + model, preprocess = Classifier.build(weights=str(weights), device=torch.device("cpu"), dtype=torch.float32) + model.eval() + size = {"synthetic": 8, "mnist": 28, "blair": 64}[dataset] + + def read(records): + _, loader = get_inference_dataloader( + images=[str(root / record["path"]) for record in records], + resize_size=size, + batch_size=batch_size, + num_workers=0, + device=torch.device("cpu"), + dtype=torch.float32, + ) + with torch.no_grad(): + return torch.cat([preprocess(images) for images in loader]) + + images = read(train) + targets = torch.tensor([record["targets"] if dataset == "blair" else record["label"] for record in train]) + example = images[:batch_size] + output.mkdir(parents=True) + provenance = {"split": "train", "seed": seed, "manifest_sha256": digest(manifest_path), "records": train} + (output / "calibration.json").write_text(json.dumps(provenance, indent=2) + "\n") + models = {"float": model} + coverage = {} + for mode in ("ptq", "qat"): + prepared = prepare_int8(model, example, qat=mode == "qat") + if mode == "ptq": + with torch.no_grad(): + for batch in images.split(batch_size): + prepared(batch) + else: + loader = DataLoader(TensorDataset(images, targets), batch_size=batch_size, shuffle=False) + optimizer = torch.optim.AdamW(prepared.parameters(), lr=1e-4, weight_decay=0) + scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=100000) + scaler = torch.amp.GradScaler("cpu", enabled=False) + with torch.no_grad(): + original_output = model(example) + criterion = ( + MultiLevelWeightedCrossEntropyLoss([v.shape[1] for v in original_output], torch.device("cpu"), torch.float32) + if isinstance(original_output, list) + else torch.nn.CrossEntropyLoss() + ) + teacher = EMATeacher(enable=False, total_steps=qat_epochs * len(loader)) + logger = MultiLogger(loader, loader, epochs=qat_epochs, output=None, name="qat", logger_cls=[]) + for epoch in range(qat_epochs): + train_one_epoch(prepared, teacher, criterion, optimizer, scaler, scheduler, loader, epoch, logger) + torch.save( + { + "model": prepared.state_dict(), + "optimizer": optimizer.state_dict(), + "lr_scheduler": scheduler.state_dict(), + "scaler": scaler.state_dict(), + "epoch": qat_epochs - 1, + }, + output / "qat_checkpoint.pt", + ) + prepared.convert().save( + output / mode, + example, + preprocessing={ + "source_checkpoint_sha256": digest(weights), + "resize_size": size, + "recipe": "Classifier.build checkpoint preprocessing; RGB uint8 input", + }, + calibration=provenance, + ) + models[mode], coverage[mode] = load_int8(output / mode).lower(example) + # Test images enter the process only after conversion and all training finish. + heldout = read(test) + labels = np.array([record["targets"] if dataset == "blair" else [record["label"]] for record in test]).T + results = {} + for name, inference in models.items(): + collected = [] + with torch.no_grad(): + inference(example) # Warm up outside timing. + started = time.perf_counter() + for batch in heldout.split(batch_size): + n = len(batch) + if n < batch_size: + batch = torch.cat([batch, batch[:1].expand(batch_size - n, *batch.shape[1:])]) + scores = inference(batch) + collected.append([value[:n] for value in (scores if isinstance(scores, list) else [scores])]) + seconds = time.perf_counter() - started + scores = [torch.cat(level).numpy() for level in zip(*collected, strict=True)] + if not all(np.isfinite(level).all() for level in scores): + raise ValueError(f"Nonfinite {name} predictions") + accuracies = [float((level.argmax(1) == truth).mean()) for level, truth in zip(scores, labels, strict=True)] + np.savez( + output / f"{name}_predictions.npz", + **{f"scores_{i}": v for i, v in enumerate(scores)}, + labels=labels, + paths=np.array([record["path"] for record in test]), + ) + results[name] = { + "level_accuracies": accuracies, + "test_inference_seconds": seconds, + "artifact_bytes": weights.stat().st_size if name == "float" else (output / name / "model.pt2").stat().st_size, + } + repository = Path(__file__).resolve().parents[2] + source_digest = hashlib.sha256() + for source in sorted((repository / "mini_trainer").rglob("*.py")) + sorted(Path(__file__).parent.glob("*.py")): + source_digest.update(str(source.relative_to(repository)).encode()) + source_digest.update(source.read_bytes()) + report = { + "source_sha256": source_digest.hexdigest(), + "lock_sha256": digest(repository / "uv.lock"), + "schema_version": 1, + "dataset": dataset, + "seed": seed, + "device": "cpu", + "amp": False, + "batch_size": batch_size, + "threads": torch.get_num_threads(), + "calibration_samples": count, + "qat_epochs": qat_epochs, + "qat_optimizer": {"name": "AdamW", "lr": 1e-4, "weight_decay": 0}, + "baseline_report_sha256": digest(baseline / "report.json"), + "checkpoint_sha256": digest(weights), + "calibration_sha256": digest(output / "calibration.json"), + "results": results, + "lowering": coverage, + "git_revision": subprocess.check_output(["git", "rev-parse", "HEAD"], text=True, cwd=repository).strip(), + "status": "passed" + if synthetic and all(r["level_accuracies"][0] == 1 for r in results.values()) + else "failed" + if synthetic + else "completed", + "timing_scope": "single warmed full-test inference pass including padding and concatenation; not a speedup claim", + } + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--baseline", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--data-root", type=Path) + parser.add_argument("--calibration-samples", type=int, default=256) + parser.add_argument("--qat-epochs", type=int, default=2) + parser.add_argument("--batch-size", type=int, default=32) + args = parser.parse_args() + torch.set_num_threads(1) + report = run(**vars(args)) + print(json.dumps(report["results"], indent=2)) + if report["status"] == "failed": + raise SystemExit(1) + + +if __name__ == "__main__": + main() From 87eaeaa2b1a063e27620168d0fa85bf2011b171b Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 21:38:04 +0200 Subject: [PATCH 005/155] experiment: validate INT8 training kernels and performance boundaries Add a CUDA probe with INT8 weights, saved activations, and forward/backward GEMMs plus numerical and storage checks. Record workload-dependent speed and memory results, loader measurements, and the remaining optimizer, checkpoint, and convergence requirements. --- dev/benchmarks/README.md | 58 +++++++++++ dev/benchmarks/quantized_training.py | 141 +++++++++++++++++++++++++++ docs/quantization.md | 6 +- docs/roadmap.md | 8 +- docs/training-feature-validation.md | 8 +- tests/test_quantized_training.py | 46 +++++++++ 6 files changed, 263 insertions(+), 4 deletions(-) create mode 100644 dev/benchmarks/quantized_training.py create mode 100644 tests/test_quantized_training.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 66b4625..a262d05 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -124,3 +124,61 @@ from this development session; the equivalent local commands have been exercised References: [Actions job summaries](https://docs.github.com/en/actions/reference/workflows-and-actions/workflow-commands#adding-a-job-summary) and [artifact retention](https://github.com/actions/upload-artifact#retention-period). + +## Quantized training and loader performance + +Actual quantized training is the next implementation target. PTQ/QAT and AMP do +not establish reduced training memory or faster training. + +Two developer probes make the remaining work measurable: + +```bash +OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.loader +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ + .venv/bin/python -m dev.benchmarks.quantized_training +# Also test smaller GEMMs and full precision; benefit is workload-dependent: +CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.quantized_training --width 2048 --batch-size 512 +CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.quantized_training --dtype float32 +``` + +The QT probe stores weights and saved linear inputs in INT8, uses scaled INT8 +forward/input-gradient/weight-gradient GEMMs, and writes SGD updates back using +stochastic rounding. There is no retained floating-point master weight copy. +Gradients and update arithmetic remain floating point. It uses TorchAO's +experimental weight storage and Triton kernels with `torch.compile` fusion. +It currently exercises only SGD without momentum or weight decay, not the +repository's complete optimizer or checkpoint contracts. It is not an `mt_train` +feature yet, and a synthetic linear-stack MSE is not a convergence study. + +Local RTX 3080 Ti evidence (four 4096-wide layers, batch 2048, FP16 input/output, +three warm-up steps, ten measured forward/backward/SGD steps): FP16 took 29.10 ms +per step and peaked at 386,139,648 allocated bytes; INT8 took 12.92 ms and peaked +at 302,302,720 bytes. Stored weight bytes fell from 134,217,728 to 67,141,632. +These are single-run kernel-probe observations, not general end-to-end speedups. +The smaller 2048-wide/batch-512 probe was slower in INT8, and the unfused prototype +used more peak memory. Preserve those negative results when choosing dispatch. + +The loader probe compares identical uint8 cached batches against the previous +scalar-fetch/default-collate route. Batched `index_select` avoids restacking and +measured about 1.5x faster locally at 2048 RGB 64x64 images, batch 64, one thread, +zero workers. It excludes cache construction, preprocessing, GPU copies and model +compute; no end-to-end training gain is implied. + +Loaders retain their sampling/drop-last policies and worker caps. CPU batches +bound for CUDA are pinned after gathering, including CPU-cached and inference +batches. Optional `prefetch_factor` and `multiprocessing_context="spawn"` pass +through loader builders; both are inactive with zero workers. Spawn is useful +when parent code already has background threads, because fork may deadlock. +Use importable/pickleable readers and hooks with spawn. + +For the separate PTQ/QAT baseline comparison: + +```bash +.venv/bin/python -m dev.benchmarks.quantization --baseline /path/to/synthetic-cpu --output /tmp/int8-synthetic +.venv/bin/python -m dev.benchmarks.quantization --baseline /path/to/mnist-cpu --data-root examples/mnist --output /tmp/int8-mnist +``` + +This checks baseline checkpoint/manifest/file hashes, selects training-only +calibration samples, performs two small QAT epochs, then evaluates reloaded native +integer artifacts on the held-out split. The report and per-level predictions +retain baseline/PTQ/QAT results. It does not establish QT training speed or memory. diff --git a/dev/benchmarks/quantized_training.py b/dev/benchmarks/quantized_training.py new file mode 100644 index 0000000..b6f42c9 --- /dev/null +++ b/dev/benchmarks/quantized_training.py @@ -0,0 +1,141 @@ +"""Experimental CUDA QT kernel probe; not yet integrated with mt_train. + +Uses INT8 stored weights (no floating-point master copy), INT8 saved linear +inputs, and scaled INT8 forward/dgrad/wgrad GEMMs. Gradients and SGD update math +remain floating point. TorchAO stochastic rounding writes updates back to INT8. +Only zero-decay, zero-momentum SGD is exercised; do not infer AdamW/Muon support. +""" + +import gc +import json +import time +from argparse import ArgumentParser + +import torch + + +def dependencies(): + from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise + from torchao.prototype.quantized_training.int8_mm import scaled_int8_mm + + return Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise, scaled_int8_mm + + +class IntegerLinear(torch.autograd.Function): + """Row-scaled INT8 GEMMs, including approximate input and weight gradients.""" + + @staticmethod + def forward(ctx, inputs, weight): + _, quantize, mm = dependencies() + quantized, scale = quantize(inputs) + ctx.save_for_backward(quantized, scale, weight.int_data, weight.scale) + return mm(quantized.contiguous(), weight.int_data.T, scale.contiguous(), weight.scale.contiguous()) + + @staticmethod + def backward(ctx, grad_output): + _, quantize, mm = dependencies() + inputs, input_scale, weight, weight_scale = ctx.saved_tensors + ones = torch.ones(weight.shape[1], device=grad_output.device, dtype=grad_output.dtype) + grad_input = None + if ctx.needs_input_grad[0]: + # Weight scales lie along the contraction axis: absorb them into + # dY before its row quantization, not into the result columns. + quantized_grad, scale = quantize(grad_output * weight_scale) + grad_input = mm(quantized_grad.contiguous(), weight.contiguous(), scale.contiguous(), ones) + grad_weight = None + if ctx.needs_input_grad[1]: + # Similarly absorb saved activation scales into dY.T for dW. + quantized_grad, scale = quantize(grad_output.T * input_scale) + grad_weight = mm(quantized_grad.contiguous(), inputs.contiguous(), scale.contiguous(), ones) + return grad_input, grad_weight + + +class Layer(torch.nn.Module): + def __init__(self, width, quantized, dtype): + super().__init__() + self.quantized = quantized + weights = torch.randn(width, width, device="cuda", dtype=dtype) / width**0.5 + self.weight = torch.nn.Parameter(dependencies()[0].from_float(weights) if quantized else weights) + + def forward(self, inputs): + return IntegerLinear.apply(inputs, self.weight) if self.quantized else torch.nn.functional.linear(inputs, self.weight) + + +def run(width=4096, batch_size=2048, layers=4, steps=10, dtype="float16", compiled=True): + if not torch.cuda.is_available(): + raise RuntimeError("The QT probe requires an accessible CUDA GPU.") + results = {} + dependencies() + for quantized in (False, True): + torch.manual_seed(42) + gc.collect() + torch.cuda.empty_cache() + model = torch.nn.Sequential(*[Layer(width, quantized, getattr(torch, dtype)) for _ in range(layers)]) + if compiled: + model = torch.compile(model, fullgraph=True) + inputs = torch.randn(batch_size, width, device="cuda", dtype=getattr(torch, dtype)) + optimizer = torch.optim.SGD(model.parameters(), lr=1e-3, foreach=False) + if compiled: + optimizer.step = torch.compile(optimizer.step, fullgraph=False) + + def step(model=model, inputs=inputs, optimizer=optimizer): + optimizer.zero_grad(set_to_none=True) + loss = model(inputs).float().square().mean() + loss.backward() + optimizer.step() + return loss + + for _ in range(3): + step() + torch.cuda.synchronize() + torch.cuda.reset_peak_memory_stats() + started = time.perf_counter() + for _ in range(steps): + loss = step() + torch.cuda.synchronize() + seconds = (time.perf_counter() - started) / steps + results["int8" if quantized else dtype] = { + "seconds_per_step": seconds, + "peak_allocated_bytes": torch.cuda.max_memory_allocated(), + "loss": float(loss.detach()), + "weight_bytes": sum( + parameter.int_data.numel() + parameter.scale.numel() * parameter.scale.element_size() + if quantized + else parameter.numel() * parameter.element_size() + for parameter in model.parameters() + ), + } + del step, model, inputs, optimizer, loss + return { + "device": torch.cuda.get_device_name(), + "torch": str(torch.__version__), + "width": width, + "batch_size": batch_size, + "layers": layers, + "steps": steps, + "compiled": compiled, + "seed": 42, + "scope": "synthetic square linear stack and MSE; warmed forward/backward/SGD; excludes compilation and loading", + "optimizer": {"name": "SGD", "lr": 1e-3, "momentum": 0, "weight_decay": 0}, + "results": results, + } + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--width", type=int, default=4096) + parser.add_argument("--batch-size", type=int, default=2048) + parser.add_argument("--layers", type=int, default=4) + parser.add_argument("--steps", type=int, default=10) + parser.add_argument("--dtype", choices=["float32", "float16"], default="float16") + parser.add_argument("--eager", action="store_true") + args = vars(parser.parse_args()) + args["compiled"] = not args.pop("eager") + if min(args[key] for key in ("width", "batch_size", "layers", "steps")) < 1: + parser.error("Dimensions, layers and steps must be positive") + torch.set_num_threads(1) + print(json.dumps(run(**args), indent=2)) + + +if __name__ == "__main__": + main() diff --git a/docs/quantization.md b/docs/quantization.md index d1f8963..40317a8 100644 --- a/docs/quantization.md +++ b/docs/quantization.md @@ -1,7 +1,8 @@ # INT8 quantization: initial x86 backend This is an opt-in Python API for **static 8-bit weights and 8-bit activations**, -using TorchAO PT2E. It supports post-training calibration (PTQ) and +using TorchAO PT2E. Actual quantized training with reduced memory and training +time is separate ongoing work; see the [QT/loader probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance). It supports post-training calibration (PTQ) and quantization-aware training (QAT). QAT uses fake quantization with float32 master parameters/gradients; it does not promise integer backward computation or reduced training memory. Converted inference executes native oneDNN integer Conv/Linear @@ -21,6 +22,7 @@ ONNX export are unchanged. The new API is in `mini_trainer.modeling.quantization ## Calibration and inference ```python +import torch from mini_trainer.modeling.quantization import prepare_int8, load_int8 # model is a loaded floating-point mini_trainer model. All inputs below are @@ -57,7 +59,7 @@ as native integer inference. The bundle contains a reference `model.pt2` graph, checksum, input shape, class metadata, structured output mapping, bit widths, dependency versions, -preprocessing/calibration provenance, and verified lowering coverage. Packing +preprocessing/calibration provenance, and verified lowering coverage. Real calibration tensors are excluded from the saved program. Packing is performed again on the deployment CPU. A reference graph alone is not an accelerated runtime. Existing output directories are never overwritten. diff --git a/docs/roadmap.md b/docs/roadmap.md index 3400c9b..5760054 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -88,7 +88,13 @@ explicit upload commands or a serving deployment. ## 4. Training efficiency and augmentation -The primary implementation target is **deeper quantization in training and inference**. +The primary implementation target is **actual quantized training and faster data loading**. +QT must reduce retained training storage and demonstrate lower peak memory and faster +training on supported workloads. QAT with floating-point master weights is a separate +capability and does not complete this target. The initial CUDA integer forward/backward +kernel probe and cached-loader benchmark are documented in [the benchmark guide](../dev/benchmarks/README.md). +Optimizer support, checkpoint integration, real-data convergence and end-to-end +measurements remain required before claiming a supported QT training path. Loader hardening, float16/bfloat16 AMP and benchmark infrastructure do not complete that target. The implementation and comparison plan is in [training feature validation](training-feature-validation.md). diff --git a/docs/training-feature-validation.md b/docs/training-feature-validation.md index 2985405..804ebb8 100644 --- a/docs/training-feature-validation.md +++ b/docs/training-feature-validation.md @@ -4,7 +4,13 @@ This is planned work. Existing CPU/GPU benchmark results establish a baseline; they do not measure the benefit of the features below. EMA is temporarily nonfunctional and excluded from these experiments; repair is deferred. -## Quantization is the primary implementation target +## Actual quantized training is the primary implementation target + +The priority is now QT that lowers training memory and increases speed, together +with faster data loading for floating-point and quantized workloads. PTQ/QAT do +not satisfy that objective. See the [QT and loader probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance); +optimizer integration, checkpoint/resume, convergence and end-to-end measurement +remain requirements, not optional follow-ups. The initial [INT8 PTQ/QAT Python backend](quantization.md) is implemented on the `quant` branch. Its CPU tests establish a training-to-integer-inference path; diff --git a/tests/test_quantized_training.py b/tests/test_quantized_training.py new file mode 100644 index 0000000..aabc2c0 --- /dev/null +++ b/tests/test_quantized_training.py @@ -0,0 +1,46 @@ +"""CUDA numerical and saved-storage checks for the QT kernel prototype.""" + +import importlib.util +import os + +import pytest +import torch + +from dev.benchmarks.quantized_training import IntegerLinear, dependencies + + +def test_integer_training_gradients_and_saved_storage(): + if importlib.util.find_spec("torchao") is None: + pytest.skip("Install mini_trainer[quantization]") + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for the native INT8 training kernel test") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + torch.manual_seed(42) + weight_type, _, _ = dependencies() + inputs = torch.randn(32, 64, device="cuda", dtype=torch.float16, requires_grad=True) + original_weight = torch.randn(64, 64, device="cuda", dtype=torch.float16) / 8 + weight = torch.nn.Parameter(weight_type.from_float(original_weight)) + grad_output = torch.randn(32, 64, device="cuda", dtype=torch.float16) + saved = [] + + def record(tensor): + saved.append((tensor.dtype, tuple(tensor.shape))) + return tensor + + with torch.autograd.graph.saved_tensors_hooks(record, lambda tensor: tensor): + output = IntegerLinear.apply(inputs, weight) + output.backward(grad_output) + assert (torch.int8, (32, 64)) in saved and (torch.int8, (64, 64)) in saved + assert not any(dtype.is_floating_point and len(shape) == 2 for dtype, shape in saved) + expected_output = inputs.detach().float() @ original_weight.float().T + expected_dx = grad_output.float() @ original_weight.float() + expected_dw = grad_output.float().T @ inputs.detach().float() + for actual, expected in ((output, expected_output), (inputs.grad, expected_dx), (weight.grad, expected_dw)): + relative_error = (actual.float() - expected).norm() / expected.norm() + assert relative_error < 0.06 # Quantized arithmetic is approximate, not FP parity. + before = weight.int_data.clone() + optimizer = torch.optim.SGD([weight], lr=0.1, foreach=False) + optimizer.step() + assert not torch.equal(weight.int_data, before) + assert weight.int_data.dtype == torch.int8 From fc7406e953ded9b6803d25ca34f219616e71b754 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 21:51:33 +0200 Subject: [PATCH 006/155] fix: preserve small INT8 training gradients and validate optimizer updates Add isolated SGD and AdamW parameter dispatch, stochastic checkpoint continuation checks, and ordinary Linear coverage. Form backward scales in float32 to avoid FP16 underflow. Reject zero-gradient compiled probes and record corrected eager learning plus negative performance results; production QT integration remains ongoing. --- dev/benchmarks/README.md | 44 ++++++++++- dev/benchmarks/_int8_weight.py | 85 +++++++++++++++++++++ dev/benchmarks/quantized_training.py | 71 ++++++++++++++---- tests/test_quantized_training.py | 108 ++++++++++++++++++++++++++- 4 files changed, 287 insertions(+), 21 deletions(-) create mode 100644 dev/benchmarks/_int8_weight.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index a262d05..4a92a03 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -135,7 +135,7 @@ Two developer probes make the remaining work measurable: ```bash OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.loader CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ - .venv/bin/python -m dev.benchmarks.quantized_training + .venv/bin/python -m dev.benchmarks.quantized_training --eager # Also test smaller GEMMs and full precision; benefit is workload-dependent: CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.quantized_training --width 2048 --batch-size 512 CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.quantized_training --dtype float32 @@ -146,7 +146,11 @@ forward/input-gradient/weight-gradient GEMMs, and writes SGD updates back using stochastic rounding. There is no retained floating-point master weight copy. Gradients and update arithmetic remain floating point. It uses TorchAO's experimental weight storage and Triton kernels with `torch.compile` fusion. -It currently exercises only SGD without momentum or weight decay, not the +The experimental parameter dispatch supports SGD (including momentum/Nesterov +and weight decay) and AdamW. CPU tests check update error against floating-point +optimizer math and exact next-step continuation when weights, optimizer state +and RNG are restored together. Foreach parameter updates currently dispatch per +tensor; no fused-optimizer speed benefit is claimed. This does not establish the repository's complete optimizer or checkpoint contracts. It is not an `mt_train` feature yet, and a synthetic linear-stack MSE is not a convergence study. @@ -154,7 +158,9 @@ Local RTX 3080 Ti evidence (four 4096-wide layers, batch 2048, FP16 input/output three warm-up steps, ten measured forward/backward/SGD steps): FP16 took 29.10 ms per step and peaked at 386,139,648 allocated bytes; INT8 took 12.92 ms and peaked at 302,302,720 bytes. Stored weight bytes fell from 134,217,728 to 67,141,632. -These are single-run kernel-probe observations, not general end-to-end speedups. +These historical measurements predate the backward-scale underflow correction +and must not be used as evidence for the corrected implementation. They are +retained to document the investigation, not as valid training speedup claims. The smaller 2048-wide/batch-512 probe was slower in INT8, and the unfused prototype used more peak memory. Preserve those negative results when choosing dispatch. @@ -182,3 +188,35 @@ This checks baseline checkpoint/manifest/file hashes, selects training-only calibration samples, performs two small QAT epochs, then evaluates reloaded native integer artifacts on the held-out split. The report and per-level predictions retain baseline/PTQ/QAT results. It does not establish QT training speed or memory. + +The QT kernel probe also accepts `--optimizer-name adamw --weight-decay 0.1` +or `--momentum 0.9 --weight-decay 0.1` for SGD. AdamW uses an explicit +`--epsilon 1e-4` for both compared models: its usual `1e-8` underflows in FP16 +state. Use `--dtype float32 --epsilon 1e-8` to test ordinary float32 optimizer +state. This changes the experimental recipe, not the training CLI defaults. +CUDA regression tests exercise ordinary `nn.Linear` dispatch with non-square +weights, bias and batched inputs. Weight normalization, masked classifier +weights, convolutional QT, Muon, DDP and `mt_train` resume remain unverified. + +The original FP16 backward scale products could underflow before quantization, +suppressing gradients from mean-reduced losses. The corrected kernel forms +those products and their row scales in float32 while retaining INT8 GEMMs and +saved inputs. A CUDA regression compares input/weight gradients at both ordinary +and `1e-6` upstream gradient magnitudes. This was discovered through the AdamW +learning probe, beyond the original large-gradient numerical test. + +Model compilation currently drops the experimental subclass's parameter gradients +in the exercised linear-stack probe. Eager execution produces gradients close to +the floating-point reference and learns; compiling only the optimizer also +permits learning. The probe now rejects missing/zero parameter gradients during +warm-up, so a compiled run cannot be reported as successful training. Use +`--eager` for validated arithmetic while compiler compatibility is investigated. +No corrected end-to-end memory or speed benefit has yet been established. + +After the scale correction, the eager two-layer 2048-wide, batch-512 AdamW probe +(seed 42, three warm-up and three measured steps, decay 0.1, epsilon 1e-4) +reduced MSE from 0.99986 to 0.25309 in INT8, versus 0.99959 to 0.25218 in +FP16. INT8 took 5.85 ms/step and peaked at 171,218,944 bytes, versus 2.39 ms +and 111,412,224 bytes for FP16. This establishes similar short-run learning in +that diagnostic, but is a negative speed and peak-memory result. Stored weight +bytes alone fell from 16,777,216 to 8,396,800; that does not complete the QT goal. diff --git a/dev/benchmarks/_int8_weight.py b/dev/benchmarks/_int8_weight.py new file mode 100644 index 0000000..f22ac22 --- /dev/null +++ b/dev/benchmarks/_int8_weight.py @@ -0,0 +1,85 @@ +"""Experimental INT8 parameter dispatch for the CUDA training probe. + +Kept outside the runtime package until optimizer, checkpoint and model coverage +are established. Changes are local to this subclass, never TorchAO's dispatch. +""" + +import torch +from torch.utils._python_dispatch import return_and_correct_aliasing +from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight + + +class TrainingWeight(Int8QuantizedTrainingLinearWeight): + """INT8 storage with integer linear arithmetic and ordinary optimizer updates.""" + + +@TrainingWeight.implements_torch_function(torch.nn.functional.linear) +def linear(func, types, args, kwargs): + from .quantized_training import IntegerLinear + + inputs = args[0] if args else kwargs["input"] + weight = args[1] if len(args) > 1 else kwargs["weight"] + bias = args[2] if len(args) > 2 else kwargs.get("bias") + output = IntegerLinear.apply(inputs.reshape(-1, inputs.shape[-1]), weight) + output = output.reshape(*inputs.shape[:-1], weight.shape[0]) + return output if bias is None else output + bias + + +@TrainingWeight.implements([torch.ops.aten.detach.default, torch.ops.aten.clone.default]) +def preserve_type(func, types, args, kwargs): + original = args[0] + out = TrainingWeight(func(original.int_data, **kwargs), func(original.scale, **kwargs)) + return return_and_correct_aliasing(func, args, kwargs, out) + + +@TrainingWeight.implements(torch.ops.aten._to_copy.default) +def to_copy(func, types, args, kwargs): + original = args[0] + integer_kwargs = {key: value for key, value in kwargs.items() if key != "dtype"} + out = TrainingWeight(func(original.int_data, **integer_kwargs), func(original.scale, **kwargs)) + return return_and_correct_aliasing(func, args, kwargs, out) + + +@TrainingWeight.implements(torch.ops.aten.add.Tensor) +def add(func, types, args, kwargs): + # Coupled weight decay (SGD) adds the dequantized parameter to its gradient. + return func(*(value.dequantize() if isinstance(value, TrainingWeight) else value for value in args), **kwargs) + + +@TrainingWeight.implements(torch.ops.aten.mul_.Tensor) +def multiply_inplace(func, types, args, kwargs): + original, multiplier = args + # AdamW's decoupled decay is a scalar rescale. Preserve the integer codes + # exactly instead of adding a second stochastic rounding to every update. + if isinstance(multiplier, (float, int)): + original.scale.mul_(multiplier) + return original + return original.copy_(original.dequantize() * multiplier) + + +@TrainingWeight.implements(torch.ops.aten._foreach_add.List) +def foreach_add(func, types, args, kwargs): + return [torch.add(left, right, **kwargs) for left, right in zip(*args, strict=True)] + + +@TrainingWeight.implements(torch.ops.aten._foreach_add_.List) +def foreach_add_inplace(func, types, args, kwargs): + for left, right in zip(*args, strict=True): + left.add_(right, **kwargs) + return None + + +@TrainingWeight.implements(torch.ops.aten._foreach_mul_.Scalar) +def foreach_mul_inplace(func, types, args, kwargs): + for value in args[0]: + value.mul_(args[1]) + return None + + +@TrainingWeight.implements([torch.ops.aten._foreach_addcdiv_.Scalar, torch.ops.aten._foreach_addcdiv_.ScalarList]) +def foreach_addcdiv_inplace(func, types, args, kwargs): + factors = args[3] if len(args) > 3 else 1 + for index, (target, numerator, denominator) in enumerate(zip(*args[:3], strict=True)): + factor = factors[index] if isinstance(factors, (tuple, list)) else factors + target.addcdiv_(numerator, denominator, value=factor) + return None diff --git a/dev/benchmarks/quantized_training.py b/dev/benchmarks/quantized_training.py index b6f42c9..f6e05e6 100644 --- a/dev/benchmarks/quantized_training.py +++ b/dev/benchmarks/quantized_training.py @@ -3,11 +3,12 @@ Uses INT8 stored weights (no floating-point master copy), INT8 saved linear inputs, and scaled INT8 forward/dgrad/wgrad GEMMs. Gradients and SGD update math remain floating point. TorchAO stochastic rounding writes updates back to INT8. -Only zero-decay, zero-momentum SGD is exercised; do not infer AdamW/Muon support. +SGD and AdamW are available; do not infer support for other optimizers. """ import gc import json +import math import time from argparse import ArgumentParser @@ -15,10 +16,12 @@ def dependencies(): - from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise + from torchao.prototype.quantized_training.int8 import quantize_int8_rowwise from torchao.prototype.quantized_training.int8_mm import scaled_int8_mm - return Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise, scaled_int8_mm + from ._int8_weight import TrainingWeight + + return TrainingWeight, quantize_int8_rowwise, scaled_int8_mm class IntegerLinear(torch.autograd.Function): @@ -35,19 +38,22 @@ def forward(ctx, inputs, weight): def backward(ctx, grad_output): _, quantize, mm = dependencies() inputs, input_scale, weight, weight_scale = ctx.saved_tensors - ones = torch.ones(weight.shape[1], device=grad_output.device, dtype=grad_output.dtype) + ones = torch.ones(weight.shape[1], device=grad_output.device, dtype=torch.float32) grad_input = None if ctx.needs_input_grad[0]: # Weight scales lie along the contraction axis: absorb them into # dY before its row quantization, not into the result columns. - quantized_grad, scale = quantize(grad_output * weight_scale) + quantized_grad, scale = quantize(grad_output.float() * weight_scale.float()) grad_input = mm(quantized_grad.contiguous(), weight.contiguous(), scale.contiguous(), ones) grad_weight = None if ctx.needs_input_grad[1]: # Similarly absorb saved activation scales into dY.T for dW. - quantized_grad, scale = quantize(grad_output.T * input_scale) + quantized_grad, scale = quantize(grad_output.T.float() * input_scale.float()) grad_weight = mm(quantized_grad.contiguous(), inputs.contiguous(), scale.contiguous(), ones) - return grad_input, grad_weight + return ( + grad_input.to(grad_output.dtype) if grad_input is not None else None, + grad_weight.to(weight_scale.dtype) if grad_weight is not None else None, + ) class Layer(torch.nn.Module): @@ -58,10 +64,21 @@ def __init__(self, width, quantized, dtype): self.weight = torch.nn.Parameter(dependencies()[0].from_float(weights) if quantized else weights) def forward(self, inputs): - return IntegerLinear.apply(inputs, self.weight) if self.quantized else torch.nn.functional.linear(inputs, self.weight) - - -def run(width=4096, batch_size=2048, layers=4, steps=10, dtype="float16", compiled=True): + return torch.nn.functional.linear(inputs, self.weight) + + +def run( + width=4096, + batch_size=2048, + layers=4, + steps=10, + dtype="float16", + compiled=True, + optimizer_name="sgd", + weight_decay=0.0, + momentum=0.0, + epsilon=1e-4, +): if not torch.cuda.is_available(): raise RuntimeError("The QT probe requires an accessible CUDA GPU.") results = {} @@ -74,7 +91,9 @@ def run(width=4096, batch_size=2048, layers=4, steps=10, dtype="float16", compil if compiled: model = torch.compile(model, fullgraph=True) inputs = torch.randn(batch_size, width, device="cuda", dtype=getattr(torch, dtype)) - optimizer = torch.optim.SGD(model.parameters(), lr=1e-3, foreach=False) + optimizer_cls = {"sgd": torch.optim.SGD, "adamw": torch.optim.AdamW}[optimizer_name] + optimizer_options = {"momentum": momentum} if optimizer_name == "sgd" else {"eps": epsilon} + optimizer = optimizer_cls(model.parameters(), lr=1e-3, weight_decay=weight_decay, foreach=False, **optimizer_options) if compiled: optimizer.step = torch.compile(optimizer.step, fullgraph=False) @@ -85,8 +104,16 @@ def step(model=model, inputs=inputs, optimizer=optimizer): optimizer.step() return loss + warmup_losses = [] for _ in range(3): - step() + warmup_losses.append(float(step().detach())) + # This random linear-stack MSE must train every weight. AOT can + # otherwise report fast steps while silently dropping gradients + # for the experimental tensor subclass. + if any(parameter.grad is None or not bool(torch.count_nonzero(parameter.grad)) for parameter in model.parameters()): + raise RuntimeError( + "Missing or zero weight gradients in the QT probe; compiled tensor-subclass training is not verified. Try --eager." + ) torch.cuda.synchronize() torch.cuda.reset_peak_memory_stats() started = time.perf_counter() @@ -98,6 +125,7 @@ def step(model=model, inputs=inputs, optimizer=optimizer): "seconds_per_step": seconds, "peak_allocated_bytes": torch.cuda.max_memory_allocated(), "loss": float(loss.detach()), + "warmup_losses": warmup_losses, "weight_bytes": sum( parameter.int_data.numel() + parameter.scale.numel() * parameter.scale.element_size() if quantized @@ -105,6 +133,8 @@ def step(model=model, inputs=inputs, optimizer=optimizer): for parameter in model.parameters() ), } + if not math.isfinite(results["int8" if quantized else dtype]["loss"]): + raise RuntimeError("Nonfinite training loss; check optimizer precision, epsilon and learning rate.") del step, model, inputs, optimizer, loss return { "device": torch.cuda.get_device_name(), @@ -115,8 +145,13 @@ def step(model=model, inputs=inputs, optimizer=optimizer): "steps": steps, "compiled": compiled, "seed": 42, - "scope": "synthetic square linear stack and MSE; warmed forward/backward/SGD; excludes compilation and loading", - "optimizer": {"name": "SGD", "lr": 1e-3, "momentum": 0, "weight_decay": 0}, + "scope": "synthetic square linear stack and MSE; warmed forward/backward/optimizer; excludes compilation and loading", + "optimizer": { + "name": optimizer_name, + "lr": 1e-3, + "weight_decay": weight_decay, + **({"momentum": momentum} if optimizer_name == "sgd" else {"eps": epsilon}), + }, "results": results, } @@ -129,10 +164,16 @@ def main(): parser.add_argument("--steps", type=int, default=10) parser.add_argument("--dtype", choices=["float32", "float16"], default="float16") parser.add_argument("--eager", action="store_true") + parser.add_argument("--optimizer-name", choices=["sgd", "adamw"], default="sgd") + parser.add_argument("--weight-decay", type=float, default=0.0) + parser.add_argument("--momentum", type=float, default=0.0) + parser.add_argument("--epsilon", type=float, default=1e-4, help="AdamW epsilon; 1e-4 avoids FP16 underflow") args = vars(parser.parse_args()) args["compiled"] = not args.pop("eager") if min(args[key] for key in ("width", "batch_size", "layers", "steps")) < 1: parser.error("Dimensions, layers and steps must be positive") + if args["optimizer_name"] != "sgd" and args["momentum"]: + parser.error("--momentum applies only to SGD") torch.set_num_threads(1) print(json.dumps(run(**args), indent=2)) diff --git a/tests/test_quantized_training.py b/tests/test_quantized_training.py index aabc2c0..e7ac59e 100644 --- a/tests/test_quantized_training.py +++ b/tests/test_quantized_training.py @@ -9,7 +9,8 @@ from dev.benchmarks.quantized_training import IntegerLinear, dependencies -def test_integer_training_gradients_and_saved_storage(): +@pytest.mark.parametrize("gradient_scale", [1.0, 1e-6]) +def test_integer_training_gradients_and_saved_storage(gradient_scale): if importlib.util.find_spec("torchao") is None: pytest.skip("Install mini_trainer[quantization]") if os.environ.get("RUN_CUDA_TESTS") != "1": @@ -21,7 +22,7 @@ def test_integer_training_gradients_and_saved_storage(): inputs = torch.randn(32, 64, device="cuda", dtype=torch.float16, requires_grad=True) original_weight = torch.randn(64, 64, device="cuda", dtype=torch.float16) / 8 weight = torch.nn.Parameter(weight_type.from_float(original_weight)) - grad_output = torch.randn(32, 64, device="cuda", dtype=torch.float16) + grad_output = torch.randn(32, 64, device="cuda", dtype=torch.float16) * gradient_scale saved = [] def record(tensor): @@ -40,7 +41,108 @@ def record(tensor): relative_error = (actual.float() - expected).norm() / expected.norm() assert relative_error < 0.06 # Quantized arithmetic is approximate, not FP parity. before = weight.int_data.clone() - optimizer = torch.optim.SGD([weight], lr=0.1, foreach=False) + optimizer = torch.optim.SGD([weight], lr=0.1 / gradient_scale, foreach=False) optimizer.step() assert not torch.equal(weight.int_data, before) assert weight.int_data.dtype == torch.int8 + + +@pytest.mark.parametrize("optimizer_name", ["sgd", "adamw"]) +@pytest.mark.parametrize("foreach", [None, False, True]) +def test_integer_optimizer_updates_and_continuation(optimizer_name, foreach): + """Compare update math before rounding and resume the stochastic trajectory.""" + import copy + import io + + pytest.importorskip("torchao") + from dev.benchmarks._int8_weight import TrainingWeight + + torch.manual_seed(123) + parameter = torch.nn.Parameter(TrainingWeight.from_float(torch.randn(32, 64))) + reference = torch.nn.Parameter(parameter.dequantize().clone()) + cls = torch.optim.SGD if optimizer_name == "sgd" else torch.optim.AdamW + options = dict(lr=0.03, weight_decay=0.1, foreach=foreach) + if optimizer_name == "sgd": + options.update(momentum=0.9, nesterov=True) + optimizer, reference_optimizer = cls([parameter], **options), cls([reference], **options) + for _ in range(4): + # Start from the same represented weights, keeping optimizer state. + with torch.no_grad(): + reference.copy_(parameter.dequantize()) + gradient = torch.randn_like(reference) + parameter.grad, reference.grad = gradient.clone(), gradient.clone() + optimizer.step() + reference_optimizer.step() + error = (parameter.dequantize() - reference).abs() + # One stochastic quantization adds less than one quantization interval + # per element (plus scale calculation rounding). + assert torch.all(error <= reference.detach().abs().amax(1, keepdim=True) / 127 + 1e-6) + checkpoint = io.BytesIO() + torch.save({"weight": parameter.detach(), "optimizer": optimizer.state_dict(), "rng": torch.get_rng_state()}, checkpoint) + checkpoint.seek(0) + with torch.serialization.safe_globals([TrainingWeight]): + restored = torch.load(checkpoint, weights_only=True) + resumed = torch.nn.Parameter(restored["weight"]) + resumed_optimizer = cls([resumed], **options) + resumed_optimizer.load_state_dict(restored["optimizer"]) + gradient = torch.randn_like(reference) + parameter.grad, resumed.grad = gradient.clone(), gradient.clone() + torch.set_rng_state(restored["rng"]) + optimizer.step() + torch.set_rng_state(restored["rng"]) + resumed_optimizer.step() + assert torch.equal(parameter.int_data, resumed.int_data) + assert torch.equal(parameter.scale, resumed.scale) + assert isinstance(copy.deepcopy(parameter), TrainingWeight) + assert isinstance(parameter.to(torch.float64), TrainingWeight) + assert isinstance(parameter.detach(), TrainingWeight) + for key, value in optimizer.state[parameter].items(): + other = resumed_optimizer.state[resumed][key] + if isinstance(value, torch.Tensor): + assert torch.equal(value, other) + else: + assert value == other + + +def test_integer_decay_preserves_codes_and_upstream_dispatch(): + pytest.importorskip("torchao") + from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight + + from dev.benchmarks._int8_weight import TrainingWeight + + original = torch.randn(8, 16) + weight = TrainingWeight.from_float(original) + codes, represented = weight.int_data.clone(), weight.dequantize() + weight.mul_(0.9) + assert torch.equal(weight.int_data, codes) + torch.testing.assert_close(weight.dequantize(), represented * 0.9) + upstream = Int8QuantizedTrainingLinearWeight.from_float(original) + assert type(upstream.detach()) is Int8QuantizedTrainingLinearWeight + with pytest.raises(NotImplementedError): + upstream.mul_(0.9) + + +@pytest.mark.parametrize("dtype,epsilon", [(torch.float16, 1e-4), (torch.float32, 1e-8)]) +def test_integer_linear_module_cuda(dtype, epsilon): + pytest.importorskip("torchao") + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for CUDA linear dispatch") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + from dev.benchmarks._int8_weight import TrainingWeight + + torch.manual_seed(456) + layer = torch.nn.Linear(64, 32, device="cuda", dtype=dtype) + layer.weight = torch.nn.Parameter(TrainingWeight.from_float(layer.weight)) + inputs = torch.randn(2, 16, 64, device="cuda", dtype=dtype, requires_grad=True) + expected = torch.nn.functional.linear(inputs.float(), layer.weight.dequantize().float(), layer.bias.float()) + output = layer(inputs) + assert output.shape == (2, 16, 32) + assert (output.float() - expected).norm() / expected.norm() < 0.04 + output.float().square().mean().backward() + assert torch.isfinite(inputs.grad).all() + assert torch.isfinite(layer.bias.grad).all() + optimizer = torch.optim.AdamW(layer.parameters(), lr=0.01, weight_decay=0.1, eps=epsilon) + optimizer.step() + assert isinstance(layer.weight, TrainingWeight) + assert torch.isfinite(layer.weight.dequantize()).all() From 7a095a25a9d9cfefb064f0d56c8c571dab869bad Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 21:59:36 +0200 Subject: [PATCH 007/155] fix: invalidate compiled INT8 graphs when training math changes Include implementation source and tensor metadata in the experimental parameter cache key. Add compiled-gradient regression coverage and replace the stale-cache diagnosis with corrected learning and performance measurements. --- dev/benchmarks/README.md | 33 ++++++++++++++++++------- dev/benchmarks/_int8_weight.py | 17 +++++++++++++ tests/test_quantized_training.py | 41 ++++++++++++++++++++++++++++++++ 3 files changed, 83 insertions(+), 8 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 4a92a03..515cd57 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -135,7 +135,7 @@ Two developer probes make the remaining work measurable: ```bash OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.loader CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ - .venv/bin/python -m dev.benchmarks.quantized_training --eager + .venv/bin/python -m dev.benchmarks.quantized_training # Also test smaller GEMMs and full precision; benefit is workload-dependent: CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.quantized_training --width 2048 --batch-size 512 CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.quantized_training --dtype float32 @@ -205,13 +205,20 @@ saved inputs. A CUDA regression compares input/weight gradients at both ordinary and `1e-6` upstream gradient magnitudes. This was discovered through the AdamW learning probe, beyond the original large-gradient numerical test. -Model compilation currently drops the experimental subclass's parameter gradients -in the exercised linear-stack probe. Eager execution produces gradients close to -the floating-point reference and learns; compiling only the optimizer also -permits learning. The probe now rejects missing/zero parameter gradients during -warm-up, so a compiled run cannot be reported as successful training. Use -`--eager` for validated arithmetic while compiler compatibility is investigated. -No corrected end-to-end memory or speed benefit has yet been established. +The apparent compiled zero-gradient failure was traced to an AOT disk-cache hit +for the old FP16 backward scale products, not the corrected source. The +experimental tensor now supplies a stable key containing its implementation +source digest and tensor metadata. Changing backward/dispatch code invalidates +that key; changing weight values does not. CUDA tests compare compiled and eager +parameter gradients for a small mean-reduced loss. The probe retains its +missing/zero-gradient gate, and records warm-up loss trajectories before timing. + +With the versioned key, the compiled two-layer 2048-wide, batch-512 AdamW run +(decay 0.1, epsilon 1e-4, three warm-up and three measured steps) ended at MSE +0.36759 for INT8 versus 0.36605 for FP16. INT8 took 2.16 ms/step and peaked at +103,033,344 bytes versus 1.53 ms and 92,539,392 bytes. Compilation improves on +the eager INT8 result below, but this remains a negative speed and memory result +against compiled FP16. Larger-workload and end-to-end gains require measurement. After the scale correction, the eager two-layer 2048-wide, batch-512 AdamW probe (seed 42, three warm-up and three measured steps, decay 0.1, epsilon 1e-4) @@ -220,3 +227,13 @@ FP16. INT8 took 5.85 ms/step and peaked at 171,218,944 bytes, versus 2.39 ms and 111,412,224 bytes for FP16. This establishes similar short-run learning in that diagnostic, but is a negative speed and peak-memory result. Stored weight bytes alone fell from 16,777,216 to 8,396,800; that does not complete the QT goal. + +Repeating the original large compiled SGD probe after both corrections (four +4096-wide layers, batch 2048, three warm-up and ten measured steps) passed the +nonzero-gradient gate. INT8 took 18.12 ms/step and peaked at 319,063,552 bytes, +versus 30.39 ms and 386,139,648 bytes for FP16: about 1.68x faster and 17% less +peak allocated memory on this RTX 3080 Ti workload. Weight storage remained +67,141,632 versus 134,217,728 bytes. Final MSE was 0.99880 versus 0.99870; the +small SGD updates in this probe do not establish convergence. This replaces the +pre-correction large-workload timing above. Real-model training, optimizer-state +memory, data loading and task quality still require end-to-end validation. diff --git a/dev/benchmarks/_int8_weight.py b/dev/benchmarks/_int8_weight.py index f22ac22..94ef79f 100644 --- a/dev/benchmarks/_int8_weight.py +++ b/dev/benchmarks/_int8_weight.py @@ -4,14 +4,31 @@ are established. Changes are local to this subclass, never TorchAO's dispatch. """ +import hashlib +from pathlib import Path + import torch from torch.utils._python_dispatch import return_and_correct_aliasing from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight +# Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary +# graph key. Include both implementation files so changing backward math cannot +# reuse a graph compiled for an earlier version. Compute once when importing. +_IMPLEMENTATION_HASH = hashlib.sha256( + Path(__file__).read_bytes() + Path(__file__).with_name("quantized_training.py").read_bytes() +).hexdigest() + class TrainingWeight(Int8QuantizedTrainingLinearWeight): """INT8 storage with integer linear arithmetic and ordinary optimizer updates.""" + def _stable_hash_for_caching(self): + metadata = [ + (tuple(value.shape), tuple(value.stride()), str(value.dtype), str(value.device), value.requires_grad) + for value in (self, self.int_data, self.scale) + ] + return hashlib.sha256(repr((_IMPLEMENTATION_HASH, metadata)).encode()).hexdigest() + @TrainingWeight.implements_torch_function(torch.nn.functional.linear) def linear(func, types, args, kwargs): diff --git a/tests/test_quantized_training.py b/tests/test_quantized_training.py index e7ac59e..c6171e6 100644 --- a/tests/test_quantized_training.py +++ b/tests/test_quantized_training.py @@ -146,3 +146,44 @@ def test_integer_linear_module_cuda(dtype, epsilon): optimizer.step() assert isinstance(layer.weight, TrainingWeight) assert torch.isfinite(layer.weight.dequantize()).all() + + +def test_integer_compiler_key_tracks_code_and_metadata(monkeypatch): + pytest.importorskip("torchao") + from dev.benchmarks import _int8_weight + + weight = _int8_weight.TrainingWeight.from_float(torch.randn(8, 16)) + key = weight._stable_hash_for_caching() + other_values = _int8_weight.TrainingWeight.from_float(torch.randn(8, 16)) + assert other_values._stable_hash_for_caching() == key + assert weight.to(torch.float64)._stable_hash_for_caching() != key + weight.requires_grad_(True) + assert weight._stable_hash_for_caching() != key + weight.requires_grad_(False) + monkeypatch.setattr(_int8_weight, "_IMPLEMENTATION_HASH", "changed-backward-implementation") + assert weight._stable_hash_for_caching() != key + + +def test_compiled_integer_parameter_gradients(): + import copy + + pytest.importorskip("torchao") + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for compiled INT8 gradient validation") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + from dev.benchmarks.quantized_training import Layer + + torch.manual_seed(42) + eager = torch.nn.Sequential(Layer(64, True, torch.float16), Layer(64, True, torch.float16)) + compiled = torch.compile(copy.deepcopy(eager), fullgraph=True) + # Ordinary training input does not require gradients. A small mean-reduced + # loss exercises scale products below FP16's representable range. + inputs = torch.randn(32, 64, device="cuda", dtype=torch.float16) + for model in (eager, compiled): + (model(inputs).float().square().mean() / 1000).backward() + for expected, actual in zip(eager.parameters(), compiled.parameters(), strict=True): + assert actual.grad is not None + assert torch.count_nonzero(actual.grad) > 0 + error = (actual.grad.float() - expected.grad.float()).norm() / expected.grad.float().norm() + assert error < 0.04 From 1b89cdda63a2d4e225e9bc75e3fa8c4a8723a653 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 22:11:16 +0200 Subject: [PATCH 008/155] perf: bound cache construction and expose reader thread controls Replace unbounded queues and the daemon writer with bounded ordered futures and direct validated writes. Decode each sample once, propagate failures with reader cleanup, and cap automatic cache threads at 16 with a synchronous override. Add CPU/CUDA regressions and a reproducible comparison showing similar PNG throughput. --- dev/benchmarks/README.md | 39 ++++++++ dev/benchmarks/cache.py | 95 ++++++++++++++++++ mini_trainer/data/io.py | 131 +++++++++++-------------- mini_trainer/data/loader.py | 3 +- tests/test_integration_lazy_dataset.py | 113 +++++++++++++++++++++ tests/utils/test_loader.py | 41 +++++++- 6 files changed, 344 insertions(+), 78 deletions(-) create mode 100644 dev/benchmarks/cache.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 515cd57..82d850b 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -237,3 +237,42 @@ peak allocated memory on this RTX 3080 Ti workload. Weight storage remained small SGD updates in this probe do not establish convergence. This replaces the pre-correction large-workload timing above. Real-model training, optimizer-state memory, data loading and task quality still require end-to-end validation. + +## Bounded cache construction + +CPU/CUDA cache construction now reserves four available CPUs and caps automatic +reader threads at 16 (previously up to 128, reserving two). Four or fewer CPUs +select synchronous construction. `LazyDataset(..., cache_workers=0)` and +`get_dataset_dataloader(..., cache_workers=0)` explicitly disable cache reader +threads; positive values override the automatic selection. Existing builder +keyword forwarding supports this option. DataLoader worker selection is separate. + +At most twice the selected reader count is submitted ahead, plus a write batch +of at most 64 samples. There is no unbounded reorder/write queue or daemon writer. +Readers run once per sample, including the first shape probe. Read errors, +inconsistent sample shapes/structures and write errors propagate to the caller; +pending work is cancelled and active readers finish before construction returns. +Cached ordering, labels and dtypes remain unchanged for valid inputs. + +Compare against the previous implementation using the developer benchmark: + +```bash +git show 7a095a2:mini_trainer/data/io.py > /tmp/cache-baseline.py +OMP_NUM_THREADS=1 CUDA_VISIBLE_DEVICES='' .venv/bin/python -m dev.benchmarks.cache --baseline-source /tmp/cache-baseline.py +# Explicit synchronous cache construction: +OMP_NUM_THREADS=1 CUDA_VISIBLE_DEVICES='' .venv/bin/python -m dev.benchmarks.cache --workers 0 +``` + +`--baseline-source` executes developer-supplied Python source. The benchmark uses +seed 42, 2048 samples cycling through 64 generated RGB 128x128 PNGs, a warm +filesystem cache, one PyTorch thread and five measured repeats in alternating +order. File generation is excluded. It verifies image and label output and +records source hashes. This is cache-construction timing, not training throughput. + +On the local 20-CPU allocation, automatic construction used 16 reader threads +versus the previous 18. The final comparison measured 1,722 versus 1,686 samples +per second (about 1.02x); an earlier repeat was effectively tied. Treat this as +similar throughput, not evidence of a meaningful speedup. The concrete gain is +bounded read-ahead and reliable failure handling, with explicit synchronous +construction available for shared-node environments. CUDA checks also verify +pinned CPU transfer batches and exact CPU/CUDA cache contents. diff --git a/dev/benchmarks/cache.py b/dev/benchmarks/cache.py new file mode 100644 index 0000000..bb893da --- /dev/null +++ b/dev/benchmarks/cache.py @@ -0,0 +1,95 @@ +"""Measure CPU cache construction from reproducible PNGs; exclude file creation. + +An optional developer-supplied baseline io.py is executed as Python source to +compare an earlier implementation in the same dependency environment. +""" + +import gc +import hashlib +import importlib.util +import json +import statistics +import sys +import tempfile +import time +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np +import torch +from PIL import Image + +from mini_trainer.data._workers import _available_cpu_count +from mini_trainer.data.io import LazyDataset, make_read_and_resize_fn +from mini_trainer.data.loader import PathLabelProcessor + + +def run(samples=2048, size=128, repeats=5, workers=None, baseline_source=None): + implementations = {"bounded": LazyDataset} + if baseline_source is not None: + name = "mini_trainer.data._benchmark_baseline" + spec = importlib.util.spec_from_file_location(name, baseline_source) + module = importlib.util.module_from_spec(spec) + sys.modules[name] = module + spec.loader.exec_module(module) + implementations["baseline"] = module.LazyDataset + reader = PathLabelProcessor(make_read_and_resize_fn((size, size), torch.device("cpu"), torch.uint8), None, False) + timings = {name: [] for name in implementations} + generator = np.random.default_rng(42) + with tempfile.TemporaryDirectory(prefix="mini-trainer-cache-probe-") as root: + paths = [] + # Repeat a fixed small corpus, measuring warm filesystem-cache decoding. + for index in range(min(64, samples)): + path = Path(root) / f"{index}.png" + Image.fromarray(generator.integers(0, 256, (size, size, 3), dtype=np.uint8)).save(path) + paths.append(str(path)) + items = ([paths[index % len(paths)] for index in range(samples)], list(range(samples))) + expected_first = reader((items[0][0], 0))[0] + for trial in range(repeats + 1): + order = list(implementations) if trial % 2 else list(reversed(implementations)) + for name in order: + gc.collect() + options = {"cache_workers": workers} if name == "bounded" else {} + started = time.perf_counter() + dataset = implementations[name](reader, items, cache="cpu", **options) + seconds = time.perf_counter() - started + assert len(dataset) == samples + torch.testing.assert_close(dataset[0][0], expected_first, rtol=0, atol=0) + assert torch.equal(dataset._ram_cache.tensors[1], torch.arange(samples)) + if trial: + timings[name].append(seconds) + del dataset + return { + "seed": 42, + "samples": samples, + "shape": [3, size, size], + "threads": torch.get_num_threads(), + "cache_workers": workers, + "available_cpus": _available_cpu_count(), + "source_sha256": { + name: hashlib.sha256(Path(sys.modules[implementation.__module__].__file__).read_bytes()).hexdigest() + for name, implementation in implementations.items() + }, + "scope": "CPU cache construction from 64 repeated PNGs; warm filesystem cache; excludes file generation", + "seconds": timings, + "median_samples_per_second": {name: samples / statistics.median(values) for name, values in timings.items()}, + "speedup": statistics.median(timings["baseline"]) / statistics.median(timings["bounded"]) if baseline_source else None, + } + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--samples", type=int, default=2048) + parser.add_argument("--size", type=int, default=128) + parser.add_argument("--repeats", type=int, default=5) + parser.add_argument("--workers", type=int) + parser.add_argument("--baseline-source", type=Path) + args = parser.parse_args() + if min(args.samples, args.size, args.repeats) < 1 or (args.workers is not None and args.workers < 0): + parser.error("Sizes and repeats must be positive; workers must be nonnegative") + torch.set_num_threads(1) + print(json.dumps(run(**vars(args)), indent=2)) + + +if __name__ == "__main__": + main() diff --git a/mini_trainer/data/io.py b/mini_trainer/data/io.py index 5a582e5..af8ddcb 100644 --- a/mini_trainer/data/io.py +++ b/mini_trainer/data/io.py @@ -3,11 +3,12 @@ import operator import os import warnings +from collections import deque from collections.abc import Callable, Iterable, Iterator, Sequence from concurrent.futures import ThreadPoolExecutor +from contextlib import closing from enum import Enum -from queue import Queue -from threading import Thread +from itertools import batched from typing import Any, TypeVar, cast import numpy as np @@ -365,9 +366,16 @@ def __init__( # noqa: D107 func: Callable[[Any], torch.Tensor | tuple[torch.Tensor, ...] | list[torch.Tensor]], items: Sequence[Sequence], cache: str | int | CACHE_MODE | None = None, + *, + cache_workers: int | None = None, ): + if cache_workers is not None and (isinstance(cache_workers, bool) or not isinstance(cache_workers, int) or cache_workers < 0): + raise ValueError("cache_workers must be a nonnegative integer or None.") + self._cache_workers = cache_workers self.func = func self.items = tuple(np.asarray(seq, dtype=_infer_numeric_dtype(seq)) if len(seq) > 0 else np.empty((0,)) for seq in items) + if self.items and any(len(seq) != len(self.items[0]) for seq in self.items): + raise ValueError("Dataset input sequences must have equal lengths.") self._init_cache(CACHE_MODE(cache)) @staticmethod @@ -420,7 +428,7 @@ def _cache_ram(self, desc: str = "Writing to CPU RAM cache..."): self._ram_was_single_tensor = False templates = [e.new_empty(e.shape) for e in first_item_processed] else: - raise TypeError(f"The provided function must return a tensor ora tuple/list of tensors, but got {type(first_item_processed)}") + raise TypeError(f"The provided function must return a tensor or a tuple/list of tensors, but got {type(first_item_processed)}") stacked_tensors = [ torch.empty( @@ -432,79 +440,56 @@ def _cache_ram(self, desc: str = "Writing to CPU RAM cache..."): for template in templates ] - max_workers = _default_worker_count(128, reserve=2, minimum=1) - batch_size = min(256, 4 * max_workers) - fetch_pool = ThreadPoolExecutor(max_workers=max_workers, thread_name_prefix="fetcher") - fetched_queue: Queue[tuple[int, torch.Tensor | Sequence[torch.Tensor] | Exception]] = Queue(max(32, batch_size * 4)) - insert_buffer: dict[int, torch.Tensor | Sequence[torch.Tensor]] = dict() - insert_queue: Queue[tuple[int, torch.Tensor | Sequence[torch.Tensor]]] = Queue() - - def _fetch_one(idx_item): - idx, item = idx_item - try: - data = self.func(item) - fetched_queue.put((idx, data)) - except Exception as e: - fetched_queue.put((idx, e)) - - def _contiguous_write(idx: list[int], data: list[torch.Tensor] | list[list[torch.Tensor]]) -> None: - """Write data to indexes. - - Args: - idx: A list of contigous increasing indices for each corresponding torch.Tensor (element) in `data`. - data: A list of torch.Tensor with the same length as `idx`. - """ - if len(idx) == 0: + max_workers = _default_worker_count(16) if self._cache_workers is None else self._cache_workers + max_workers = min(max_workers, len(self) - 1) + batch_size = min(64, max(1, 4 * max_workers)) + + def processed_items(): + # Reuse the shape probe: every reader/hook runs exactly once. + yield first_item_processed + items = iter(zip(*self.items)) + next(items) + if max_workers == 0: + yield from map(self.func, items) return - slc = slice(idx[0], idx[-1] + 1) - # Insert data into slice along first dimension in dst (in-place) - if self._ram_was_single_tensor: - assert not data or isinstance(data[0], torch.Tensor) - data = cast(list[torch.Tensor], data) - torch.stack(data, out=stacked_tensors[0][slc]) - else: - for i, elements in enumerate(zip(*data)): - torch.stack(elements, out=stacked_tensors[i][slc]) - - def _write(): - end_idx = len(self) - 1 - with TQDM(range(len(self)), desc=desc, leave=False) as pbar: - batch = ([], []) - while True: - idx, data = insert_queue.get() - batch[0].append(idx) - batch[1].append(data) - if len(batch[0]) >= batch_size: - _contiguous_write(*batch) - batch = ([], []) - pbar.update() - if idx == end_idx: + pool = ThreadPoolExecutor(max_workers=max_workers, thread_name_prefix="mini-trainer-cache") + pending = deque() + try: + # Bound both submitted work and decoded samples even when an + # early reader or the cache writer is slower than later reads. + for _ in range(2 * max_workers): + item = next(items, None) + if item is None: break - _contiguous_write(*batch) - - write_thread = Thread(target=_write, daemon=True) - write_thread.start() - - try: - fetch_pool.map(_fetch_one, enumerate(zip(*self.items))) - - nxt_idx = 0 - for _ in range(len(self)): - idx, data = fetched_queue.get() - if isinstance(data, Exception): - raise data - if idx == nxt_idx: - insert_queue.put((idx, data)) - nxt_idx += 1 - while nxt_idx in insert_buffer: - insert_queue.put((nxt_idx, insert_buffer.pop(nxt_idx))) - nxt_idx += 1 + pending.append(pool.submit(self.func, item)) + while pending: + yield pending.popleft().result() + item = next(items, None) + if item is not None: + pending.append(pool.submit(self.func, item)) + finally: + pool.shutdown(wait=True, cancel_futures=True) + + # Writes happen here so shape/type errors reach the caller. Closing the + # generator also shuts down readers when stacking a batch fails. + with closing(processed_items()) as records, TQDM(total=len(self), desc=desc, leave=False) as pbar: + offset = 0 + for batch in batched(records, batch_size): + for index, record in enumerate(batch, offset): + values = (record,) if self._ram_was_single_tensor else record + if not isinstance(values, (tuple, list)) or len(values) != len(templates): + raise ValueError(f"Cached sample {index} has an inconsistent tensor structure.") + for value, template in zip(values, templates, strict=True): + if not isinstance(value, torch.Tensor) or value.shape != template.shape: + raise ValueError(f"Cached sample {index} has an inconsistent tensor shape.") + destination = slice(offset, offset + len(batch)) + if self._ram_was_single_tensor: + torch.stack(batch, out=stacked_tensors[0][destination]) else: - insert_buffer[idx] = data - - write_thread.join() - finally: - fetch_pool.shutdown(wait=False) + for target, values in zip(stacked_tensors, zip(*batch, strict=True), strict=True): + torch.stack(values, out=target[destination]) + offset += len(batch) + pbar.update(len(batch)) self._ram_cache = torch.utils.data.TensorDataset(*[t for t in stacked_tensors]) diff --git a/mini_trainer/data/loader.py b/mini_trainer/data/loader.py index 6ce3018..08877e5 100644 --- a/mini_trainer/data/loader.py +++ b/mini_trainer/data/loader.py @@ -128,6 +128,7 @@ def get_dataset_dataloader( # noqa: D103 device: torch.device | str = torch.device("cpu"), dtype: torch.dtype = torch.float32, cache: CACHE_MODE | str | int | None = None, + cache_workers: int | None = None, multilabel: bool = False, prefetch_factor: int | None = None, multiprocessing_context: str | None = None, @@ -158,7 +159,7 @@ def get_dataset_dataloader( # noqa: D103 for mode, data in zip(modes, metadata): if mode.strip().lower() == "train" and resample: raise NotImplementedError("Resampling is currently not supported.") - dset = LazyDataset(func=proc_path_label, items=(data["path"], data["class"]), cache=cache) + dset = LazyDataset(func=proc_path_label, items=(data["path"], data["class"]), cache=cache, cache_workers=cache_workers) datasets.append(dset) if cache is CACHE_MODE.CUDA: diff --git a/tests/test_integration_lazy_dataset.py b/tests/test_integration_lazy_dataset.py index 212a3ac..96aac17 100644 --- a/tests/test_integration_lazy_dataset.py +++ b/tests/test_integration_lazy_dataset.py @@ -109,3 +109,116 @@ def test_lazy_dataset_caching_exception_propagation(self): paths = ["ok1.png", "fail.png", "ok2.png"] with pytest.raises(ValueError, match="Simulated read failure"): LazyDataset(failing_loader, (paths,), cache="cpu") + + +@pytest.mark.parametrize("workers", [0, 1, 3]) +@pytest.mark.parametrize("structured", [False, True]) +def test_cache_decodes_once_and_preserves_order(workers, structured): + from collections import Counter + from threading import Lock + + calls, lock = Counter(), Lock() + + def reader(item): + index = int(item[0]) + with lock: + calls[index] += 1 + image = torch.full((2, 3), index, dtype=torch.uint8) + return (image, torch.tensor(index)) if structured else image + + dataset = LazyDataset(reader, (range(37),), cache="cpu", cache_workers=workers) + assert calls == Counter(range(37)) + for index in range(37): + result = dataset[index] + image = result[0] if structured else result + assert torch.equal(image, torch.full((2, 3), index, dtype=torch.uint8)) + if structured: + assert result[1].item() == index + + +def test_cache_read_ahead_is_bounded(monkeypatch): + from concurrent.futures import ThreadPoolExecutor + from threading import Event, Thread + + from mini_trainer.data import io + + release, window_ready = Event(), Event() + submitted, outcome = [], [] + + class RecordingPool(ThreadPoolExecutor): + def submit(self, func, item): + submitted.append(int(item[0])) + result = super().submit(func, item) + if len(submitted) == 4: + window_ready.set() + return result + + monkeypatch.setattr(io, "ThreadPoolExecutor", RecordingPool) + + def reader(item): + if item[0] == 1: + assert release.wait(10), "Test failed to release slow reader" + return torch.tensor(item[0]) + + def construct(): + try: + outcome.append(LazyDataset(reader, (range(100),), cache="cpu", cache_workers=2)) + except Exception as error: + outcome.append(error) + + constructor = Thread(target=construct) + constructor.start() + try: + assert window_ready.wait(10) + assert submitted == [1, 2, 3, 4] + finally: + release.set() + constructor.join(10) + assert not constructor.is_alive() + assert len(outcome) == 1 and isinstance(outcome[0], LazyDataset) + assert torch.equal(outcome[0][:], torch.arange(100)) + + +@pytest.mark.parametrize("failure", ["read", "shape", "arity"]) +@pytest.mark.parametrize("workers", [0, 2]) +def test_cache_failure_shuts_down_readers(failure, workers): + import threading + + before = set(threading.enumerate()) + + def reader(item): + if item[0] == 3: + if failure == "read": + raise ValueError("broken reader") + if failure == "shape": + return (torch.zeros(5),) + return (torch.zeros(2), torch.zeros(2)) + return (torch.zeros(2),) + + with pytest.raises((ValueError, RuntimeError)): + LazyDataset(reader, (range(100),), cache="cpu", cache_workers=workers) + assert not [thread for thread in threading.enumerate() if thread not in before and thread.name.startswith("mini-trainer-cache")] + + +def test_cache_automatic_workers_respect_small_cpu_budget(monkeypatch): + from mini_trainer.data import _workers, io + + monkeypatch.setattr(_workers, "_available_cpu_count", lambda: 4) + + def unexpected_pool(*args, **kwargs): + pytest.fail("Four available CPUs should not start cache reader threads") + + monkeypatch.setattr(io, "ThreadPoolExecutor", unexpected_pool) + dataset = LazyDataset(lambda item: torch.tensor(item[0]), (range(10),), cache="cpu") + assert torch.equal(dataset[:], torch.arange(10)) + + +@pytest.mark.parametrize("workers", [-1, True, 1.5]) +def test_cache_invalid_worker_count(workers): + with pytest.raises(ValueError, match="cache_workers"): + LazyDataset(lambda item: torch.tensor(0), ([0],), cache="cpu", cache_workers=workers) + + +def test_cache_rejects_incomplete_input_columns(): + with pytest.raises(ValueError, match="equal lengths"): + LazyDataset(lambda item: torch.tensor(item[0]), (range(3), range(2)), cache="cpu") diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index 715ce67..9f63a19 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -128,7 +128,7 @@ def test_cuda_cache_still_disables_loader_workers(metadata, monkeypatch, workers assert not loaders[0].persistent_workers -@pytest.mark.parametrize("available,expected", [(0, 1), (1, 1), (2, 1), (8, 6), (256, 128)]) +@pytest.mark.parametrize("available,expected", [(0, 0), (1, 0), (2, 0), (4, 0), (8, 4), (256, 16)]) def test_ram_cache_thread_budget_and_contents(monkeypatch, available, expected): monkeypatch.setattr(_workers, "_available_cpu_count", lambda: available) executor = data_io.ThreadPoolExecutor @@ -139,9 +139,9 @@ def capture_executor(**kwargs): return executor(**kwargs) monkeypatch.setattr(data_io, "ThreadPoolExecutor", capture_executor) - dataset = data_io.LazyDataset(lambda item: torch.tensor([item[0]]), (list(range(5)),), cache="cpu") - assert selected == [expected] - torch.testing.assert_close(dataset[:], torch.arange(5).reshape(5, 1)) + dataset = data_io.LazyDataset(lambda item: torch.tensor([item[0]]), (list(range(33)),), cache="cpu") + assert selected == ([expected] if expected else []) + torch.testing.assert_close(dataset[:], torch.arange(33).reshape(33, 1)) def test_distributed_loader_retains_spawn_and_sampler(monkeypatch): @@ -204,3 +204,36 @@ def test_cuda_transfer_batches_are_pinned(metadata, cache): torch.testing.assert_close(images.to(device, non_blocking=True).cpu(), images) _, inference = get_inference_dataloader(metadata["path"], resize_size=4, batch_size=2, num_workers=0, device=device) assert next(iter(inference)).is_pinned() + + +def test_cache_worker_override_reaches_dataset(metadata, monkeypatch): + def unexpected_pool(*args, **kwargs): + pytest.fail("cache_workers=0 must bypass the reader thread pool") + + monkeypatch.setattr(data_io, "ThreadPoolExecutor", unexpected_pool) + _, loaders = get_dataset_dataloader(metadata, resize_size=4, modes=("val",), cache="cpu", cache_workers=0, num_workers=0, batch_size=5) + _, labels = next(iter(loaders[0])) + assert labels.tolist() == list(range(5)) + + +def test_bounded_cuda_cache_matches_cpu_cache(metadata): + import os + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to validate CUDA cache construction") + if not torch.cuda.is_available(): + pytest.fail("CUDA checks requested but no CUDA device is accessible") + device = torch.device("cuda:0") + _, cpu_loaders = get_dataset_dataloader( + metadata, resize_size=4, modes=("val",), cache="cpu", cache_workers=0, num_workers=0, batch_size=5 + ) + with pytest.warns(UserWarning, match="CUDA caching"): + _, gpu_loaders = get_dataset_dataloader( + metadata, resize_size=4, modes=("val",), cache="cuda", cache_workers=2, num_workers=3, batch_size=5, device=device + ) + assert gpu_loaders[0].num_workers == 0 + expected = next(iter(cpu_loaders[0])) + actual = next(iter(gpu_loaders[0])) + for cpu, gpu in zip(expected, actual, strict=True): + assert gpu.device == device + torch.testing.assert_close(gpu.cpu(), cpu, rtol=0, atol=0) From 87ac5cd0d8c21e8e60e0c4163456b63a33dc4d08 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 22:37:41 +0200 Subject: [PATCH 009/155] feat: integrate opt-in CUDA INT8 training and checkpoint restoration Move integer kernels into the optional runtime backend, prepare eligible Linear weights before optimizer creation, and report floating-point coverage. Preserve selected weight ties and restore quantized parameter types from checkpoint recipes. Cover AMP, masked heads, inference aliases, regularization, diagnostics, MuonAuxAdamW and training resume; validate minimal and CUDA installed wheels. --- README.md | 4 + dev/benchmarks/README.md | 6 +- dev/benchmarks/_int8_weight.py | 103 +-------- dev/benchmarks/quantized_training.py | 41 +--- docs/quantization.md | 2 +- docs/quantized-training.md | 97 ++++++++ docs/roadmap.md | 6 +- docs/training-feature-validation.md | 5 +- mini_trainer/modeling/_quantized_training.py | 178 +++++++++++++++ mini_trainer/modeling/classifier.py | 7 +- mini_trainer/modeling/distance.py | 2 + mini_trainer/modeling/quantized_training.py | 150 +++++++++++++ mini_trainer/train.py | 27 +++ mini_trainer/training/loss.py | 2 + tests/test_quantized_training.py | 2 +- tests/test_quantized_training_model.py | 219 +++++++++++++++++++ 16 files changed, 705 insertions(+), 146 deletions(-) create mode 100644 docs/quantized-training.md create mode 100644 mini_trainer/modeling/_quantized_training.py create mode 100644 mini_trainer/modeling/quantized_training.py create mode 100644 tests/test_quantized_training_model.py diff --git a/README.md b/README.md index 5252748..244beb9 100644 --- a/README.md +++ b/README.md @@ -139,3 +139,7 @@ deferred. See [known limitations](docs/roadmap.md). An opt-in [PTQ and QAT Python API](docs/quantization.md) targets native x86 INT8 inference. This is an initial backend increment; CPU float32 QAT, integer inference and ordinary AMP are distinct capabilities. + +Opt-in [CUDA INT8 training](docs/quantized-training.md) now has an initial Linear +model/checkpoint integration. Its documented coverage and performance limits are +separate from [x86 PTQ/QAT inference](docs/quantization.md). diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 82d850b..d9a733d 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -151,8 +151,10 @@ and weight decay) and AdamW. CPU tests check update error against floating-point optimizer math and exact next-step continuation when weights, optimizer state and RNG are restored together. Foreach parameter updates currently dispatch per tensor; no fused-optimizer speed benefit is claimed. This does not establish the -repository's complete optimizer or checkpoint contracts. It is not an `mt_train` -feature yet, and a synthetic linear-stack MSE is not a convergence study. +repository's complete optimizer or checkpoint contracts. The backend now has an +initial [model and trainer integration](../../docs/quantized-training.md), including +checkpoint restoration. The linear-stack MSE remains a kernel probe, not a +convergence study. Local RTX 3080 Ti evidence (four 4096-wide layers, batch 2048, FP16 input/output, three warm-up steps, ten measured forward/backward/SGD steps): FP16 took 29.10 ms diff --git a/dev/benchmarks/_int8_weight.py b/dev/benchmarks/_int8_weight.py index 94ef79f..161cbf9 100644 --- a/dev/benchmarks/_int8_weight.py +++ b/dev/benchmarks/_int8_weight.py @@ -1,102 +1,5 @@ -"""Experimental INT8 parameter dispatch for the CUDA training probe. +"""Compatibility import for earlier developer probes.""" -Kept outside the runtime package until optimizer, checkpoint and model coverage -are established. Changes are local to this subclass, never TorchAO's dispatch. -""" +from mini_trainer.modeling._quantized_training import TrainingWeight -import hashlib -from pathlib import Path - -import torch -from torch.utils._python_dispatch import return_and_correct_aliasing -from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight - -# Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary -# graph key. Include both implementation files so changing backward math cannot -# reuse a graph compiled for an earlier version. Compute once when importing. -_IMPLEMENTATION_HASH = hashlib.sha256( - Path(__file__).read_bytes() + Path(__file__).with_name("quantized_training.py").read_bytes() -).hexdigest() - - -class TrainingWeight(Int8QuantizedTrainingLinearWeight): - """INT8 storage with integer linear arithmetic and ordinary optimizer updates.""" - - def _stable_hash_for_caching(self): - metadata = [ - (tuple(value.shape), tuple(value.stride()), str(value.dtype), str(value.device), value.requires_grad) - for value in (self, self.int_data, self.scale) - ] - return hashlib.sha256(repr((_IMPLEMENTATION_HASH, metadata)).encode()).hexdigest() - - -@TrainingWeight.implements_torch_function(torch.nn.functional.linear) -def linear(func, types, args, kwargs): - from .quantized_training import IntegerLinear - - inputs = args[0] if args else kwargs["input"] - weight = args[1] if len(args) > 1 else kwargs["weight"] - bias = args[2] if len(args) > 2 else kwargs.get("bias") - output = IntegerLinear.apply(inputs.reshape(-1, inputs.shape[-1]), weight) - output = output.reshape(*inputs.shape[:-1], weight.shape[0]) - return output if bias is None else output + bias - - -@TrainingWeight.implements([torch.ops.aten.detach.default, torch.ops.aten.clone.default]) -def preserve_type(func, types, args, kwargs): - original = args[0] - out = TrainingWeight(func(original.int_data, **kwargs), func(original.scale, **kwargs)) - return return_and_correct_aliasing(func, args, kwargs, out) - - -@TrainingWeight.implements(torch.ops.aten._to_copy.default) -def to_copy(func, types, args, kwargs): - original = args[0] - integer_kwargs = {key: value for key, value in kwargs.items() if key != "dtype"} - out = TrainingWeight(func(original.int_data, **integer_kwargs), func(original.scale, **kwargs)) - return return_and_correct_aliasing(func, args, kwargs, out) - - -@TrainingWeight.implements(torch.ops.aten.add.Tensor) -def add(func, types, args, kwargs): - # Coupled weight decay (SGD) adds the dequantized parameter to its gradient. - return func(*(value.dequantize() if isinstance(value, TrainingWeight) else value for value in args), **kwargs) - - -@TrainingWeight.implements(torch.ops.aten.mul_.Tensor) -def multiply_inplace(func, types, args, kwargs): - original, multiplier = args - # AdamW's decoupled decay is a scalar rescale. Preserve the integer codes - # exactly instead of adding a second stochastic rounding to every update. - if isinstance(multiplier, (float, int)): - original.scale.mul_(multiplier) - return original - return original.copy_(original.dequantize() * multiplier) - - -@TrainingWeight.implements(torch.ops.aten._foreach_add.List) -def foreach_add(func, types, args, kwargs): - return [torch.add(left, right, **kwargs) for left, right in zip(*args, strict=True)] - - -@TrainingWeight.implements(torch.ops.aten._foreach_add_.List) -def foreach_add_inplace(func, types, args, kwargs): - for left, right in zip(*args, strict=True): - left.add_(right, **kwargs) - return None - - -@TrainingWeight.implements(torch.ops.aten._foreach_mul_.Scalar) -def foreach_mul_inplace(func, types, args, kwargs): - for value in args[0]: - value.mul_(args[1]) - return None - - -@TrainingWeight.implements([torch.ops.aten._foreach_addcdiv_.Scalar, torch.ops.aten._foreach_addcdiv_.ScalarList]) -def foreach_addcdiv_inplace(func, types, args, kwargs): - factors = args[3] if len(args) > 3 else 1 - for index, (target, numerator, denominator) in enumerate(zip(*args[:3], strict=True)): - factor = factors[index] if isinstance(factors, (tuple, list)) else factors - target.addcdiv_(numerator, denominator, value=factor) - return None +__all__ = ["TrainingWeight"] diff --git a/dev/benchmarks/quantized_training.py b/dev/benchmarks/quantized_training.py index f6e05e6..304b6f5 100644 --- a/dev/benchmarks/quantized_training.py +++ b/dev/benchmarks/quantized_training.py @@ -1,4 +1,4 @@ -"""Experimental CUDA QT kernel probe; not yet integrated with mt_train. +"""CUDA QT kernel probe using the opt-in model backend. Uses INT8 stored weights (no floating-point master copy), INT8 saved linear inputs, and scaled INT8 forward/dgrad/wgrad GEMMs. Gradients and SGD update math @@ -14,48 +14,15 @@ import torch +from mini_trainer.modeling.quantized_training import IntegerLinear as IntegerLinear -def dependencies(): - from torchao.prototype.quantized_training.int8 import quantize_int8_rowwise - from torchao.prototype.quantized_training.int8_mm import scaled_int8_mm - from ._int8_weight import TrainingWeight +def dependencies(): + from mini_trainer.modeling._quantized_training import TrainingWeight, quantize_int8_rowwise, scaled_int8_mm return TrainingWeight, quantize_int8_rowwise, scaled_int8_mm -class IntegerLinear(torch.autograd.Function): - """Row-scaled INT8 GEMMs, including approximate input and weight gradients.""" - - @staticmethod - def forward(ctx, inputs, weight): - _, quantize, mm = dependencies() - quantized, scale = quantize(inputs) - ctx.save_for_backward(quantized, scale, weight.int_data, weight.scale) - return mm(quantized.contiguous(), weight.int_data.T, scale.contiguous(), weight.scale.contiguous()) - - @staticmethod - def backward(ctx, grad_output): - _, quantize, mm = dependencies() - inputs, input_scale, weight, weight_scale = ctx.saved_tensors - ones = torch.ones(weight.shape[1], device=grad_output.device, dtype=torch.float32) - grad_input = None - if ctx.needs_input_grad[0]: - # Weight scales lie along the contraction axis: absorb them into - # dY before its row quantization, not into the result columns. - quantized_grad, scale = quantize(grad_output.float() * weight_scale.float()) - grad_input = mm(quantized_grad.contiguous(), weight.contiguous(), scale.contiguous(), ones) - grad_weight = None - if ctx.needs_input_grad[1]: - # Similarly absorb saved activation scales into dY.T for dW. - quantized_grad, scale = quantize(grad_output.T.float() * input_scale.float()) - grad_weight = mm(quantized_grad.contiguous(), inputs.contiguous(), scale.contiguous(), ones) - return ( - grad_input.to(grad_output.dtype) if grad_input is not None else None, - grad_weight.to(weight_scale.dtype) if grad_weight is not None else None, - ) - - class Layer(torch.nn.Module): def __init__(self, width, quantized, dtype): super().__init__() diff --git a/docs/quantization.md b/docs/quantization.md index 40317a8..5f38231 100644 --- a/docs/quantization.md +++ b/docs/quantization.md @@ -2,7 +2,7 @@ This is an opt-in Python API for **static 8-bit weights and 8-bit activations**, using TorchAO PT2E. Actual quantized training with reduced memory and training -time is separate ongoing work; see the [QT/loader probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance). It supports post-training calibration (PTQ) and +time now has an initial [CUDA model integration](quantized-training.md); see the [QT/loader probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance). It supports post-training calibration (PTQ) and quantization-aware training (QAT). QAT uses fake quantization with float32 master parameters/gradients; it does not promise integer backward computation or reduced training memory. Converted inference executes native oneDNN integer Conv/Linear diff --git a/docs/quantized-training.md b/docs/quantized-training.md new file mode 100644 index 0000000..08b4d6d --- /dev/null +++ b/docs/quantized-training.md @@ -0,0 +1,97 @@ +# CUDA INT8 training integration + +The opt-in training path stores eligible Linear weights and saved linear inputs +in INT8 and uses integer matrix products for forward, input gradients and weight +gradients. It retains no floating-point master copy of those weights. Gradients, +optimizer state, biases, normalization, convolutions and auxiliary regularization +remain floating point. This is separate from fake-quantized QAT and x86 PTQ. + +Install the optional `quantization` extra while explicitly retaining the intended +PyTorch CUDA backend, as described in the README. The current implementation uses +TorchAO's experimental Triton kernels. It has been exercised on an RTX 3080 Ti; +CPU preparation and checkpoint inspection do not establish CPU execution support. + +```bash +mt_train -i /path/to/data --device cuda --quantized-training --dtype float16 --cache cpu --cache-workers 0 +``` + +`--quantized-training` prepares weights before building the optimizer and logs +coverage. `--cache-workers 0` makes cache construction synchronous; its selection +is separate from DataLoader workers. Ordinary training defaults are unchanged. + +The Python API also supports explicit module selection: + +```python +import torch +from mini_trainer.modeling.quantized_training import prepare_quantized_training + +# Load floating weights and move the model to its intended device first. +coverage = prepare_quantized_training(model) # in place, before optimizer creation +# Or: prepare_quantized_training(model, module_names=["encoder.projection"]) +optimizer = torch.optim.AdamW(model.parameters(), lr=1e-3) +``` + +The returned recipe lists quantized modules, skipped operations, remaining +floating-point parameters and physical versus reference weight storage. Automatic +selection covers ordinary `nn.Linear` modules, including their functional use by +Classifier heads. It preserves shared weights when every owner is selected. +Parametrized weights (including normalized heads) and weights shared with an +unselected operation stay floating point and are reported. Explicit unsupported +selections and models with no eligible weights fail before changing weights. +Quantizing the hidden linear layer of a convolutional classifier does not make +its convolutions integer operations. + +CUDA execution supports batched inputs, bias, masked classifier rows, +single-sample inference and float16/bfloat16 autocast with float32 parameters. +Float32 optimizer parameters avoid the FP16 AdamW epsilon underflow discussed in +the developer probe. Eager SGD, AdamW and the repository's MuonAuxAdamW update +paths are covered; fused optimizer variants are not established. Quantized +regularization uses a differentiable floating view of the represented weights. + +## Checkpoints and inference + +Prepared models include their recipe in `state_dict`. Ordinary `mt_train` +checkpoints retain INT8 parameter storage, and `Classifier.build(weights=...)` +restores the parameter types before loading weights. Known tensor classes are +allowed only within a scoped `weights_only=True` load. To resume through +`mt_train`, enable `--quantized-training` again so optimizer construction sees +the correct parameters. Existing stochastic-resume limitations still apply: +the trainer does not generally persist sampler or RNG state. + +For a custom architecture outside `Classifier.build`: + +```python +from mini_trainer.modeling.quantized_training import load_training_weights, restore_quantized_training + +state = load_training_weights("weights.pt", map_location="cpu") +restore_quantized_training(model, state) +model.load_state_dict(state) +``` + +Use the same architecture and intended dtype. This restores model state; create +and restore optimizer/scheduler/scaler state in their normal order separately. +The same model supports CUDA inference with `eval()` and `inference_mode()`. +ONNX export, checkpoint averaging, DDP/FSDP, quantized normalization and integer +convolution training are not established for this path. Distributed training and +EMA are rejected by the training entry point. + +## Evidence and remaining work + +The [developer probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance) +record both positive and negative workload-dependent results. Compiler cache keys +include backend source and tensor metadata to avoid reusing obsolete backward +graphs. Numerical checks include small gradients, compiled/eager agreement, +weight storage, optimizer updates, regularization and model-state restoration. + +Kernel-probe results do not establish a real-model speedup or convergence. The +integrated path still needs paired synthetic-oracle, MNIST and hierarchical Blair +runs, complete optimizer/resume coverage, end-to-end memory and throughput +measurements, and broader quantized operation coverage. These are requirements +for the overall QT goal, not conclusions implied by this initial integration. + +The integrated backend was rerun on the four-layer, 4096-wide, batch-2048 compiled +SGD probe: 15.61 ms/step and 319,063,552 peak allocated bytes for INT8, versus +30.05 ms and 386,139,648 bytes for FP16. These single-run numbers are about +1.93x faster and 17% lower peak memory for that workload, with nonzero gradients +checked before timing. They remain kernel-probe evidence, not a claim about +MNIST, Blair or typical convolutional models. diff --git a/docs/roadmap.md b/docs/roadmap.md index 5760054..3cf3137 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -93,8 +93,10 @@ QT must reduce retained training storage and demonstrate lower peak memory and f training on supported workloads. QAT with floating-point master weights is a separate capability and does not complete this target. The initial CUDA integer forward/backward kernel probe and cached-loader benchmark are documented in [the benchmark guide](../dev/benchmarks/README.md). -Optimizer support, checkpoint integration, real-data convergence and end-to-end -measurements remain required before claiming a supported QT training path. +An initial [CUDA INT8 Linear integration](quantized-training.md) connects model +preparation and checkpoint loading to the training entry point. Broader operator +and optimizer coverage, real-data convergence and end-to-end measurements remain +required to complete this target. Loader hardening, float16/bfloat16 AMP and benchmark infrastructure do not complete that target. The implementation and comparison plan is in [training feature validation](training-feature-validation.md). diff --git a/docs/training-feature-validation.md b/docs/training-feature-validation.md index 804ebb8..962ff69 100644 --- a/docs/training-feature-validation.md +++ b/docs/training-feature-validation.md @@ -9,8 +9,9 @@ nonfunctional and excluded from these experiments; repair is deferred. The priority is now QT that lowers training memory and increases speed, together with faster data loading for floating-point and quantized workloads. PTQ/QAT do not satisfy that objective. See the [QT and loader probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance); -optimizer integration, checkpoint/resume, convergence and end-to-end measurement -remain requirements, not optional follow-ups. +an initial [model/trainer integration](quantized-training.md) is available, while +broader optimizer/resume coverage, convergence and end-to-end measurement remain +requirements, not optional follow-ups. The initial [INT8 PTQ/QAT Python backend](quantization.md) is implemented on the `quant` branch. Its CPU tests establish a training-to-integer-inference path; diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py new file mode 100644 index 0000000..3c990a2 --- /dev/null +++ b/mini_trainer/modeling/_quantized_training.py @@ -0,0 +1,178 @@ +"""Optional CUDA INT8 training backend. Import through quantized_training. + +Changes are local to this subclass, never TorchAO's global dispatch. +""" + +import hashlib +from pathlib import Path + +import torch +from torch.utils._python_dispatch import return_and_correct_aliasing +from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise +from torchao.prototype.quantized_training.int8_mm import scaled_int8_mm as _native_scaled_int8_mm + +# Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary +# graph key. Include the backend implementation so changing backward math cannot +# reuse a graph compiled for an earlier version. Compute once when importing. +_IMPLEMENTATION_HASH = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() + + +class TrainingWeight(Int8QuantizedTrainingLinearWeight): + """INT8 storage with integer linear arithmetic and ordinary optimizer updates.""" + + _is_quantized_training = True + + def dequantize(self): + return _Dequantize.apply(self) + + def _stable_hash_for_caching(self): + metadata = [ + (tuple(value.shape), tuple(value.stride()), str(value.dtype), str(value.device), value.requires_grad) + for value in (self, self.int_data, self.scale) + ] + return hashlib.sha256(repr((_IMPLEMENTATION_HASH, metadata)).encode()).hexdigest() + + +@TrainingWeight.implements_torch_function(torch.nn.functional.linear) +def linear(func, types, args, kwargs): + inputs = args[0] if args else kwargs["input"] + weight = args[1] if len(args) > 1 else kwargs["weight"] + bias = args[2] if len(args) > 2 else kwargs.get("bias") + if inputs.device.type == "cuda" and torch.is_autocast_enabled("cuda"): + inputs = inputs.to(torch.get_autocast_dtype("cuda")) + output = IntegerLinear.apply(inputs.reshape(-1, inputs.shape[-1]), weight) + output = output.reshape(*inputs.shape[:-1], weight.shape[0]) + return output if bias is None else output + bias.to(output.dtype) + + +@TrainingWeight.implements([torch.ops.aten.detach.default, torch.ops.aten.clone.default]) +def preserve_type(func, types, args, kwargs): + original = args[0] + if func == torch.ops.aten.detach.default: + # Detach aliases the original version counter. Constructing its wrapper + # under inference_mode would discard that counter for normal weights. + with torch.inference_mode(original.is_inference()): + out = TrainingWeight(func(original.int_data, **kwargs), func(original.scale, **kwargs)) + else: + out = TrainingWeight(func(original.int_data, **kwargs), func(original.scale, **kwargs)) + return return_and_correct_aliasing(func, args, kwargs, out) + + +@TrainingWeight.implements(torch.ops.aten._to_copy.default) +def to_copy(func, types, args, kwargs): + original = args[0] + integer_kwargs = {key: value for key, value in kwargs.items() if key != "dtype"} + out = TrainingWeight(func(original.int_data, **integer_kwargs), func(original.scale, **kwargs)) + return return_and_correct_aliasing(func, args, kwargs, out) + + +@TrainingWeight.implements(torch.ops.aten.add.Tensor) +def add(func, types, args, kwargs): + # Coupled weight decay (SGD) adds the dequantized parameter to its gradient. + return func(*(value.dequantize() if isinstance(value, TrainingWeight) else value for value in args), **kwargs) + + +@TrainingWeight.implements(torch.ops.aten.mul_.Tensor) +def multiply_inplace(func, types, args, kwargs): + original, multiplier = args + # AdamW's decoupled decay is a scalar rescale. Preserve the integer codes + # exactly instead of adding a second stochastic rounding to every update. + if isinstance(multiplier, (float, int)): + original.scale.mul_(multiplier) + return original + return original.copy_(original.dequantize() * multiplier) + + +@TrainingWeight.implements(torch.ops.aten._foreach_add.List) +def foreach_add(func, types, args, kwargs): + return [torch.add(left, right, **kwargs) for left, right in zip(*args, strict=True)] + + +@TrainingWeight.implements(torch.ops.aten._foreach_add_.List) +def foreach_add_inplace(func, types, args, kwargs): + for left, right in zip(*args, strict=True): + left.add_(right, **kwargs) + return None + + +@TrainingWeight.implements(torch.ops.aten._foreach_mul_.Scalar) +def foreach_mul_inplace(func, types, args, kwargs): + for value in args[0]: + value.mul_(args[1]) + return None + + +@TrainingWeight.implements([torch.ops.aten._foreach_addcdiv_.Scalar, torch.ops.aten._foreach_addcdiv_.ScalarList]) +def foreach_addcdiv_inplace(func, types, args, kwargs): + factors = args[3] if len(args) > 3 else 1 + for index, (target, numerator, denominator) in enumerate(zip(*args[:3], strict=True)): + factor = factors[index] if isinstance(factors, (tuple, list)) else factors + target.addcdiv_(numerator, denominator, value=factor) + return None + + +class IntegerLinear(torch.autograd.Function): + """Row-scaled INT8 GEMMs, including approximate input and weight gradients.""" + + @staticmethod + def forward(ctx, inputs, weight): + quantize, mm = quantize_int8_rowwise, scaled_int8_mm + if inputs.device.type != "cuda": + raise ValueError("INT8 training linear execution requires CUDA.") + quantized, scale = quantize(inputs.float()) + ctx.weight_dtype = weight.dtype + ctx.save_for_backward(quantized, scale, weight.int_data, weight.scale.float()) + return mm(quantized.contiguous(), weight.int_data.T, scale.contiguous(), weight.scale.float().contiguous()).to(inputs.dtype) + + @staticmethod + def backward(ctx, grad_output): + quantize, mm = quantize_int8_rowwise, scaled_int8_mm + inputs, input_scale, weight, weight_scale = ctx.saved_tensors + ones = torch.ones(weight.shape[1], device=grad_output.device, dtype=torch.float32) + grad_input = None + if ctx.needs_input_grad[0]: + # Weight scales lie along the contraction axis: absorb them into + # dY before its row quantization, not into the result columns. + quantized_grad, scale = quantize(grad_output.float() * weight_scale.float()) + grad_input = mm(quantized_grad.contiguous(), weight.contiguous(), scale.contiguous(), ones) + grad_weight = None + if ctx.needs_input_grad[1]: + # Similarly absorb saved activation scales into dY.T for dW. + quantized_grad, scale = quantize(grad_output.T.float() * input_scale.float()) + grad_weight = mm(quantized_grad.contiguous(), inputs.contiguous(), scale.contiguous(), ones) + return ( + grad_input.to(grad_output.dtype) if grad_input is not None else None, + grad_weight.to(ctx.weight_dtype) if grad_weight is not None else None, + ) + + +def scaled_int8_mm(left, right, row_scale, column_scale): + # TorchAO's Python validation squeezes the row scale, rejecting M=1 even + # though its native kernel supports it. Validate that case without squeeze. + if left.shape[0] == 1: + if row_scale.shape != (1,) or column_scale.shape != (right.shape[1],): + raise ValueError("Invalid INT8 matrix scales.") + if left.shape[1] != right.shape[0] or row_scale.dtype != column_scale.dtype: + raise ValueError("Incompatible INT8 matrix shapes or scale dtypes.") + return torch.ops.torchao.scaled_int8_mm(left, right, row_scale, column_scale) + return _native_scaled_int8_mm(left, right, row_scale, column_scale) + + +@TrainingWeight.implements(torch.ops.aten.index_select.default) +def select_weight_rows(func, types, args, kwargs): + weight, dimension, indices = args + if dimension not in (0, -2): + raise ValueError("Quantized training weights support row selection only.") + return TrainingWeight(weight.int_data.index_select(0, indices), weight.scale.index_select(0, indices)) + + +class _Dequantize(torch.autograd.Function): + """Expose represented values to floating-point auxiliary losses with STE.""" + + @staticmethod + def forward(ctx, weight): + return weight.int_data * weight.scale.view(-1, 1) + + @staticmethod + def backward(ctx, gradient): + return gradient diff --git a/mini_trainer/modeling/classifier.py b/mini_trainer/modeling/classifier.py index 404317b..5fbc589 100644 --- a/mini_trainer/modeling/classifier.py +++ b/mini_trainer/modeling/classifier.py @@ -16,6 +16,7 @@ from .architectures import get_model from .prior import prior_from_labels +from .quantized_training import load_training_weights, restore_quantized_training try: from torch.nn.utils.parametrizations import weight_norm @@ -156,6 +157,9 @@ def _load_from_state_dict(self, state_dict, prefix, local_metadata, strict, miss self._dirty_cache.clear() return retval + def _on_quantized_training_prepared(self): + self._dirty_cache.clear() + def set_active_features(self, indices: list[int] | torch.Tensor | np.ndarray | None = None): """Mask a selection of output features (classes). @@ -314,6 +318,7 @@ def load( for k, v in cfg.items(): setattr(architecture, f"_{k}", v) if state is not None: + restore_quantized_training(architecture, state) try: load_result = architecture.load_state_dict(state, strict=strict) except RuntimeError as e: @@ -349,7 +354,7 @@ def build( # Parse metadata stored in .pt file if available if weights is not None: if isinstance(weights, str): - state = torch.load(f=weights, map_location=device, weights_only=True) + state = load_training_weights(weights, map_location=device) state = state.get("model", state) # type: ignore else: state = weights diff --git a/mini_trainer/modeling/distance.py b/mini_trainer/modeling/distance.py index 08966f0..45d1363 100644 --- a/mini_trainer/modeling/distance.py +++ b/mini_trainer/modeling/distance.py @@ -7,6 +7,8 @@ def _class_similarity(W: torch.Tensor, cdf: bool = True) -> torch.Tensor: + if getattr(W, "_is_quantized_training", False): + W = W.dequantize() W = W.detach().clone().float() WN = W.norm(2, 1, True) Z = cosine_to_zscore((W @ W.T) / (WN @ WN.T), W.shape[1]) diff --git a/mini_trainer/modeling/quantized_training.py b/mini_trainer/modeling/quantized_training.py new file mode 100644 index 0000000..bed80b7 --- /dev/null +++ b/mini_trainer/modeling/quantized_training.py @@ -0,0 +1,150 @@ +"""Opt-in CUDA INT8 weight/activation training for ordinary linear modules. + +Prepare before constructing an optimizer. Biases, normalization, convolutions, +and unselected weights remain floating point and are reported explicitly. +""" + +import torch +from torch import nn + + +def _backend(): + try: + from . import _quantized_training + except ImportError as error: + raise ImportError("CUDA INT8 training requires mini_trainer[quantization] and a compatible CUDA/Triton installation.") from error + return _quantized_training + + +class IntegerLinear: + """Lazy access to the backend autograd function for numerical validation.""" + + @staticmethod + def apply(inputs, weight): + return _backend().IntegerLinear.apply(inputs, weight) + + +def prepare_quantized_training(model: nn.Module, *, module_names=None) -> dict: + """Replace eligible linear weights in place and return the coverage recipe. + + No floating master weights are retained. Recreate optimizers after this call. + CPU preparation supports checkpoint inspection; execution requires CUDA. + Explicitly selected unsupported modules raise instead of silently skipping. + Parametrized weights and weights shared with unselected operations remain + floating point until their quantized training contracts are implemented. + """ + backend = _backend() + modules = dict(model.named_modules(remove_duplicate=False)) + requested = None if module_names is None else set([module_names] if isinstance(module_names, str) else module_names) + if requested is not None and requested - modules.keys(): + raise ValueError(f"Unknown quantized training modules: {sorted(requested - modules.keys())}") + candidates, skipped = {}, {} + for name, module in modules.items(): + if requested is not None and name not in requested: + continue + reason = None + if not isinstance(module, nn.Linear): + if requested is not None or isinstance(module, (nn.Conv1d, nn.Conv2d, nn.Conv3d)): + reason = "integer training is currently implemented for Linear" + else: + continue + elif nn.utils.parametrize.is_parametrized(module, "weight"): + reason = "parametrized weights require a separate quantized gradient contract" + elif not isinstance(module.weight, nn.Parameter) or module.weight.numel() == 0: + reason = "requires a nonempty weight Parameter" + elif module.weight.dtype not in (torch.float16, torch.bfloat16, torch.float32): + reason = "requires float16, bfloat16 or float32 compute metadata" + if reason: + skipped[name] = reason + else: + candidates[name] = (module, module.weight) + selected_owners = {(id(module), "weight") for module, _ in candidates.values()} + owners = {} + for module in modules.values(): + for name, parameter in module.named_parameters(recurse=False): + owners.setdefault(id(parameter), set()).add((id(module), name)) + for name, (_, parameter) in list(candidates.items()): + if owners[id(parameter)] - selected_owners: + skipped[name] = "weight is shared with an unselected operation" + del candidates[name] + if requested is not None and skipped: + raise ValueError(f"Unsupported quantized training selection: {skipped}") + if not candidates: + raise ValueError(f"No eligible Linear weights for quantized training. Unsupported modules: {skipped}") + # Validate before mutating the caller's model. + for _, parameter in candidates.values(): + values = parameter.dequantize() if isinstance(parameter, backend.TrainingWeight) else parameter + if not torch.isfinite(values).all(): + raise ValueError("Quantized training requires finite initial weights.") + replacements = {} + for module, parameter in candidates.values(): + if id(parameter) not in replacements: + replacements[id(parameter)] = ( + parameter + if isinstance(parameter, backend.TrainingWeight) + else nn.Parameter(backend.TrainingWeight.from_float(parameter), requires_grad=parameter.requires_grad) + ) + module.weight = replacements[id(parameter)] + for module in modules.values(): + invalidate = getattr(module, "_on_quantized_training_prepared", None) + if invalidate is not None: + invalidate() + report = { + "schema_version": 1, + "backend": "cuda-int8-linear", + "quantized_modules": sorted(candidates), + "skipped_modules": skipped, + "floating_parameter_names": [ + name for name, parameter in model.named_parameters() if not isinstance(parameter, backend.TrainingWeight) + ], + "weight_bits": 8, + "saved_linear_input_bits": 8, + "floating_master_weights": False, + "quantized_weight_bytes": sum( + weight.int_data.numel() + weight.scale.numel() * weight.scale.element_size() for weight in replacements.values() + ), + "reference_weight_bytes": sum(weight.numel() * weight.element_size() for weight in replacements.values()), + } + model._quantized_training_recipe = report + if not getattr(model, "_quantized_training_hooks", False): + model.register_state_dict_post_hook(_store_recipe) + model.register_load_state_dict_pre_hook(_consume_recipe) + model._quantized_training_hooks = True + return report + + +def _store_recipe(module, state_dict, prefix, local_metadata): + import copy + + state_dict[prefix + "_quantized_training"] = copy.deepcopy(module._quantized_training_recipe) + + +def _consume_recipe(module, state_dict, prefix, local_metadata, strict, missing_keys, unexpected_keys, error_msgs): + recipe = state_dict.pop(prefix + "_quantized_training", None) + expected = module._quantized_training_recipe + if recipe is None or any(recipe.get(key) != expected[key] for key in ("schema_version", "backend", "quantized_modules")): + error_msgs.append("Quantized training checkpoint recipe does not match the prepared model.") + + +def restore_quantized_training(model: nn.Module, state_dict: dict) -> None: + """Restore quantized parameter types before loading a model state dictionary.""" + for key, recipe in state_dict.items(): + if key == "_quantized_training" or key.endswith("._quantized_training"): + if not isinstance(recipe, dict) or recipe.get("schema_version") != 1 or recipe.get("backend") != "cuda-int8-linear": + raise ValueError("Unsupported quantized training checkpoint recipe.") + target = model if key == "_quantized_training" else model.get_submodule(key.removesuffix("._quantized_training")) + prepare_quantized_training(target, module_names=recipe["quantized_modules"]) + + +def load_training_weights(path, *, map_location="cpu"): + """Load ordinary or INT8 model weights using a scoped known-class allowlist.""" + import pickle + + try: + return torch.load(path, map_location=map_location, weights_only=True) + except pickle.UnpicklingError as error: + name = "mini_trainer.modeling._quantized_training.TrainingWeight" + if name not in str(error): + raise + with torch.serialization.safe_globals([_backend().TrainingWeight]): + return torch.load(path, map_location=map_location, weights_only=True) diff --git a/mini_trainer/train.py b/mini_trainer/train.py index 0d9cba5..18406bd 100644 --- a/mini_trainer/train.py +++ b/mini_trainer/train.py @@ -49,6 +49,7 @@ def main( # noqa: D417 dtype: str | torch.dtype = "float16", ema: bool = False, compile: bool = False, + quantized_training: bool = False, seed: int | None = None, builder: type[BaseBuilder] = BaseBuilder, spec_model_dataloader_kwargs: dict[str, Any] = {}, @@ -195,6 +196,17 @@ def main( # noqa: D417 **{**class_spec_data, **model_builder_kwargs}, ) validate_type(nn_model, torch.nn.Module) + if quantized_training or getattr(nn_model, "_quantized_training_recipe", None): + from mini_trainer.modeling.quantized_training import prepare_quantized_training + + if device.type != "cuda": + raise ValueError("Quantized training requires a CUDA device.") + if ddp_info and ddp_info.get("world_size", 1) > 1: + raise NotImplementedError("Distributed INT8 training is not validated yet.") + if ema: + raise ValueError("EMA is not supported for quantized training.") + coverage = prepare_quantized_training(nn_model) + log.info(f"INT8 training coverage: {coverage}") log.info(f"Using model `{nn_model.__class__.__name__}` with head `{classification_module(nn_model).__class__.__name__}`") # Resolve input size if not explicitly set @@ -256,6 +268,8 @@ def main( # noqa: D417 else: checkpoint_data = average_checkpoints(checkpoint_files, map_location=device, weights_only=False) assert checkpoint_data is not None + if "_quantized_training" in checkpoint_data["model"] and not getattr(nn_model, "_quantized_training_recipe", None): + raise ValueError("Resume an INT8 training checkpoint with --quantized-training enabled.") nn_model.load_state_dict(checkpoint_data["model"]) optimizer.load_state_dict(checkpoint_data["optimizer"]) lr_scheduler.load_state_dict(checkpoint_data["lr_scheduler"]) @@ -535,7 +549,20 @@ def cli(description="Train a classifier", **extra_kwargs): # noqa: D103 'Valid options are `None`, "disk", "cpu", "cuda" or "guess" (CUDA not supported yet).\n' "Mainly relevant for inefficiently stored training data or slow filesystems.", ) + cfg_args.add_argument( + "--cache-workers", + type=int, + default=None, + dest="dataloader_builder_kwargs.cache_workers", + help="Cache construction reader threads; 0 is synchronous, automatic selection is capped at 16.", + ) cfg_args.add_argument("--device", type=str, default=None, required=False, help='Device used for training (default="cuda").') + cfg_args.add_argument( + "--quantized-training", + action="store_true", + dest="quantized_training", + help="Opt into CUDA INT8 linear training; reports remaining floating-point operations.", + ) cfg_args.add_argument( "--compile", action="store_true", diff --git a/mini_trainer/training/loss.py b/mini_trainer/training/loss.py index b2a11c5..c655ca9 100644 --- a/mini_trainer/training/loss.py +++ b/mini_trainer/training/loss.py @@ -135,6 +135,8 @@ def class_weight_distribution_regularization(W: torch.Tensor, sparse: bool = Tru Returns: A scalar tensor representing the regularization loss. """ + if getattr(W, "_is_quantized_training", False): + W = W.dequantize() # Select a subset of classes to regularize _n = min(len(W), max(32, 2 * round(len(W) ** 0.5))) if sparse and _n < len(W): diff --git a/tests/test_quantized_training.py b/tests/test_quantized_training.py index c6171e6..acc0a28 100644 --- a/tests/test_quantized_training.py +++ b/tests/test_quantized_training.py @@ -150,7 +150,7 @@ def test_integer_linear_module_cuda(dtype, epsilon): def test_integer_compiler_key_tracks_code_and_metadata(monkeypatch): pytest.importorskip("torchao") - from dev.benchmarks import _int8_weight + from mini_trainer.modeling import _quantized_training as _int8_weight weight = _int8_weight.TrainingWeight.from_float(torch.randn(8, 16)) key = weight._stable_hash_for_caching() diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py new file mode 100644 index 0000000..d828945 --- /dev/null +++ b/tests/test_quantized_training_model.py @@ -0,0 +1,219 @@ +"""Model selection, serialization and CUDA execution contracts for real QT.""" + +import importlib.util +import os + +import pytest +import torch +from torch import nn + +from mini_trainer.modeling.quantized_training import load_training_weights, prepare_quantized_training, restore_quantized_training + +pytestmark = pytest.mark.skipif(importlib.util.find_spec("torchao") is None, reason="Install mini_trainer[quantization]") + + +def test_selection_preserves_ties_and_reports_float_operations(): + from mini_trainer.modeling._quantized_training import TrainingWeight + + model = nn.ModuleDict({"a": nn.Linear(8, 8), "b": nn.Linear(8, 8), "conv": nn.Conv2d(3, 3, 1), "norm": nn.Linear(8, 2)}) + model["b"].weight = model["a"].weight + nn.utils.parametrizations.weight_norm(model["norm"]) + report = prepare_quantized_training(model) + assert report["quantized_modules"] == ["a", "b"] + assert set(report["skipped_modules"]) == {"conv", "norm"} + assert model["a"].weight is model["b"].weight + assert isinstance(model["a"].weight, TrainingWeight) + assert report["quantized_weight_bytes"] < report["reference_weight_bytes"] + assert "conv.weight" in report["floating_parameter_names"] + state = model.state_dict() + assert state["_quantized_training"]["quantized_modules"] == ["a", "b"] + + +def test_invalid_selection_is_not_partially_applied(): + model = nn.Sequential(nn.Linear(8, 8), nn.Conv2d(3, 3, 1)) + before = model[0].weight + with pytest.raises(ValueError, match="Unsupported"): + prepare_quantized_training(model, module_names=["0", "1"]) + assert model[0].weight is before + model = nn.ModuleDict({"embedding": nn.Embedding(8, 8), "linear": nn.Linear(8, 8)}) + model["linear"].weight = model["embedding"].weight + with pytest.raises(ValueError, match="shared"): + prepare_quantized_training(model, module_names=["linear"]) + assert model["linear"].weight is model["embedding"].weight + + +def test_quantized_checkpoint_restores_storage_and_recipe(tmp_path): + from mini_trainer.modeling._quantized_training import TrainingWeight + + original = nn.Sequential(nn.Linear(8, 16), nn.ReLU(), nn.Linear(16, 3)) + prepare_quantized_training(original) + path = tmp_path / "weights.pt" + torch.save(original.state_dict(), path) + safe_before = set(torch.serialization.get_safe_globals()) + state = load_training_weights(path) + assert set(torch.serialization.get_safe_globals()) == safe_before + restored = nn.Sequential(nn.Linear(8, 16), nn.ReLU(), nn.Linear(16, 3)) + restore_quantized_training(restored, state) + restored.load_state_dict(state) + for left, right in zip(original.parameters(), restored.parameters(), strict=True): + if isinstance(left, TrainingWeight): + assert isinstance(right, TrainingWeight) + assert torch.equal(left.int_data, right.int_data) + assert torch.equal(left.scale, right.scale) + else: + torch.testing.assert_close(left, right, rtol=0, atol=0) + state["_quantized_training"]["quantized_modules"] = [] + with pytest.raises(RuntimeError, match="recipe"): + restored.load_state_dict(state) + + +def cuda(): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for model QT execution") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + return torch.device("cuda:0") + + +@pytest.mark.parametrize("autocast_dtype", [torch.float16, torch.bfloat16]) +def test_model_autocast_single_sample_and_gradients(autocast_dtype): + model = nn.Sequential(nn.Linear(64, 32), nn.ReLU(), nn.Linear(32, 1)).to(cuda()) + prepare_quantized_training(model) + inputs = torch.randn(1, 64, device=cuda()) + with torch.autocast("cuda", dtype=autocast_dtype): + output = model(inputs) + assert output.dtype == autocast_dtype and output.shape == (1, 1) + output.float().square().sum().backward() + assert all(parameter.grad is not None and torch.isfinite(parameter.grad).all() for parameter in model.parameters()) + assert all(parameter.grad.dtype == torch.float32 for parameter in model.parameters()) + optimizer = torch.optim.AdamW(model.parameters(), lr=0.001) + optimizer.step() + with torch.no_grad(), torch.autocast("cuda", dtype=autocast_dtype): + assert torch.isfinite(model(inputs)).all() + + +def test_masked_classifier_quantized_training_and_eval_cache(): + from mini_trainer.modeling import Classifier + from mini_trainer.modeling._quantized_training import TrainingWeight + + model = Classifier(64, 4, hidden=32, normalized=False).to(cuda()).eval() + inputs = torch.randn(4, 64, device=cuda()) + model(inputs) # Populate the old floating-point evaluation cache. + prepare_quantized_training(model) + with torch.no_grad(): + model(inputs) + assert isinstance(model._linear_weight, TrainingWeight) + model.set_active_features([0, 2, 3]) + model.train() + model(inputs).square().mean().backward() + assert model.linear.weight.grad is not None + assert torch.count_nonzero(model.linear.weight.grad[1]) == 0 + assert torch.count_nonzero(model.linear.weight.grad[[0, 2, 3]]) > 0 + + +def test_mixed_muon_adamw_updates_and_counter(): + from mini_trainer.training.muon import MuonAuxAdamW + + model = nn.Linear(8, 8) + prepare_quantized_training(model) + optimizer = MuonAuxAdamW([{"name": "mixed", "params": list(model.parameters())}], lr=0.01, weight_decay=0.1) + before = model.weight.int_data.clone() + for parameter in model.parameters(): + parameter.grad = torch.randn(parameter.shape) + optimizer.step() + assert optimizer._step_count == 1 + assert optimizer.muon is not None and optimizer.adamw is not None + assert not torch.equal(before, model.weight.int_data) + state = optimizer.state_dict() + restored = MuonAuxAdamW([{"name": "mixed", "params": list(model.parameters())}], lr=0.01, weight_decay=0.1) + restored.load_state_dict(state) + # The existing counter is process-local, not checkpoint state. Its role is + # successful-step detection; restoring must not change that contract. + before_count = restored._step_count + restored.step() + assert restored._step_count == before_count + 1 + + +def test_training_entrypoint_checkpoint_and_inference(tmp_path): + from mini_trainer.modeling import Classifier + from mini_trainer.modeling._quantized_training import TrainingWeight + from mini_trainer.train import main + from tests.test_checkpoint_contract import DeterministicBuilder + from tests.test_integration_train import TinyMockModel + + device = cuda() + for label in ("class_a", "class_b"): + (tmp_path / "data" / label).mkdir(parents=True) + args = { + "input": str(tmp_path / "data"), + "output": str(tmp_path), + "name": "quantized", + "epochs": 1, + "device": device, + "dtype": "float16", + "quantized_training": True, + "seed": 42, + "builder": DeterministicBuilder, + "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": False}, + "dataloader_builder_kwargs": {"batch_size": 4}, + "lr_schedule_builder_kwargs": {"warmup_epochs": 0}, + "ema": False, + "logger_builder_kwargs": {"verbose": False}, + } + main(**args) + path = tmp_path / "quantized/weights/last.pt" + restored, preprocess = Classifier.build(weights=str(path), device=device) + assert any(isinstance(parameter, TrainingWeight) for parameter in restored.parameters()) + images = torch.randn(1, 3, 5, 5, device=device) + restored.eval() + with torch.no_grad(): + assert torch.isfinite(restored(preprocess(images))).all() + state = load_training_weights(tmp_path / "quantized/weights/checkpoint_last.pth") + assert state["optimizer"] and state["scaler"] + # Resume through the ordinary training entry point. This checks state + # plumbing, not identical stochastic continuation with a changed schedule. + args["epochs"] = 2 + args["name"] = "resumed" + args["checkpoint"] = str(tmp_path / "quantized/weights/checkpoint_last.pth") + args["model_builder_kwargs"]["model_type"] = TinyMockModel() + main(**args) + resumed = load_training_weights(tmp_path / "resumed/weights/checkpoint_last.pth") + assert resumed["epoch"] == 1 + assert isinstance(resumed["model"]["fc.linear.weight"], TrainingWeight) + + +def test_quantized_regularizer_retains_weight_gradients(): + from mini_trainer.training.loss import class_weight_distribution_regularization + + model = nn.Linear(8, 4) + prepare_quantized_training(model) + reference = model.weight.dequantize().detach().requires_grad_() + expected = class_weight_distribution_regularization(reference, sparse=False) + actual = class_weight_distribution_regularization(model.weight, sparse=False) + expected.backward() + actual.backward() + torch.testing.assert_close(actual, expected) + torch.testing.assert_close(model.weight.grad, reference.grad) + + +def test_quantized_detach_in_inference_mode_preserves_alias(): + model = nn.Linear(8, 4) + prepare_quantized_training(model) + with torch.inference_mode(): + detached = model.weight.detach() + assert not detached.is_inference() + assert detached.int_data.data_ptr() == model.weight.int_data.data_ptr() + with torch.no_grad(): + model.weight.mul_(0.9) + assert torch.equal(detached.scale, model.weight.scale) + + +def test_quantized_class_similarity_uses_represented_values(): + from mini_trainer.modeling.distance import _class_similarity + + model = nn.Linear(8, 4) + prepare_quantized_training(model) + with torch.inference_mode(): + actual = _class_similarity(model.weight) + expected = _class_similarity(model.weight.dequantize()) + torch.testing.assert_close(actual, expected) From e1844e5eb28d32d1e4281d6ea56ae362e60c86ab Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 22:59:26 +0200 Subject: [PATCH 010/155] fix: preserve quantized checkpoint selection and coverage on reload --- mini_trainer/modeling/quantized_training.py | 5 ++++- mini_trainer/train.py | 2 +- tests/test_quantized_training_model.py | 10 ++++++++++ 3 files changed, 15 insertions(+), 2 deletions(-) diff --git a/mini_trainer/modeling/quantized_training.py b/mini_trainer/modeling/quantized_training.py index bed80b7..390d071 100644 --- a/mini_trainer/modeling/quantized_training.py +++ b/mini_trainer/modeling/quantized_training.py @@ -133,7 +133,10 @@ def restore_quantized_training(model: nn.Module, state_dict: dict) -> None: if not isinstance(recipe, dict) or recipe.get("schema_version") != 1 or recipe.get("backend") != "cuda-int8-linear": raise ValueError("Unsupported quantized training checkpoint recipe.") target = model if key == "_quantized_training" else model.get_submodule(key.removesuffix("._quantized_training")) - prepare_quantized_training(target, module_names=recipe["quantized_modules"]) + restored = prepare_quantized_training(target, module_names=recipe["quantized_modules"]) + # Explicit restoration selects only recorded weights, so keep the + # original reasons why other modules were not selected. + restored["skipped_modules"] = dict(recipe.get("skipped_modules", {})) def load_training_weights(path, *, map_location="cpu"): diff --git a/mini_trainer/train.py b/mini_trainer/train.py index 18406bd..4a0d54d 100644 --- a/mini_trainer/train.py +++ b/mini_trainer/train.py @@ -205,7 +205,7 @@ def main( # noqa: D417 raise NotImplementedError("Distributed INT8 training is not validated yet.") if ema: raise ValueError("EMA is not supported for quantized training.") - coverage = prepare_quantized_training(nn_model) + coverage = getattr(nn_model, "_quantized_training_recipe", None) or prepare_quantized_training(nn_model) log.info(f"INT8 training coverage: {coverage}") log.info(f"Using model `{nn_model.__class__.__name__}` with head `{classification_module(nn_model).__class__.__name__}`") diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index d828945..536dffa 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -217,3 +217,13 @@ def test_quantized_class_similarity_uses_represented_values(): actual = _class_similarity(model.weight) expected = _class_similarity(model.weight.dequantize()) torch.testing.assert_close(actual, expected) + + +def test_restoration_preserves_skipped_operation_reasons(): + original = nn.ModuleDict({"conv": nn.Conv2d(3, 4, 1), "linear": nn.Linear(4, 2)}) + prepare_quantized_training(original) + state = original.state_dict() + restored = nn.ModuleDict({"conv": nn.Conv2d(3, 4, 1), "linear": nn.Linear(4, 2)}) + restore_quantized_training(restored, state) + restored.load_state_dict(state) + assert restored._quantized_training_recipe["skipped_modules"] == original._quantized_training_recipe["skipped_modules"] From ea5085f774b733bae38a1be60312fdb45b49eec7 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 23:00:15 +0200 Subject: [PATCH 011/155] bench: add paired INT8 dataset profiles and visible CI results --- .github/workflows/benchmarks.yml | 32 +++++++++-- dev/benchmarks/README.md | 44 ++++++++++++++-- dev/benchmarks/run.py | 88 +++++++++++++++++++++++++------ dev/benchmarks/summarize.py | 19 +++++-- dev/check-benchmarks.sh | 40 +++++++++++++- docs/benchmarks.md | 51 +++++++++++++++++- docs/quantized-training.md | 7 +-- docs/roadmap.md | 5 +- tests/test_benchmark_datasets.py | 34 +++++++++++- tests/test_benchmark_synthetic.py | 15 +++++- 10 files changed, 294 insertions(+), 41 deletions(-) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index db71e94..f49890d 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -13,6 +13,10 @@ on: description: Run GPU profiles on a configured self-hosted GPU runner type: boolean default: false + qt: + description: Also run paired float and INT8 training profiles (requires gpu) + type: boolean + default: false real_data: description: Also run MNIST and Blair using configured runner data paths type: boolean @@ -32,9 +36,9 @@ jobs: timeout-minutes: 15 steps: - name: Validate manual profile selection - if: github.event_name == 'workflow_dispatch' && inputs.real_data && !inputs.gpu + if: github.event_name == 'workflow_dispatch' && (inputs.real_data || inputs.qt) && !inputs.gpu run: | - echo 'The real_data profile requires gpu=true.' >&2 + echo 'The real_data and qt profiles require gpu=true.' >&2 exit 2 - uses: actions/checkout@v6 - uses: astral-sh/setup-uv@v8.1.0 @@ -64,6 +68,7 @@ jobs: env: BENCHMARK_DATA_ROOT: ${{ vars.BENCHMARK_DATA_ROOT }} BLAIR_CLASS_SPEC: ${{ vars.BLAIR_CLASS_SPEC }} + QT_ENABLED: ${{ inputs.qt || vars.BENCHMARK_QT == 'true' }} CUDA_BACKEND: ${{ inputs.cuda_backend || vars.BENCHMARK_CUDA_BACKEND || 'cu130' }} steps: - uses: actions/checkout@v6 @@ -72,11 +77,24 @@ jobs: run: | export UV_PROJECT_ENVIRONMENT="$RUNNER_TEMP/benchmark-env-$GITHUB_RUN_ID-$GITHUB_RUN_ATTEMPT" printf 'UV_PROJECT_ENVIRONMENT=%s\nBENCHMARK_PYTHON=%s/bin/python\n' "$UV_PROJECT_ENVIRONMENT" "$UV_PROJECT_ENVIRONMENT" >> "$GITHUB_ENV" - uv sync --locked --extra "$CUDA_BACKEND" --python 3.13 + qt_extras=() + if [[ "$QT_ENABLED" == "true" ]]; then qt_extras=(--extra quantization); fi + uv sync --locked --extra "$CUDA_BACKEND" "${qt_extras[@]}" --python 3.13 - name: CUDA optimizer and AMP step regressions env: RUN_CUDA_TESTS: "1" run: '"$BENCHMARK_PYTHON" -m pytest tests/test_optimizer_steps.py -k cuda' + - name: CUDA INT8 model and checkpoint regressions + if: inputs.qt || vars.BENCHMARK_QT == 'true' + env: + RUN_CUDA_TESTS: "1" + run: '"$BENCHMARK_PYTHON" -m pytest tests/test_quantized_training.py tests/test_quantized_training_model.py' + - name: Paired synthetic quantized training + if: (inputs.qt || vars.BENCHMARK_QT == 'true') && !cancelled() + run: bash dev/check-benchmarks.sh qt benchmark-qt + - name: Paired real-data quantized training + if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() + run: bash dev/check-benchmarks.sh qt-real benchmark-qt-real - name: Synthetic GPU precision profiles run: bash dev/check-benchmarks.sh gpu benchmark-gpu - name: Real-data progression @@ -86,6 +104,11 @@ jobs: if: always() run: | python3 -m dev.benchmarks.summarize benchmark-gpu >> "$GITHUB_STEP_SUMMARY" + for qt_results in benchmark-qt benchmark-qt-real; do + if [[ -d "$qt_results" ]]; then + python3 -m dev.benchmarks.summarize "$qt_results" >> "$GITHUB_STEP_SUMMARY" + fi + done if [[ -d benchmark-real ]]; then python3 -m dev.benchmarks.summarize benchmark-real >> "$GITHUB_STEP_SUMMARY" fi @@ -99,6 +122,9 @@ jobs: path: | benchmark-gpu/ benchmark-real/ + benchmark-qt/ + benchmark-qt-real/ + !benchmark-qt/**/data/**/*.png !benchmark-gpu/**/data/**/*.png - name: Remove the disposable GPU environment if: always() diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index d9a733d..b05489c 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -127,8 +127,9 @@ and [artifact retention](https://github.com/actions/upload-artifact#retention-pe ## Quantized training and loader performance -Actual quantized training is the next implementation target. PTQ/QAT and AMP do -not establish reduced training memory or faster training. +Actual quantized training has an initial Linear integration. PTQ/QAT and AMP do +not establish reduced training memory or faster training. Broader coverage and +real-workload speedups remain implementation targets. Two developer probes make the remaining work measurable: @@ -197,8 +198,9 @@ or `--momentum 0.9 --weight-decay 0.1` for SGD. AdamW uses an explicit state. Use `--dtype float32 --epsilon 1e-8` to test ordinary float32 optimizer state. This changes the experimental recipe, not the training CLI defaults. CUDA regression tests exercise ordinary `nn.Linear` dispatch with non-square -weights, bias and batched inputs. Weight normalization, masked classifier -weights, convolutional QT, Muon, DDP and `mt_train` resume remain unverified. +weights, bias, batched inputs and masked classifier rows. Model tests also cover +MuonAuxAdamW and controlled `mt_train` resume. Weight normalization, convolutional +QT, DDP and arbitrary stochastic continuation remain unverified. The original FP16 backward scale products could underflow before quantization, suppressing gradients from mean-reduced losses. The corrected kernel forms @@ -278,3 +280,37 @@ similar throughput, not evidence of a meaningful speedup. The concrete gain is bounded read-ahead and reliable failure handling, with explicit synchronous construction available for shared-node environments. CUDA checks also verify pinned CPU transfer batches and exact CPU/CUDA cache contents. + +## Integrated QT dataset profiles + +Install the optional `quantization` extra with the intended CUDA backend explicitly +selected (see the repository README), then use new output directories: + +```bash +CUDA_VISIBLE_DEVICES=0 TORCHINDUCTOR_COMPILE_THREADS=1 \ + bash dev/check-benchmarks.sh qt /tmp/benchmarks-qt +CUDA_VISIBLE_DEVICES=0 TORCHINDUCTOR_COMPILE_THREADS=1 \ + BENCHMARK_DATA_ROOT=/path/to/datasets BLAIR_CLASS_SPEC=/path/to/class_spec.json \ + bash dev/check-benchmarks.sh qt-real /tmp/benchmarks-qt-real +``` + +The first command pairs floating and INT8 synthetic training. The second pairs +MNIST and hierarchical Blair, using a reviewed existing Blair class specification. +Both use FP16 AMP, CPU caching and zero cache/loader workers; Blair uses hidden +size 64 in both paths to exercise an eligible Linear while normalized heads remain +floating point. The runner exposes `--hidden`, `--batch-size`, `--compile` and +`--cache-workers` for explicit additional profiles. Defaults remain unchanged. +`--cache RAM` is retained as an alias for `CPU`. + +For Actions, enable `gpu=true` and `qt=true`, adding `real_data=true` for both +real datasets. Scheduled GPU runs can enable `BENCHMARK_QT=true` together with the +existing GPU and real-data variables described above. The disposable GPU +environment installs the quantization extra only when QT is enabled. CUDA model +regressions precede the paired profiles; summaries and artifacts retain measured +coverage, storage, timing and failures. This workflow wiring has been checked +locally but has not been dispatched on a self-hosted runner from this session. + +[Local measured results](../../docs/benchmarks.md#integrated-int8-training) include +slower QT training on these small workloads. Whole-model compilation defaults off; +first-use kernel compilation is still included in wall time. Compare matching +configurations and compiler cache conditions before making performance claims. diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index f7975bc..a028066 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -11,6 +11,7 @@ from importlib.metadata import version from pathlib import Path +import matplotlib import numpy as np import torch @@ -42,15 +43,24 @@ def run( dataset: str = "synthetic", data_root: str | Path | None = None, class_spec: str | Path | None = None, + quantized_training: bool = False, + compile: bool = False, + hidden: int = 0, + batch_size: int = 32, + cache_workers: int | None = None, ): + matplotlib.use("Agg", force=True) + cache = "CPU" if cache == "RAM" else cache target_device = torch.device(device) precision = getattr(torch, dtype) if target_device.type not in ("cpu", "cuda") or dtype not in ("float32", "float16", "bfloat16"): raise ValueError("Benchmark profiles support CPU/CUDA with float32, float16 or bfloat16.") if target_device.type == "cpu" and (dtype == "float16" or cache == "CUDA"): raise ValueError("Use CUDA for the float16 or CUDA-cache profiles.") - if num_workers < 0 or epochs < 1: - raise ValueError("Workers must be nonnegative and epochs positive.") + if num_workers < 0 or epochs < 1 or hidden < 0 or batch_size < 2: + raise ValueError("Workers/hidden must be nonnegative, epochs positive and batch size at least two.") + if quantized_training and target_device.type != "cuda": + raise ValueError("Quantized training benchmark profiles require CUDA.") if target_device.type == "cuda": if not torch.cuda.is_available(): raise RuntimeError("CUDA profile requested but no accessible CUDA device is available.") @@ -59,6 +69,14 @@ def run( raise RuntimeError("The selected CUDA device does not support bfloat16.") torch.cuda.synchronize(target_device) torch.cuda.reset_peak_memory_stats(target_device) + repository = Path(__file__).resolve().parents[2] + revision = ( + subprocess.run(["git", "rev-parse", "HEAD"], cwd=repository, capture_output=True, text=True, check=False).stdout.strip() or None + ) + code_digest = hashlib.sha256() + for source in sorted((repository / "mini_trainer").rglob("*.py")) + sorted(Path(__file__).parent.glob("*.py")): + code_digest.update(str(source.relative_to(repository)).encode()) + code_digest.update(source.read_bytes()) output = Path(output).absolute() if output.exists(): raise FileExistsError(output) @@ -108,13 +126,21 @@ def run( seed=seed, builder=HierarchicalBenchmarkBuilder if hierarchical else NoAugmentationBuilder, ema=False, + quantized_training=quantized_training, + compile=compile, model_builder_kwargs={ "model_type": model_type, - "hidden": False, + "hidden": hidden if hidden else False, "normalized": hierarchical, "cls": HierarchicalClassifier if hierarchical else Classifier, }, - dataloader_builder_kwargs={"batch_size": 32, "num_workers": num_workers, "data_index": str(data_index), "cache": cache}, + dataloader_builder_kwargs={ + "batch_size": batch_size, + "num_workers": num_workers, + "data_index": str(data_index), + "cache": cache, + "cache_workers": cache_workers, + }, optimizer_builder_kwargs={"optimizer_cls": MuonAuxAdamW, "lr": 0.1 if dataset == "synthetic" else 0.01, "weight_decay": 0.0}, criterion_builder_kwargs={"label_smoothing": 0.0}, regularizer_builder_kwargs={"strength": 0.0}, @@ -128,11 +154,14 @@ def run( weights = output / "training/weights/last.pt" model, preprocess = Classifier.build(weights=str(weights), device=target_device, dtype=torch.float32) model.eval() + quantization_recipe = getattr(model, "_quantized_training_recipe", None) + if quantized_training and not quantization_recipe: + raise RuntimeError("Quantized training recipe was not restored from the checkpoint.") test_records = [record for record in records if record["split"] == "test"] _, loader = get_inference_dataloader( images=[str(root / record["path"]) for record in test_records], resize_size=size, - batch_size=32, + batch_size=batch_size, num_workers=0, device=target_device, dtype=torch.float32, @@ -154,14 +183,6 @@ def run( arrays.update({f"scores_{level}": scores for level, scores in enumerate(scores_by_level)}) arrays.update({f"labels_{level}": labels for level, labels in enumerate(labels_by_level)}) np.savez(output / "predictions.npz", **arrays) - repository = Path(__file__).resolve().parents[2] - revision = ( - subprocess.run(["git", "rev-parse", "HEAD"], cwd=repository, capture_output=True, text=True, check=False).stdout.strip() or None - ) - code_digest = hashlib.sha256() - for source in sorted((repository / "mini_trainer").rglob("*.py")) + sorted(Path(__file__).parent.glob("*.py")): - code_digest.update(str(source.relative_to(repository)).encode()) - code_digest.update(source.read_bytes()) result = { "schema_version": 1, "status": "passed" if dataset == "synthetic" and accuracy == 1.0 else "failed" if dataset == "synthetic" else "completed", @@ -179,11 +200,22 @@ def run( "cuda_cache": cache == "CUDA", "ema": False, "distributed": False, - "quantization": False, + "quantization": bool(quantization_recipe), "augmentation": False, "onnx": False, }, "dataset": dataset, + "quantization_recipe": quantization_recipe, + "compile": compile, + "hidden": hidden, + "batch_size": batch_size, + "cache_workers": cache_workers, + "parameter_bytes": sum( + parameter.int_data.numel() + parameter.scale.numel() * parameter.scale.element_size() + if getattr(parameter, "_is_quantized_training", False) + else parameter.numel() * parameter.element_size() + for parameter in model.parameters() + ), "level_accuracies": accuracies, "split_counts": {split: sum(record["split"] == split for record in records) for split in ("train", "val", "test")}, "seed": seed, @@ -192,7 +224,7 @@ def run( "chance_accuracy": 1 / scores_by_level[0].shape[1], "test_accuracy": accuracy, "training_wall_seconds": elapsed, - "training_wall_scope": "setup, training, validation, logging and checkpoints", + "training_wall_scope": "setup, training, validation, logging and checkpoints; includes first-use compilation/autotuning", "num_workers_requested": num_workers, "num_workers": 0 if cache == "CUDA" else num_workers, "cache": cache, @@ -207,7 +239,14 @@ def run( "platform": platform.platform(), "cuda_version": torch.version.cuda, "cudnn_version": torch.backends.cudnn.version() if target_device.type == "cuda" else None, - "versions": {name: version(name) for name in ("torch", "torchvision", "numpy", "mini_trainer")}, + "versions": { + name: version(name) + for name in ( + ("torch", "torchvision", "numpy", "mini_trainer", "torchao") + if quantization_recipe + else ("torch", "torchvision", "numpy", "mini_trainer") + ) + }, "dataset_manifest_sha256": hashlib.sha256(manifest_path.read_bytes()).hexdigest(), "checkpoint_sha256": hashlib.sha256(weights.read_bytes()).hexdigest(), "score_semantics": "model_eval_forward", @@ -229,8 +268,13 @@ def main(): parser.add_argument("--threads", type=int, default=1) parser.add_argument("--device", default="cpu") parser.add_argument("--dtype", choices=["float32", "float16", "bfloat16"], default="float32") - parser.add_argument("--cache", choices=["NONE", "RAM", "CUDA"], default="NONE") + parser.add_argument("--cache", choices=["NONE", "CPU", "RAM", "CUDA"], default="NONE") parser.add_argument("--num-workers", type=int, default=0) + parser.add_argument("--cache-workers", type=int) + parser.add_argument("--quantized-training", action="store_true") + parser.add_argument("--compile", action="store_true") + parser.add_argument("--hidden", type=int, default=0) + parser.add_argument("--batch-size", type=int, default=32) parser.add_argument( "--allow-nondeterministic", action="store_true", @@ -256,6 +300,11 @@ def main(): dataset=args.dataset, data_root=args.data_root, class_spec=args.class_spec, + quantized_training=args.quantized_training, + compile=args.compile, + hidden=args.hidden, + batch_size=args.batch_size, + cache_workers=args.cache_workers, ) except Exception as error: args.output.mkdir(parents=True, exist_ok=True) @@ -268,6 +317,11 @@ def main(): "cache": args.cache, "seed": args.seed, "epochs": args.epochs, + "quantized_training": args.quantized_training, + "compile": args.compile, + "hidden": args.hidden, + "batch_size": args.batch_size, + "cache_workers": args.cache_workers, "error": {"type": type(error).__name__, "message": str(error)}, } (args.output / "report.json").write_text(json.dumps(failure, indent=2) + "\n") diff --git a/dev/benchmarks/summarize.py b/dev/benchmarks/summarize.py index ecb3a7a..8a4d3ea 100644 --- a/dev/benchmarks/summarize.py +++ b/dev/benchmarks/summarize.py @@ -9,8 +9,8 @@ def summarize(directory: Path) -> str: lines = [ "# Dataset benchmark results", "", - "| Run | Status | Device / precision | Accuracy by level | Training wall time |", - "| --- | --- | --- | --- | --- |", + "| Run | Status | Device / precision | QT coverage | Accuracy by level | Parameter bytes | Peak CUDA MiB | Training wall time |", + "| --- | --- | --- | --- | --- | --- | --- | --- |", ] reports = sorted(directory.rglob("report.json")) for path in reports: @@ -20,9 +20,18 @@ def summarize(directory: Path) -> str: duration = f"{seconds:.2f}s" if seconds is not None else "—" name = path.parent.relative_to(directory).as_posix() device = f"{report.get('device', '?')} / {report.get('dtype', '?')}" - lines.append(f"| {name} | {report['status']} | {device} | {accuracy} | {duration} |") + recipe = report.get("quantization_recipe") + quantization = ( + f"INT8 ({len(recipe['quantized_modules'])} Linear)" if recipe else "requested" if report.get("quantized_training") else "off" + ) + parameter_bytes = report.get("parameter_bytes", "—") + peak = report.get("peak_cuda_allocated_bytes") + peak_memory = f"{peak / 2**20:.2f}" if peak is not None else "—" + lines.append( + f"| {name} | {report['status']} | {device} | {quantization} | {accuracy} | {parameter_bytes} | {peak_memory} | {duration} |" + ) if not reports: - lines.append("| No reports produced | incomplete | — | — | — |") + lines.append("| No reports produced | incomplete | — | — | — | — | — | — |") lines.extend( [ "", @@ -32,6 +41,8 @@ def summarize(directory: Path) -> str: "Wall times include setup, training, validation, logging and checkpoints. Compare timings", "only with matching hardware, dataset/configuration and timing scope. See JSON reports", "for provenance, errors and explicit coverage flags. CPU results do not validate GPU behavior.", + "QT coverage counts quantized Linear modules; other operations may remain floating point.", + "Parameter bytes describe stored parameters, while peak CUDA memory covers the full run.", ] ) return "\n".join(lines) + "\n" diff --git a/dev/check-benchmarks.sh b/dev/check-benchmarks.sh index ef23e37..e7adc31 100644 --- a/dev/check-benchmarks.sh +++ b/dev/check-benchmarks.sh @@ -9,15 +9,32 @@ if [[ -e "$results" ]]; then echo 'Results directory must be new.' >&2 exit 2 fi -case "$mode" in cpu|gpu|real) ;; *) echo 'Mode must be cpu, gpu or real.' >&2; exit 2 ;; esac +case "$mode" in cpu|gpu|real|qt|qt-real) ;; *) echo 'Mode must be cpu, gpu, real, qt or qt-real.' >&2; exit 2 ;; esac mkdir -p -- "$results" status=0 run_profile() { local profile="$1" shift - if ! OMP_NUM_THREADS=1 "$benchmark_python" -m dev.benchmarks.run --output "$results/$profile" "$@" > "$results/$profile.log" 2>&1; then + if OMP_NUM_THREADS=1 MPLBACKEND=Agg "$benchmark_python" -m dev.benchmarks.run --output "$results/$profile" "$@" > "$results/$profile.log" 2>&1; then + return + else + local exit_code="$?" echo "Benchmark failed: $profile; see $results/$profile.log" >&2 status=1 + if [[ ! -f "$results/$profile/report.json" ]]; then + "$benchmark_python" - "$results/$profile/report.json" "$exit_code" "$@" <<'PY_REPORT' +import json +import sys +from pathlib import Path +path = Path(sys.argv[1]) +path.parent.mkdir(parents=True, exist_ok=True) +path.write_text(json.dumps({ + "schema_version": 1, "status": "failed", "arguments": sys.argv[3:], + "quantized_training": "--quantized-training" in sys.argv[3:], + "error": {"type": "ProcessFailure", "exit_code": int(sys.argv[2]), "message": "Process exited without a report; see profile log."}, +}, indent=2) + "\n") +PY_REPORT + fi fi } if [[ "$mode" == cpu ]]; then @@ -26,6 +43,25 @@ elif [[ "$mode" == gpu ]]; then for precision in float32 float16 bfloat16; do run_profile "synthetic-cuda-$precision" --device cuda:0 --dtype "$precision" --cache CUDA done +elif [[ "$mode" == qt ]]; then + run_profile synthetic-float --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 + run_profile synthetic-int8 --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --quantized-training +elif [[ "$mode" == qt-real ]]; then + : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/ and blair/}" + : "${BLAIR_CLASS_SPEC:?Set BLAIR_CLASS_SPEC to a reviewed Blair class specification}" + for dataset in mnist blair; do + extra=() + if [[ "$dataset" == blair ]]; then + extra=(--class-spec "$BLAIR_CLASS_SPEC" --hidden 64) + fi + for precision in float int8; do + quantization=() + if [[ "$precision" == int8 ]]; then quantization=(--quantized-training); fi + run_profile "$dataset-$precision" --dataset "$dataset" --data-root "$BENCHMARK_DATA_ROOT/$dataset" \ + --epochs 5 --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --allow-nondeterministic \ + "${extra[@]}" "${quantization[@]}" + done + done else : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/ and blair/}" : "${BLAIR_CLASS_SPEC:?Set BLAIR_CLASS_SPEC to a reviewed Blair class specification}" diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 1a88ee2..6d99ccb 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -58,8 +58,9 @@ inference quantization or quantization-aware training. ## Coverage still needed -The reports currently mark EMA, distributed training, augmentation, quantization -and ONNX integration as unexercised by this benchmark suite. Existing focused tests +The baseline reports mark EMA, distributed training, augmentation, quantization +and ONNX integration as unexercised. The explicit QT profiles below establish +limited quantization coverage; the other capabilities remain unexercised by this suite. Existing focused tests provide other evidence, but do not make those boxes true for these dataset runs. Add explicit profiles and comparison criteria before claiming coverage or improvement. The known EMA continuation failure remains recorded in the roadmap. @@ -72,3 +73,49 @@ it does not yet measure their relative model quality. See the [step contract](.. See the [benchmark guide](../dev/benchmarks/README.md) for local commands, GPU runner configuration, real-data inputs and reproduction details. + +## Integrated INT8 training + +The shared runner now offers paired `qt` (synthetic) and `qt-real` (MNIST/Blair) +profiles. Each pair uses the same architecture, dataset manifest, seed, optimizer, +AMP setting and CPU cache with synchronous construction. QT coverage is checked +after checkpoint reload and reported by operation, alongside parameter storage, +peak CUDA allocation and training-call wall time. Runs are headless; the wrapper +also records process failures that cannot produce a Python exception report. + +Local observations on the RTX 3080 Ti Laptop GPU, Python 3.13.7, PyTorch +2.12.0+cu130 and TorchAO 0.17.0 use seed 42, batch size 32, one CPU thread, +zero loader/cache workers and FP16 AMP with float32 optimizer parameters. +Synthetic uses 12 epochs; MNIST and Blair use 5. No augmentation or EMA is used. + +| Dataset / path | Held-out accuracy | Parameter bytes | Peak CUDA MiB | Training wall seconds | +| --- | --- | ---: | ---: | ---: | +| Synthetic float | 100% | 88 | 64.04 | 4.22 | +| Synthetic INT8, repeat | 100% | 68 | 32.03 | 12.95 | +| MNIST float | 97.12% | 44,968 | 64.18 | 6.06 | +| MNIST INT8 | 97.00% | 29,648 | 32.17 | 43.08 | +| Blair float | 66.41% species / 80.28% parent | 158,792 | 64.50 | 10.64 | +| Blair INT8 | 64.86% species / 79.59% parent | 60,744 | 64.40 | 26.51 | + +Synthetic INT8 repeated with bitwise-identical held-out scores. Its first run +required 80.43 seconds, showing how compiler/autotuning cache state affects these +short runs even without whole-model compilation. Wall time includes setup, +training, validation, logging, checkpoints and first-use compilation; it excludes +final held-out inference. These are single-seed smoke comparisons, not isolated +causal estimates or steady-state throughput measurements. Real-data runs permit +nondeterministic CUDA pooling. QT stochastic rounding can also change the RNG +sequence used by dropout. + +MNIST quantizes only its final Linear. Blair uses a 64-unit hidden layer in both +paths and quantizes that layer; convolutions and normalized hierarchical heads +remain floating point. Parameter bytes exclude buffers, gradients, optimizer +state and activations. Whole-run peak allocation includes workspaces: the tiny +synthetic model's roughly 32 MiB difference cannot be explained by its 20-byte +parameter reduction. Blair's parameter reduction barely changes overall peak +allocation. None of these dataset runs demonstrates a training speedup. + +The initial Blair QT attempt aborted in Tkinter cleanup before reporting; the +headless fix allowed the successful rerun above. Reports now preserve skipped +operation reasons across reload. These local artifacts remain outside the checkout; +the shared workflow retains future reports, predictions and logs in Actions. +See [the reproduction commands](../dev/benchmarks/README.md#integrated-qt-dataset-profiles). diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 08b4d6d..51de1f3 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -84,9 +84,10 @@ graphs. Numerical checks include small gradients, compiled/eager agreement, weight storage, optimizer updates, regularization and model-state restoration. Kernel-probe results do not establish a real-model speedup or convergence. The -integrated path still needs paired synthetic-oracle, MNIST and hierarchical Blair -runs, complete optimizer/resume coverage, end-to-end memory and throughput -measurements, and broader quantized operation coverage. These are requirements +integrated path now has paired synthetic-oracle, MNIST and hierarchical Blair +smoke runs, recorded in [the dataset benchmark results](benchmarks.md#integrated-int8-training). +Complete optimizer/resume coverage, demonstrated real-workload speedups and broader +quantized operation coverage remain outstanding. These are requirements for the overall QT goal, not conclusions implied by this initial integration. The integrated backend was rerun on the four-layer, 4096-wide, batch-2048 compiled diff --git a/docs/roadmap.md b/docs/roadmap.md index 3cf3137..87c1a90 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -95,8 +95,9 @@ capability and does not complete this target. The initial CUDA integer forward/b kernel probe and cached-loader benchmark are documented in [the benchmark guide](../dev/benchmarks/README.md). An initial [CUDA INT8 Linear integration](quantized-training.md) connects model preparation and checkpoint loading to the training entry point. Broader operator -and optimizer coverage, real-data convergence and end-to-end measurements remain -required to complete this target. +and optimizer coverage, stronger convergence evidence and real-workload speedups +remain required. Paired synthetic, MNIST and hierarchical Blair smoke runs now +record quality, storage and timing; [these small workloads are slower under QT](benchmarks.md#integrated-int8-training). Loader hardening, float16/bfloat16 AMP and benchmark infrastructure do not complete that target. The implementation and comparison plan is in [training feature validation](training-feature-validation.md). diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index da26933..21c576e 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -63,8 +63,38 @@ def test_blair_requires_explicit_covering_taxonomy(tmp_path): def test_summary_preserves_failures_and_unmeasured_fields(tmp_path): path = tmp_path / "failed" path.mkdir() - (path / "report.json").write_text(json.dumps({"status": "failed", "device": "cuda:0", "dtype": "float16"})) + (path / "report.json").write_text(json.dumps({"status": "failed", "device": "cuda:0", "dtype": "float16", "quantized_training": True})) summary = summarize(tmp_path) - assert "| failed | failed | cuda:0 / float16 | — | — |" in summary + assert "| failed | failed | cuda:0 / float16 | requested | — | — | — | — |" in summary assert "CPU results do not validate GPU" in summary assert "No reports produced" in summarize(Path(tmp_path / "missing")) + + +def test_shared_harness_records_process_failures(tmp_path): + import os + import shlex + import subprocess + import sys + + runner = tmp_path / "python-wrapper" + runner.write_text( + '#!/usr/bin/env bash\nif [[ "$1" == "-m" && "$2" == "dev.benchmarks.run" ]]; then exit 134; fi\n' + + f'exec {shlex.quote(sys.executable)} "$@"\n' + ) + runner.chmod(0o755) + output = tmp_path / "reports" + result = subprocess.run( + ["bash", "dev/check-benchmarks.sh", "qt", str(output)], + env={**os.environ, "BENCHMARK_PYTHON": str(runner)}, + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 1 + for profile in ("synthetic-float", "synthetic-int8"): + report = json.loads((output / profile / "report.json").read_text()) + assert report["status"] == "failed" + assert report["error"]["exit_code"] == 134 + assert report["quantized_training"] == profile.endswith("int8") + assert "test_accuracy" not in report + assert "requested" in (output / "summary.md").read_text() diff --git a/tests/test_benchmark_synthetic.py b/tests/test_benchmark_synthetic.py index 8b14495..950bd2b 100644 --- a/tests/test_benchmark_synthetic.py +++ b/tests/test_benchmark_synthetic.py @@ -27,10 +27,11 @@ def test_synthetic_training_matches_oracle_and_repeats(tmp_path): threads = torch.get_num_threads() try: torch.set_num_threads(1) - first = run(tmp_path / "first") - second = run(tmp_path / "second") + first = run(tmp_path / "first", cache="CPU", cache_workers=0) + second = run(tmp_path / "second", cache="RAM", cache_workers=0) finally: torch.set_num_threads(threads) + assert first["cache"] == second["cache"] == "CPU" assert first["test_accuracy"] == second["test_accuracy"] == 1.0 assert first["dataset_manifest_sha256"] == second["dataset_manifest_sha256"] with np.load(tmp_path / "first/predictions.npz") as a, np.load(tmp_path / "second/predictions.npz") as b: @@ -73,3 +74,13 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): assert report["device"] == "cuda:0" assert report["error"]["type"] == "RuntimeError" assert "test_accuracy" not in report + + +def test_qt_profile_requires_cuda_before_creating_output(tmp_path): + import pytest + + from dev.benchmarks.run import run + + with pytest.raises(ValueError, match="require CUDA"): + run(tmp_path / "qt", quantized_training=True) + assert not (tmp_path / "qt").exists() From 0ddc8d72464d1e57c56bdf7a2b7f0442bc0b7cb8 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 23:07:22 +0200 Subject: [PATCH 012/155] feat: support INT8 training of row-normalized Linear weights --- dev/benchmarks/README.md | 6 +- docs/benchmarks.md | 16 +++ docs/quantized-training.md | 24 ++++- mini_trainer/modeling/_quantized_training.py | 34 +++++++ mini_trainer/modeling/quantized_training.py | 39 ++++--- tests/test_quantized_training_model.py | 101 ++++++++++++++++++- 6 files changed, 195 insertions(+), 25 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index b05489c..421ac91 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -199,7 +199,7 @@ state. Use `--dtype float32 --epsilon 1e-8` to test ordinary float32 optimizer state. This changes the experimental recipe, not the training CLI defaults. CUDA regression tests exercise ordinary `nn.Linear` dispatch with non-square weights, bias, batched inputs and masked classifier rows. Model tests also cover -MuonAuxAdamW and controlled `mt_train` resume. Weight normalization, convolutional +MuonAuxAdamW and controlled `mt_train` resume. Row-wise weight normalization is now covered separately. Convolutional QT, DDP and arbitrary stochastic continuation remain unverified. The original FP16 backward scale products could underflow before quantization, @@ -297,8 +297,8 @@ CUDA_VISIBLE_DEVICES=0 TORCHINDUCTOR_COMPILE_THREADS=1 \ The first command pairs floating and INT8 synthetic training. The second pairs MNIST and hierarchical Blair, using a reviewed existing Blair class specification. Both use FP16 AMP, CPU caching and zero cache/loader workers; Blair uses hidden -size 64 in both paths to exercise an eligible Linear while normalized heads remain -floating point. The runner exposes `--hidden`, `--batch-size`, `--compile` and +size 64 in both paths to exercise both a hidden Linear and its normalized head. Earlier recorded +profiles quantized only the hidden layer; inspect each report for actual coverage. The runner exposes `--hidden`, `--batch-size`, `--compile` and `--cache-workers` for explicit additional profiles. Defaults remain unchanged. `--cache RAM` is retained as an alias for `CPU`. diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 6d99ccb..35ffd6c 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -119,3 +119,19 @@ headless fix allowed the successful rerun above. Reports now preserve skipped operation reasons across reload. These local artifacts remain outside the checkout; the shared workflow retains future reports, predictions and logs in Actions. See [the reproduction commands](../dev/benchmarks/README.md#integrated-qt-dataset-profiles). + + +With row-wise weight normalization supported, a further matched Blair pair uses +no hidden layer and quantizes the normalized classifier direction directly: + +| Blair path, hidden size 0 | Held-out accuracy | Parameter bytes | Peak CUDA MiB | Training wall seconds | +| --- | --- | ---: | ---: | ---: | +| Float | 64.25% species / 80.45% parent | 75,848 | 64.34 | 11.65 | +| INT8 normalized direction | 62.62% species / 76.14% parent | 37,548 | 32.25 | 25.32 | + +The environment, seed, five-epoch budget, batch size and CPU cache settings match +the preceding comparisons. Dataset manifest hashes agree between the two runs. +Convolutions still remain floating point. This establishes real hierarchical +training and restored-checkpoint inference with normalized integer weights, with +roughly half the parameter storage. Accuracy is lower in this single run and QT +is slower; neither convergence parity nor a throughput improvement is established. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 51de1f3..c476eb3 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -3,7 +3,7 @@ The opt-in training path stores eligible Linear weights and saved linear inputs in INT8 and uses integer matrix products for forward, input gradients and weight gradients. It retains no floating-point master copy of those weights. Gradients, -optimizer state, biases, normalization, convolutions and auxiliary regularization +optimizer state, biases, activation normalization, convolutions and auxiliary regularization remain floating point. This is separate from fake-quantized QAT and x86 PTQ. Install the optional `quantization` extra while explicitly retaining the intended @@ -35,8 +35,13 @@ The returned recipe lists quantized modules, skipped operations, remaining floating-point parameters and physical versus reference weight storage. Automatic selection covers ordinary `nn.Linear` modules, including their functional use by Classifier heads. It preserves shared weights when every owner is selected. -Parametrized weights (including normalized heads) and weights shared with an -unselected operation stay floating point and are reported. Explicit unsupported +Row-wise PyTorch weight normalization (`dim=0`) quantizes the direction parameter +while retaining its scalar magnitude per output row in floating point. Effective +normalized weights reuse the integer codes with new row scales; normalization +backward applies its Jacobian to the approximate Linear gradient. No floating +weight matrix is retained for this operation. Other parametrizations, normalization +dimensions and weights shared with an unselected operation stay floating point +and are reported. Explicit unsupported selections and models with no eligible weights fail before changing weights. Quantizing the hidden linear layer of a convolutional classifier does not make its convolutions integer operations. @@ -71,8 +76,8 @@ model.load_state_dict(state) Use the same architecture and intended dtype. This restores model state; create and restore optimizer/scheduler/scaler state in their normal order separately. The same model supports CUDA inference with `eval()` and `inference_mode()`. -ONNX export, checkpoint averaging, DDP/FSDP, quantized normalization and integer -convolution training are not established for this path. Distributed training and +ONNX export, checkpoint averaging, DDP/FSDP, quantized activation normalization +and integer convolution training are not established for this path. Distributed training and EMA are rejected by the training entry point. ## Evidence and remaining work @@ -96,3 +101,12 @@ SGD probe: 15.61 ms/step and 319,063,552 peak allocated bytes for INT8, versus 1.93x faster and 17% lower peak memory for that workload, with nonzero gradients checked before timing. They remain kernel-probe evidence, not a claim about MNIST, Blair or typical convolutional models. + + +Row-wise normalization is checked against PyTorch's represented-value forward +and backward results, including signed scales and zero magnitudes. Tests inspect +saved tensors to exclude a retained floating direction matrix, and exercise +checkpoint restoration, eager/compiled CUDA training and masked inference. +Initial zero direction rows are rejected before preparation mutates any weights; +normalization is undefined for these rows. The mathematical contract follows +[PyTorch weight normalization](https://docs.pytorch.org/docs/2.12/generated/torch.nn.utils.parametrizations.weight_norm.html). diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index 3c990a2..df9ba6c 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -176,3 +176,37 @@ def forward(ctx, weight): @staticmethod def backward(ctx, gradient): return gradient + + +@TrainingWeight.implements_torch_function(torch._weight_norm) +def weight_norm(func, types, args, kwargs): + direction = args[0] if args else kwargs["v"] + magnitude = args[1] if len(args) > 1 else kwargs["g"] + dimension = args[2] if len(args) > 2 else kwargs.get("dim", 0) + if dimension != 0: + raise ValueError("INT8 weight normalization supports Linear output rows (dim=0) only.") + return _WeightNorm.apply(direction, magnitude) + + +class _WeightNorm(torch.autograd.Function): + """Normalize represented directions without retaining floating weight matrices. + + w = g * v / ||v||. The codes are unchanged; only their row scales change. + Backward applies the ordinary normalization Jacobian to the approximate dW + supplied by IntegerLinear. Float intermediates are transient, never masters. + """ + + @staticmethod + def forward(ctx, direction, magnitude): + codes, scales = direction.int_data, direction.scale + norm = torch.linalg.vector_norm(codes.float(), dim=1) + ctx.save_for_backward(codes, scales, magnitude, norm) + return TrainingWeight(codes, (scales.sign() * magnitude.flatten().float() / norm).to(scales.dtype)) + + @staticmethod + def backward(ctx, gradient): + codes, scales, magnitude, norm = ctx.saved_tensors + unit = codes.float() * (scales.sign() / norm).unsqueeze(1) + projection = (gradient.float() * unit).sum(dim=1, keepdim=True) + direction_gradient = (gradient.float() - projection * unit) * (magnitude.float() / (scales.float().abs() * norm).unsqueeze(1)) + return direction_gradient.to(scales.dtype), projection.to(magnitude.dtype) diff --git a/mini_trainer/modeling/quantized_training.py b/mini_trainer/modeling/quantized_training.py index 390d071..dc66dd9 100644 --- a/mini_trainer/modeling/quantized_training.py +++ b/mini_trainer/modeling/quantized_training.py @@ -1,6 +1,6 @@ -"""Opt-in CUDA INT8 weight/activation training for ordinary linear modules. +"""Opt-in CUDA INT8 weight/activation training for linear modules. -Prepare before constructing an optimizer. Biases, normalization, convolutions, +Prepare before constructing an optimizer. Biases, activation normalization, convolutions, and unselected weights remain floating point and are reported explicitly. """ @@ -30,8 +30,9 @@ def prepare_quantized_training(model: nn.Module, *, module_names=None) -> dict: No floating master weights are retained. Recreate optimizers after this call. CPU preparation supports checkpoint inspection; execution requires CUDA. Explicitly selected unsupported modules raise instead of silently skipping. - Parametrized weights and weights shared with unselected operations remain - floating point until their quantized training contracts are implemented. + Row-wise weight normalization retains a floating magnitude and quantizes its + direction. Other parametrizations and weights shared with unselected + operations remain floating point. """ backend = _backend() modules = dict(model.named_modules(remove_duplicate=False)) @@ -43,27 +44,36 @@ def prepare_quantized_training(model: nn.Module, *, module_names=None) -> dict: if requested is not None and name not in requested: continue reason = None + owner, attribute = module, "weight" if not isinstance(module, nn.Linear): if requested is not None or isinstance(module, (nn.Conv1d, nn.Conv2d, nn.Conv3d)): reason = "integer training is currently implemented for Linear" else: continue elif nn.utils.parametrize.is_parametrized(module, "weight"): - reason = "parametrized weights require a separate quantized gradient contract" - elif not isinstance(module.weight, nn.Parameter) or module.weight.numel() == 0: + parametrizations = module.parametrizations.weight + if ( + len(parametrizations) == 1 + and type(parametrizations[0]) is nn.utils.parametrizations._WeightNorm + and parametrizations[0].dim == 0 + ): + owner, attribute = parametrizations, "original1" + else: + reason = "only row-wise weight normalization is supported among parametrized weights" + if reason is None and (not isinstance(getattr(owner, attribute), nn.Parameter) or getattr(owner, attribute).numel() == 0): reason = "requires a nonempty weight Parameter" - elif module.weight.dtype not in (torch.float16, torch.bfloat16, torch.float32): + elif reason is None and getattr(owner, attribute).dtype not in (torch.float16, torch.bfloat16, torch.float32): reason = "requires float16, bfloat16 or float32 compute metadata" if reason: skipped[name] = reason else: - candidates[name] = (module, module.weight) - selected_owners = {(id(module), "weight") for module, _ in candidates.values()} + candidates[name] = (owner, attribute, getattr(owner, attribute)) + selected_owners = {(id(owner), attribute) for owner, attribute, _ in candidates.values()} owners = {} for module in modules.values(): for name, parameter in module.named_parameters(recurse=False): owners.setdefault(id(parameter), set()).add((id(module), name)) - for name, (_, parameter) in list(candidates.items()): + for name, (_, _, parameter) in list(candidates.items()): if owners[id(parameter)] - selected_owners: skipped[name] = "weight is shared with an unselected operation" del candidates[name] @@ -72,19 +82,22 @@ def prepare_quantized_training(model: nn.Module, *, module_names=None) -> dict: if not candidates: raise ValueError(f"No eligible Linear weights for quantized training. Unsupported modules: {skipped}") # Validate before mutating the caller's model. - for _, parameter in candidates.values(): + for owner, attribute, parameter in candidates.values(): values = parameter.dequantize() if isinstance(parameter, backend.TrainingWeight) else parameter if not torch.isfinite(values).all(): raise ValueError("Quantized training requires finite initial weights.") + if attribute == "original1": + if not torch.isfinite(owner.original0).all() or (values.float().norm(dim=1) == 0).any(): + raise ValueError("Quantized weight normalization requires finite magnitudes and nonzero directions.") replacements = {} - for module, parameter in candidates.values(): + for owner, attribute, parameter in candidates.values(): if id(parameter) not in replacements: replacements[id(parameter)] = ( parameter if isinstance(parameter, backend.TrainingWeight) else nn.Parameter(backend.TrainingWeight.from_float(parameter), requires_grad=parameter.requires_grad) ) - module.weight = replacements[id(parameter)] + setattr(owner, attribute, replacements[id(parameter)]) for module in modules.values(): invalidate = getattr(module, "_on_quantized_training_prepared", None) if invalidate is not None: diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index 536dffa..d419089 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -17,7 +17,7 @@ def test_selection_preserves_ties_and_reports_float_operations(): model = nn.ModuleDict({"a": nn.Linear(8, 8), "b": nn.Linear(8, 8), "conv": nn.Conv2d(3, 3, 1), "norm": nn.Linear(8, 2)}) model["b"].weight = model["a"].weight - nn.utils.parametrizations.weight_norm(model["norm"]) + nn.utils.parametrizations.weight_norm(model["norm"], dim=1) report = prepare_quantized_training(model) assert report["quantized_modules"] == ["a", "b"] assert set(report["skipped_modules"]) == {"conv", "norm"} @@ -134,7 +134,8 @@ def test_mixed_muon_adamw_updates_and_counter(): assert restored._step_count == before_count + 1 -def test_training_entrypoint_checkpoint_and_inference(tmp_path): +@pytest.mark.parametrize("normalized", [False, True]) +def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized): from mini_trainer.modeling import Classifier from mini_trainer.modeling._quantized_training import TrainingWeight from mini_trainer.train import main @@ -154,7 +155,7 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path): "quantized_training": True, "seed": 42, "builder": DeterministicBuilder, - "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": False}, + "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": normalized}, "dataloader_builder_kwargs": {"batch_size": 4}, "lr_schedule_builder_kwargs": {"warmup_epochs": 0}, "ema": False, @@ -179,7 +180,8 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path): main(**args) resumed = load_training_weights(tmp_path / "resumed/weights/checkpoint_last.pth") assert resumed["epoch"] == 1 - assert isinstance(resumed["model"]["fc.linear.weight"], TrainingWeight) + key = "fc.linear.parametrizations.weight.original1" if normalized else "fc.linear.weight" + assert isinstance(resumed["model"][key], TrainingWeight) def test_quantized_regularizer_retains_weight_gradients(): @@ -227,3 +229,94 @@ def test_restoration_preserves_skipped_operation_reasons(): restore_quantized_training(restored, state) restored.load_state_dict(state) assert restored._quantized_training_recipe["skipped_modules"] == original._quantized_training_recipe["skipped_modules"] + + +@pytest.mark.parametrize("negative_scale", [False, True]) +def test_normalized_direction_matches_float_jacobian_without_saved_float_weights(negative_scale): + from mini_trainer.modeling._quantized_training import TrainingWeight + + torch.manual_seed(19) + direction = nn.Parameter(TrainingWeight.from_float(torch.randn(5, 17))) + if negative_scale: + with torch.no_grad(): + direction.scale.neg_() + magnitude = nn.Parameter(torch.randn(5, 1)) + with torch.no_grad(): + magnitude[0].zero_() + reference_direction = direction.detach().dequantize().requires_grad_() + reference_magnitude = magnitude.detach().clone().requires_grad_() + saved = [] + with torch.autograd.graph.saved_tensors_hooks(lambda tensor: saved.append(tensor) or tensor, lambda tensor: tensor): + actual = torch._weight_norm(direction, magnitude, 0) + assert isinstance(actual, TrainingWeight) + assert actual.int_data.data_ptr() == direction.int_data.data_ptr() + assert not any(tensor.shape == direction.shape and tensor.is_floating_point() for tensor in saved) + expected = torch._weight_norm(reference_direction, reference_magnitude, 0) + torch.testing.assert_close(actual.dequantize(), expected) + gradient = torch.randn_like(expected) + actual.backward(gradient) + expected.backward(gradient) + torch.testing.assert_close(direction.grad, reference_direction.grad) + torch.testing.assert_close(magnitude.grad, reference_magnitude.grad) + + +def test_normalized_checkpoint_restores_direction_and_magnitude(tmp_path): + from mini_trainer.modeling._quantized_training import TrainingWeight + + def make(): + return nn.utils.parametrizations.weight_norm(nn.Linear(17, 5)) + + original = make() + report = prepare_quantized_training(original) + assert report["quantized_modules"] == [""] + assert report["floating_parameter_names"] == ["bias", "parametrizations.weight.original0"] + path = tmp_path / "normalized.pt" + torch.save(original.state_dict(), path) + state = load_training_weights(path) + restored = make() + restore_quantized_training(restored, state) + restored.load_state_dict(state) + assert isinstance(restored.parametrizations.weight.original1, TrainingWeight) + torch.testing.assert_close(restored.weight.dequantize(), original.weight.dequantize(), rtol=0, atol=0) + + +def test_invalid_normalized_direction_does_not_partially_prepare(): + model = nn.Sequential(nn.Linear(8, 8), nn.utils.parametrizations.weight_norm(nn.Linear(8, 4))) + original = model[0].weight + with torch.no_grad(): + model[1].parametrizations.weight.original1[0].zero_() + with pytest.raises(ValueError, match="nonzero directions"): + prepare_quantized_training(model) + assert model[0].weight is original + + +@pytest.mark.parametrize("compiled", [False, True]) +def test_normalized_classifier_integer_training_and_masked_inference(compiled): + from mini_trainer.modeling import Classifier + from mini_trainer.modeling._quantized_training import TrainingWeight + + torch.manual_seed(23) + model = Classifier(64, 4, hidden=False, normalized=True).to(cuda()) + report = prepare_quantized_training(model) + assert report["quantized_modules"] == ["linear"] + parameter = model.linear.parametrizations.weight.original1 + assert isinstance(parameter, TrainingWeight) + inputs = torch.randn(8, 64, device=cuda()) + target = torch.arange(8, device=cuda()) % 4 + optimizer = torch.optim.AdamW(model.parameters(), lr=0.01) + forward = torch.compile(model, fullgraph=True) if compiled else model + before = parameter.dequantize().detach().clone() + for _ in range(3): + optimizer.zero_grad(set_to_none=True) + with torch.autocast("cuda", dtype=torch.float16): + loss = torch.nn.functional.cross_entropy(forward(inputs), target) + loss.backward() + assert torch.isfinite(parameter.grad).all() and parameter.grad.norm() > 0 + optimizer.step() + assert not torch.equal(before, parameter.dequantize()) + model.eval() + with torch.inference_mode(): + complete = model(inputs[:1]) + model.set_active_features([0, 2, 3]) + selected = model(inputs[:1]) + torch.testing.assert_close(selected, complete[:, [0, 2, 3]]) From 828d5d7fcf7d4cda8957f7fce09f6382de92a9cd Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 23:18:41 +0200 Subject: [PATCH 013/155] perf: add opt-in CUDA batch transfer lookahead and measurements --- dev/README.md | 23 ++++++++ dev/benchmarks/README.md | 32 ++++++++++ dev/benchmarks/run.py | 7 +++ dev/benchmarks/transfer.py | 94 +++++++++++++++++++++++++++++ mini_trainer/data/_prefetch.py | 78 ++++++++++++++++++++++++ mini_trainer/data/loader.py | 15 ++++- mini_trainer/predict.py | 6 ++ mini_trainer/train.py | 6 ++ tests/utils/test_loader.py | 105 +++++++++++++++++++++++++++++++++ 9 files changed, 365 insertions(+), 1 deletion(-) create mode 100644 dev/benchmarks/transfer.py create mode 100644 mini_trainer/data/_prefetch.py diff --git a/dev/README.md b/dev/README.md index 90984d4..0507421 100644 --- a/dev/README.md +++ b/dev/README.md @@ -154,3 +154,26 @@ hardware cases. These are optimizer/AMP tests, not an EMA-functionality claim. References: [optimizer post-step hooks](https://docs.pytorch.org/docs/2.12/generated/torch.optim.Optimizer.register_step_post_hook.html) and [GradScaler](https://docs.pytorch.org/docs/2.12/amp.html). + +### CUDA batch transfer lookahead + +`mt_train --cuda-prefetch` and `mt_predict --cuda-prefetch` opt into one-batch +transfer lookahead. Python loader builders accept `cuda_prefetch=True` with a CUDA +`device`. The returned object still inherits `DataLoader`, with the same sampler, +length, worker settings and repeated-epoch behavior; its batches are already on +the requested CUDA device. Shape, dtype and order are preserved. Model +preprocessing/augmentation stays on the caller's compute stream, so this works +independently of float32, AMP or INT8 model execution. + +The option defaults off, rejects CPU targets, and is bypassed for an already +CUDA-cached dataset. It stages one additional batch and records stream usage so +the allocator cannot recycle batch storage before consumption completes. It does +not increase worker counts or add CPU reader threads. Custom CPU hooks may run a +batch earlier; stochastic hooks sharing global RNG state can therefore change +their interleaving with caller code. Returned batches may outlive iteration; +callers using them on another CUDA stream must establish their own stream handoff. +See [PyTorch stream semantics](https://docs.pytorch.org/docs/main/notes/cuda.html#cuda-streams). + +This is an opt-in throughput/memory tradeoff. Actual gains depend on the balance +between transfer and compute; compare peak allocation as well as wall time using +[the transfer probe](benchmarks/README.md#cuda-transfer-overlap). diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 421ac91..c34d3ac 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -314,3 +314,35 @@ locally but has not been dispatched on a self-hosted runner from this session. slower QT training on these small workloads. Whole-model compilation defaults off; first-use kernel compilation is still included in wall time. Compare matching configurations and compiler cache conditions before making performance claims. + +## CUDA transfer overlap + +```bash +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.transfer +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.transfer --backward +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.transfer --dtype float32 +``` + +The probe compares the ordinary loader with `cuda_prefetch=True`, using the same +CPU-cached uint8 images, pinned batches and fixed ResNet18 weights. BatchNorm is +frozen; the optional backward pass includes gradient computation but no optimizer +updates. Model/cache construction is excluded, one trial warms both paths, and +five measured trials alternate order. Every run checks bitwise-identical outputs. +JSON records timings, hardware, PyTorch version and peak allocated CUDA memory. +This measures transfer plus model computation, not convergence or complete trainer +throughput. The dataset harness also accepts `--cuda-prefetch` for actual training, +checkpoint reload and held-out inference profiles, including INT8 models. + +On the RTX 3080 Ti Laptop GPU with PyTorch 2.12.0+cu130, one CPU thread, 512 images +at 224x224, batch size 32 and FP16 AMP, median throughput increased from 3,206 to +3,310 images/s for inference (1.03x), and 1,061 to 1,078 images/s for +forward/backward (1.02x). Peak CUDA allocation increased by 9,569,792 bytes in each +case, consistent with staging an additional input batch plus small overheads. +These modest local observations are diagnostic, not portable performance gates. + +The matching float32 inference probe was slower with lookahead: 2,002 versus +1,918 images/s (0.96x). Keep the default off unless measurements on the intended +workload justify the additional stream and memory. The integrated INT8 synthetic +training/checkpoint/inference profile passed its 100% oracle gate with prefetch +and reproduced the earlier QT held-out scores bit for bit; that establishes +compatibility, not a training speedup. diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index a028066..9b7c4ab 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -43,6 +43,7 @@ def run( dataset: str = "synthetic", data_root: str | Path | None = None, class_spec: str | Path | None = None, + cuda_prefetch: bool = False, quantized_training: bool = False, compile: bool = False, hidden: int = 0, @@ -140,6 +141,7 @@ def run( "data_index": str(data_index), "cache": cache, "cache_workers": cache_workers, + "cuda_prefetch": cuda_prefetch, }, optimizer_builder_kwargs={"optimizer_cls": MuonAuxAdamW, "lr": 0.1 if dataset == "synthetic" else 0.01, "weight_decay": 0.0}, criterion_builder_kwargs={"label_smoothing": 0.0}, @@ -162,6 +164,7 @@ def run( images=[str(root / record["path"]) for record in test_records], resize_size=size, batch_size=batch_size, + cuda_prefetch=cuda_prefetch, num_workers=0, device=target_device, dtype=torch.float32, @@ -210,6 +213,7 @@ def run( "hidden": hidden, "batch_size": batch_size, "cache_workers": cache_workers, + "cuda_prefetch": cuda_prefetch, "parameter_bytes": sum( parameter.int_data.numel() + parameter.scale.numel() * parameter.scale.element_size() if getattr(parameter, "_is_quantized_training", False) @@ -272,6 +276,7 @@ def main(): parser.add_argument("--num-workers", type=int, default=0) parser.add_argument("--cache-workers", type=int) parser.add_argument("--quantized-training", action="store_true") + parser.add_argument("--cuda-prefetch", action="store_true") parser.add_argument("--compile", action="store_true") parser.add_argument("--hidden", type=int, default=0) parser.add_argument("--batch-size", type=int, default=32) @@ -300,6 +305,7 @@ def main(): dataset=args.dataset, data_root=args.data_root, class_spec=args.class_spec, + cuda_prefetch=args.cuda_prefetch, quantized_training=args.quantized_training, compile=args.compile, hidden=args.hidden, @@ -317,6 +323,7 @@ def main(): "cache": args.cache, "seed": args.seed, "epochs": args.epochs, + "cuda_prefetch": args.cuda_prefetch, "quantized_training": args.quantized_training, "compile": args.compile, "hidden": args.hidden, diff --git a/dev/benchmarks/transfer.py b/dev/benchmarks/transfer.py new file mode 100644 index 0000000..1e16470 --- /dev/null +++ b/dev/benchmarks/transfer.py @@ -0,0 +1,94 @@ +"""Compare same-stream H2D with one-batch lookahead around fixed ResNet compute.""" + +import json +import statistics +import time +from argparse import ArgumentParser + +import torch +from torchvision.models import resnet18 + +from mini_trainer.data.io import LazyDataset +from mini_trainer.data.loader import get_dataloader + + +def run(samples=512, size=224, batch_size=32, repeats=5, backward=False, dtype="float16"): + if not torch.cuda.is_available(): + raise RuntimeError("CUDA transfer benchmark requires an accessible CUDA device.") + torch.manual_seed(42) + device = torch.device("cuda:0") + images = torch.randint(0, 256, (samples, 3, size, size), dtype=torch.uint8) + dataset = LazyDataset(lambda item: images[item[0]], (list(range(samples)),), cache="cpu", cache_workers=0) + loaders = { + name: get_dataloader(dataset, "val", batch_size, 0, True, device, cuda_prefetch=prefetch) + for name, prefetch in (("same_stream", False), ("prefetch", True)) + } + model = resnet18(weights=None).eval().to(device) + precision = getattr(torch, dtype) + timings = {name: [] for name in loaders} + peaks = {name: [] for name in loaders} + reference = None + for trial in range(repeats + 1): + for name in list(loaders) if trial % 2 else list(reversed(loaders)): + model.zero_grad(set_to_none=True) + torch.cuda.synchronize(device) + torch.cuda.reset_peak_memory_stats(device) + started = time.perf_counter() + outputs = [] + with torch.set_grad_enabled(backward): + for batch in loaders[name]: + inputs = batch.to(device, non_blocking=True).float().div_(255) + with torch.autocast("cuda", dtype=precision, enabled=precision != torch.float32): + output = model(inputs) + outputs.append(output.detach()) + if backward: + output.float().square().mean().backward() + model.zero_grad(set_to_none=True) + torch.cuda.synchronize(device) + elapsed = time.perf_counter() - started + peak = torch.cuda.max_memory_allocated(device) + actual = torch.cat(outputs).cpu() + if reference is None: + reference = actual + torch.testing.assert_close(actual, reference, rtol=0, atol=0) + if trial: + timings[name].append(elapsed) + peaks[name].append(peak) + return { + "samples": samples, + "size": size, + "batch_size": batch_size, + "workers": 0, + "dtype": dtype, + "backward": backward, + "device": torch.cuda.get_device_name(device), + "torch_version": torch.__version__, + "identical_outputs": True, + "scope": ( + "CPU cache iteration, pinned H2D, scaling and fixed ResNet18 compute; " + "BN frozen; no optimizer updates; excludes cache/model construction" + ), + "seconds": timings, + "peak_cuda_allocated_bytes": peaks, + "median_samples_per_second": {name: samples / statistics.median(values) for name, values in timings.items()}, + "speedup": statistics.median(timings["same_stream"]) / statistics.median(timings["prefetch"]), + } + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--samples", type=int, default=512) + parser.add_argument("--size", type=int, default=224) + parser.add_argument("--batch-size", type=int, default=32) + parser.add_argument("--repeats", type=int, default=5) + parser.add_argument("--backward", action="store_true") + parser.add_argument("--dtype", choices=["float32", "float16", "bfloat16"], default="float16") + args = parser.parse_args() + if min(args.samples, args.size, args.batch_size, args.repeats) < 1: + parser.error("Sizes and repeats must be positive") + torch.set_num_threads(1) + print(json.dumps(run(**vars(args)), indent=2)) + + +if __name__ == "__main__": + main() diff --git a/mini_trainer/data/_prefetch.py b/mini_trainer/data/_prefetch.py new file mode 100644 index 0000000..9a1ab86 --- /dev/null +++ b/mini_trainer/data/_prefetch.py @@ -0,0 +1,78 @@ +"""One-batch CUDA transfer lookahead, independent of model precision.""" + +import torch +from torch.utils.data import DataLoader + + +def _map_tensors(batch, operation): + if isinstance(batch, torch.Tensor): + return operation(batch) + if isinstance(batch, dict): + return {key: _map_tensors(value, operation) for key, value in batch.items()} + if isinstance(batch, tuple) and hasattr(batch, "_fields"): + return type(batch)(*(_map_tensors(value, operation) for value in batch)) + if isinstance(batch, (tuple, list)): + return type(batch)(_map_tensors(value, operation) for value in batch) + return batch + + +class CUDAPrefetchLoader(DataLoader): + """DataLoader whose opt-in iterator yields tensors on the requested CUDA device. + + The sampler and worker lifecycle remain those of DataLoader. Only one extra + device batch is staged. Preprocessing and augmentation stay on the caller's + compute stream. CPU hooks may run one batch earlier than without lookahead. + """ + + def __init__(self, *args, device, **kwargs): + self.transfer_device = torch.device(device) + if self.transfer_device.type != "cuda": + raise ValueError("CUDA batch prefetch requires a CUDA target device.") + super().__init__(*args, **kwargs) + + def __iter__(self): + return self._prefetch(super().__iter__()) + + def _prefetch(self, source): + stream = torch.cuda.Stream(device=self.transfer_device) + stream.wait_stream(torch.cuda.current_stream(self.transfer_device)) + sentinel = object() + + def preload(): + batch = next(source, sentinel) + if batch is sentinel: + return sentinel + producer = torch.cuda.current_stream(self.transfer_device) + + def transfer(tensor): + if tensor.device.type == "cuda": + if tensor.device != stream.device: + raise ValueError("CUDA prefetch cannot stage tensors from another CUDA device.") + stream.wait_stream(producer) + tensor.record_stream(stream) + return tensor.to(self.transfer_device, non_blocking=True) + + with torch.cuda.stream(stream): + return _map_tensors(batch, transfer) + + pending = preload() + while pending is not sentinel: + current = torch.cuda.current_stream(self.transfer_device) + # Wait only for this batch, before queueing the next copy. + current.wait_stream(stream) + + def record(tensor): + tensor.record_stream(current) + return tensor + + batch = _map_tensors(pending, record) + error = None + try: + pending = preload() + except Exception as caught: + # Deliver the already loaded batch before surfacing a later + # reader failure, as ordinary sequential iteration would. + error = caught + yield batch + if error is not None: + raise error diff --git a/mini_trainer/data/loader.py b/mini_trainer/data/loader.py index 08877e5..8533c51 100644 --- a/mini_trainer/data/loader.py +++ b/mini_trainer/data/loader.py @@ -8,6 +8,7 @@ from mini_trainer import get_logger from mini_trainer.utils import is_dist_avail_and_initialized +from ._prefetch import CUDAPrefetchLoader from ._workers import _default_worker_count from .io import ( CACHE_MODE, @@ -83,6 +84,7 @@ def get_dataloader( # noqa: D103 pin_memory: bool, device: torch.device, *, + cuda_prefetch: bool = False, prefetch_factor: int | None = None, multiprocessing_context: str | None = None, ): @@ -105,8 +107,11 @@ def get_dataloader( # noqa: D103 sampler = BatchSampler(base_sampler, batch_size=batch_size, drop_last=drop_last) - return DataLoader( + loader_cls = CUDAPrefetchLoader if cuda_prefetch else DataLoader + transfer_kwargs = {"device": device} if cuda_prefetch else {} + return loader_cls( dataset, + **transfer_kwargs, batch_sampler=sampler, collate_fn=_collate_batch, num_workers=num_workers, @@ -130,6 +135,7 @@ def get_dataset_dataloader( # noqa: D103 cache: CACHE_MODE | str | int | None = None, cache_workers: int | None = None, multilabel: bool = False, + cuda_prefetch: bool = False, prefetch_factor: int | None = None, multiprocessing_context: str | None = None, hook: Callable[[torch.Tensor], torch.Tensor] | None = None, @@ -137,6 +143,8 @@ def get_dataset_dataloader( # noqa: D103 resize_size = _normalize_resize_size(resize_size, error_suffix=".") if isinstance(device, str): device = torch.device(device) + if cuda_prefetch and device.type != "cuda": + raise ValueError("CUDA batch prefetch requires a CUDA target device.") if len(metadata) != len(modes): raise ValueError(f"Number of supplied datasets: {len(metadata)} and modes: {len(modes)} do not match!") @@ -179,6 +187,7 @@ def get_dataset_dataloader( # noqa: D103 num_workers, pin_memory, device, + cuda_prefetch=cuda_prefetch and cache is not CACHE_MODE.CUDA, prefetch_factor=prefetch_factor, multiprocessing_context=multiprocessing_context, ) @@ -199,11 +208,14 @@ def get_inference_dataloader( # noqa: D103 hook: Callable[[torch.Tensor], torch.Tensor] | None = None, prefetch_factor: int | None = None, multiprocessing_context: str | None = None, + cuda_prefetch: bool = False, **kwargs, ): resize_size = _normalize_resize_size(resize_size) if isinstance(device, str): device = torch.device(device) + if cuda_prefetch and device.type != "cuda": + raise ValueError("CUDA batch prefetch requires a CUDA target device.") if subsample is not None and subsample > 1: images = images[::subsample] @@ -224,6 +236,7 @@ def get_inference_dataloader( # noqa: D103 num_workers, device.type == "cuda", device, + cuda_prefetch=cuda_prefetch, prefetch_factor=prefetch_factor, multiprocessing_context=multiprocessing_context, ) diff --git a/mini_trainer/predict.py b/mini_trainer/predict.py index a536c9c..8bca167 100644 --- a/mini_trainer/predict.py +++ b/mini_trainer/predict.py @@ -276,6 +276,12 @@ def cli(description="Classify images with a trained model", **extra_kwargs): # help="Number of images used in each mini-batch for training/validation (default=64).", ) cfg_args = parser.add_argument_group("Runtime [optional]") + cfg_args.add_argument( + "--cuda-prefetch", + action="store_true", + dest="dataloader_builder_kwargs.cuda_prefetch", + help="Stage one CPU batch ahead on a CUDA transfer stream (opt-in; uses extra device memory).", + ) cfg_args.add_argument( "--subsample", type=int, diff --git a/mini_trainer/train.py b/mini_trainer/train.py index 4a0d54d..9fffe3e 100644 --- a/mini_trainer/train.py +++ b/mini_trainer/train.py @@ -531,6 +531,12 @@ def cli(description="Train a classifier", **extra_kwargs): # noqa: D103 ) cfg_args = parser.add_argument_group("Runtime [optional]") + cfg_args.add_argument( + "--cuda-prefetch", + action="store_true", + dest="dataloader_builder_kwargs.cuda_prefetch", + help="Stage one CPU batch ahead on a CUDA transfer stream (opt-in; uses extra device memory).", + ) cfg_args.add_argument( "--subsample", type=int, diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index 9f63a19..0a9624d 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -237,3 +237,108 @@ def test_bounded_cuda_cache_matches_cpu_cache(metadata): for cpu, gpu in zip(expected, actual, strict=True): assert gpu.device == device torch.testing.assert_close(gpu.cpu(), cpu, rtol=0, atol=0) + + +def test_cuda_prefetch_rejects_cpu_target(metadata): + with pytest.raises(ValueError, match="CUDA target"): + get_inference_dataloader(metadata["path"], resize_size=4, num_workers=0, cuda_prefetch=True) + + +def _require_prefetch_cuda(): + import os + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify CUDA transfer streams") + assert torch.cuda.is_available() + + +@pytest.mark.parametrize("workers", [0, 1]) +def test_cuda_prefetch_order_epochs_tail_and_lifetime(metadata, workers): + _require_prefetch_cuda() + from torch.utils.data import DataLoader + + datasets, loaders = get_dataset_dataloader( + metadata, + modes=("val",), + resize_size=4, + batch_size=2, + num_workers=workers, + device="cuda:0", + cache="CPU", + cache_workers=0, + cuda_prefetch=True, + multiprocessing_context="spawn" if workers else None, + ) + loader = loaders[0] + assert isinstance(loader, DataLoader) + assert loader.dataset is datasets[0] and len(loader) == 3 + consumer = torch.cuda.Stream() + for _ in range(2): + outputs = [] + with torch.cuda.stream(consumer): + for images, labels in loader: + assert images.device == labels.device == torch.device("cuda:0") + assert images.dtype == torch.uint8 and labels.dtype == torch.long + # Queue use then release the batch while the allocator may reuse + # its copy-stream storage for subsequent batches. + outputs.append((images.float().mean((1, 2, 3)), labels + 0)) + del images, labels + consumer.synchronize() + assert torch.cat([result[1] for result in outputs]).tolist() == list(range(5)) + torch.testing.assert_close(torch.cat([result[0] for result in outputs]).cpu(), torch.arange(5).float()) + iterator = iter(loader) + next(iterator) + del iterator + assert sum(len(images) for images, _ in loader) == 5 + + +def test_cuda_prefetch_inference_empty_and_failure(metadata): + _require_prefetch_cuda() + from mini_trainer.data._prefetch import CUDAPrefetchLoader + + _, loader = get_inference_dataloader( + metadata["path"], + resize_size=4, + batch_size=2, + num_workers=0, + device="cuda:0", + cuda_prefetch=True, + ) + result = torch.cat(list(loader)) + torch.testing.assert_close(result.float().mean((1, 2, 3)).cpu(), torch.arange(5).float()) + empty = CUDAPrefetchLoader([], device="cuda:0", batch_size=2) + assert list(empty) == [] + + class Broken(torch.utils.data.Dataset): + def __len__(self): + return 3 + + def __getitem__(self, index): + if index == 1: + raise RuntimeError("broken sample") + return torch.tensor(index) + + iterator = iter(CUDAPrefetchLoader(Broken(), device="cuda:0", batch_size=1, pin_memory=True)) + assert next(iterator).item() == 0 + with pytest.raises(RuntimeError, match="broken sample"): + next(iterator) + + +def test_cuda_prefetch_nested_cuda_source(): + _require_prefetch_cuda() + from mini_trainer.data._prefetch import CUDAPrefetchLoader + + class Mixed(torch.utils.data.Dataset): + def __len__(self): + return 4 + + def __getitem__(self, index): + return {"value": torch.ones(1024, device="cuda:0") * index, "label": index, "name": str(index)} + + consumer = torch.cuda.Stream() + with torch.cuda.stream(consumer): + outputs = list(CUDAPrefetchLoader(Mixed(), device="cuda:0", batch_size=2)) + consumer.synchronize() + for index, output in enumerate(outputs): + assert output["name"] == [str(2 * index), str(2 * index + 1)] + torch.testing.assert_close(output["value"].mean(1), output["label"].float()) From dbb36dddc47c32418aa01c30286b85881faa3dc6 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 23:26:05 +0200 Subject: [PATCH 014/155] perf: gather CUDA-bound cached batches directly into pinned memory --- dev/README.md | 15 +++++++++++++++ dev/benchmarks/README.md | 27 +++++++++++++++++++++++++++ dev/benchmarks/loader.py | 25 +++++++++++++++++++++---- dev/benchmarks/transfer.py | 18 ++++++++++++++++-- mini_trainer/data/io.py | 19 ++++++++++++++++++- mini_trainer/data/loader.py | 24 +++++++++++++++--------- tests/utils/test_loader.py | 37 +++++++++++++++++++++++++++++++++++++ 7 files changed, 149 insertions(+), 16 deletions(-) diff --git a/dev/README.md b/dev/README.md index 0507421..1530a26 100644 --- a/dev/README.md +++ b/dev/README.md @@ -177,3 +177,18 @@ See [PyTorch stream semantics](https://docs.pytorch.org/docs/main/notes/cuda.htm This is an opt-in throughput/memory tradeoff. Actual gains depend on the balance between transfer and compute; compare peak allocation as well as wall time using [the transfer probe](benchmarks/README.md#cuda-transfer-overlap). + +### Direct pinned cache batches + +For a CUDA target with `cache="CPU"` and `num_workers=0`, the shared training +loader now gathers cached rows directly into pinned batch storage. This removes +the intermediate pageable batch and its second copy during pinning. Batches own +their storage: modifying one cannot change the cache, and keeping an older batch +cannot cause it to be overwritten by a later iteration. Sampling, label order, +shape and dtype are unchanged. + +This applies automatically with either ordinary transfer or `--cuda-prefetch`. +Raw `LazyDataset` users can request `pin_batches=True` for the same behavior. +Worker processes always use the ordinary gather path and DataLoader's parent-side +pinning; this option never initializes the CUDA pin allocator in a worker. CPU +training, CUDA-cached datasets and scalar indexing keep their existing behavior. diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index c34d3ac..a4b6455 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -346,3 +346,30 @@ workload justify the additional stream and memory. The integrated INT8 synthetic training/checkpoint/inference profile passed its 100% oracle gate with prefetch and reproduced the earlier QT held-out scores bit for bit; that establishes compatibility, not a training speedup. + +### Direct pinned gathering + +```bash +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.loader \ + --pin-batches --samples 512 --size 224 --batch-size 32 --repeats 7 +``` + +This compares gathering into pageable memory followed by pinning against gathering +directly into pinned storage. Both return identical pinned batches; measurements +exclude cache construction, H2D and model compute. On the RTX 3080 Ti Laptop host, +seven alternating measured trials after warmup gave 30,983 versus 72,680 images/s +(2.35x). This is a loader-only result, not a claim of a 2.35x training speedup. +The shared CUDA-target CPU-cache loader enables direct pinned gathering when +`num_workers=0`; workers retain parent-side pinning. + +The transfer probe now measures four variants in the same run: the former +same-stream path, transfer prefetch, direct pinned gathering, and their combination. +This separates the effects of eliminating a CPU copy and overlapping H2D with +compute. Each variant still checks exact output equality against the others. + +In the four-way ResNet probe on the same host, float32 inference measured 2,050 +images/s for the former path, 2,063 for direct pinned gathering, and 2,107 for +pinned gathering plus prefetch. FP16 forward/backward measured 1,040, 1,009 and +1,028 images/s respectively. These compute-inclusive differences are small and +include regressions despite the clear loader-only improvement. Eliminating the +CPU copy does not establish a training speedup for compute-bound models. diff --git a/dev/benchmarks/loader.py b/dev/benchmarks/loader.py index 631debe..6f604d5 100644 --- a/dev/benchmarks/loader.py +++ b/dev/benchmarks/loader.py @@ -25,7 +25,7 @@ def __getitem__(self, index): return self.dataset[index] -def run(samples=2048, size=64, batch_size=64, repeats=5): +def run(samples=2048, size=64, batch_size=64, repeats=5, pin_batches=False): generator = torch.Generator().manual_seed(42) images = torch.randint(0, 256, (samples, 3, size, size), dtype=torch.uint8, generator=generator) dataset = LazyDataset(lambda item: (images[item[0]], torch.tensor(item[0])), (list(range(samples)),), cache="cpu") @@ -33,7 +33,22 @@ def run(samples=2048, size=64, batch_size=64, repeats=5): "scalar": DataLoader(ScalarFetch(dataset), batch_size=batch_size, num_workers=0), "batched": get_dataloader(dataset, "val", batch_size, 0, False, torch.device("cpu")), } - for expected, actual in zip(loaders["scalar"], loaders["batched"], strict=True): + if pin_batches: + if not torch.cuda.is_available(): + raise RuntimeError("Pinned batch comparison requires an accessible CUDA device.") + direct = LazyDataset( + lambda item: (images[item[0]], torch.tensor(item[0])), + (list(range(samples)),), + cache="cpu", + cache_workers=0, + pin_batches=True, + ) + loaders = { + "gather_then_pin": get_dataloader(dataset, "val", batch_size, 0, True, torch.device("cuda:0")), + "pinned_gather": get_dataloader(direct, "val", batch_size, 0, True, torch.device("cuda:0")), + } + baseline, candidate = list(loaders) + for expected, actual in zip(loaders[baseline], loaders[candidate], strict=True): torch.testing.assert_close(actual, expected, rtol=0, atol=0) timings = {name: [] for name in loaders} for trial in range(repeats + 1): @@ -53,10 +68,11 @@ def run(samples=2048, size=64, batch_size=64, repeats=5): "workers": 0, "threads": torch.get_num_threads(), "identical_batches": True, + "pin_batches": pin_batches, "scope": "cached CPU loader iteration; excludes cache construction, preprocessing, H2D and model compute", "seconds": timings, "median_samples_per_second": {name: samples / statistics.median(values) for name, values in timings.items()}, - "speedup": statistics.median(timings["scalar"]) / statistics.median(timings["batched"]), + "speedup": statistics.median(timings[baseline]) / statistics.median(timings[candidate]), } @@ -66,8 +82,9 @@ def main(): parser.add_argument("--size", type=int, default=64) parser.add_argument("--batch-size", type=int, default=64) parser.add_argument("--repeats", type=int, default=5) + parser.add_argument("--pin-batches", action="store_true") args = parser.parse_args() - if min(vars(args).values()) < 1: + if min(args.samples, args.size, args.batch_size, args.repeats) < 1: parser.error("All sizes and repeat counts must be positive") torch.set_num_threads(1) print(json.dumps(run(**vars(args)), indent=2)) diff --git a/dev/benchmarks/transfer.py b/dev/benchmarks/transfer.py index 1e16470..bedaa33 100644 --- a/dev/benchmarks/transfer.py +++ b/dev/benchmarks/transfer.py @@ -19,9 +19,22 @@ def run(samples=512, size=224, batch_size=32, repeats=5, backward=False, dtype=" device = torch.device("cuda:0") images = torch.randint(0, 256, (samples, 3, size, size), dtype=torch.uint8) dataset = LazyDataset(lambda item: images[item[0]], (list(range(samples)),), cache="cpu", cache_workers=0) + direct = LazyDataset( + lambda item: images[item[0]], + (list(range(samples)),), + cache="cpu", + cache_workers=0, + pin_batches=True, + ) + variants = { + "same_stream": (dataset, False), + "prefetch": (dataset, True), + "pinned_gather": (direct, False), + "pinned_gather_prefetch": (direct, True), + } loaders = { - name: get_dataloader(dataset, "val", batch_size, 0, True, device, cuda_prefetch=prefetch) - for name, prefetch in (("same_stream", False), ("prefetch", True)) + name: get_dataloader(data, "val", batch_size, 0, True, device, cuda_prefetch=prefetch) + for name, (data, prefetch) in variants.items() } model = resnet18(weights=None).eval().to(device) precision = getattr(torch, dtype) @@ -72,6 +85,7 @@ def run(samples=512, size=224, batch_size=32, repeats=5, backward=False, dtype=" "peak_cuda_allocated_bytes": peaks, "median_samples_per_second": {name: samples / statistics.median(values) for name, values in timings.items()}, "speedup": statistics.median(timings["same_stream"]) / statistics.median(timings["prefetch"]), + "speedups": {name: statistics.median(timings["same_stream"]) / statistics.median(values) for name, values in timings.items()}, } diff --git a/mini_trainer/data/io.py b/mini_trainer/data/io.py index af8ddcb..9012e58 100644 --- a/mini_trainer/data/io.py +++ b/mini_trainer/data/io.py @@ -368,10 +368,12 @@ def __init__( # noqa: D107 cache: str | int | CACHE_MODE | None = None, *, cache_workers: int | None = None, + pin_batches: bool = False, ): if cache_workers is not None and (isinstance(cache_workers, bool) or not isinstance(cache_workers, int) or cache_workers < 0): raise ValueError("cache_workers must be a nonnegative integer or None.") self._cache_workers = cache_workers + self._pin_batches = pin_batches self.func = func self.items = tuple(np.asarray(seq, dtype=_infer_numeric_dtype(seq)) if len(seq) > 0 else np.empty((0,)) for seq in items) if self.items and any(len(seq) != len(self.items[0]) for seq in self.items): @@ -505,7 +507,22 @@ def __getitems__(self, indices): index = torch.where(index < 0, index + len(self), index) # index_select copies whole rows; generic advanced indexing is much # slower for uint8 image batches on CPU. Results own their storage. - data = tuple(tensor.index_select(0, index) for tensor in tensors) + pin = self._pin_batches and tensors[0].device.type == "cpu" and torch.utils.data.get_worker_info() is None + if pin: + # Gather directly into the final pinned batch. DataLoader's + # pinning pass then returns the same allocation, without a + # second full image copy. Never initialize CUDA in workers. + data = tuple( + torch.index_select( + tensor, + 0, + index, + out=torch.empty((len(index), *tensor.shape[1:]), dtype=tensor.dtype, device=tensor.device, pin_memory=True), + ) + for tensor in tensors + ) + else: + data = tuple(tensor.index_select(0, index) for tensor in tensors) return _FetchedBatch(data[0] if self._ram_was_single_tensor else data) return _FetchedBatch(self[indices]) diff --git a/mini_trainer/data/loader.py b/mini_trainer/data/loader.py index 8533c51..2c7654d 100644 --- a/mini_trainer/data/loader.py +++ b/mini_trainer/data/loader.py @@ -163,21 +163,27 @@ def get_dataset_dataloader( # noqa: D103 proc_path_label = PathLabelProcessor(reader, hook, multilabel) - datasets = [] - for mode, data in zip(modes, metadata): - if mode.strip().lower() == "train" and resample: - raise NotImplementedError("Resampling is currently not supported.") - dset = LazyDataset(func=proc_path_label, items=(data["path"], data["class"]), cache=cache, cache_workers=cache_workers) - datasets.append(dset) - if cache is CACHE_MODE.CUDA: # When the entire dataset is preloaded there is no need to use multiprocessing for dataloading num_workers = 0 elif num_workers is None: num_workers = _default_worker_count(16) - # A gather from a pinned cache allocates an unpinned result. Pin the actual - # CPU batch before asynchronous H2D, including for the CPU cache path. + datasets = [] + for mode, data in zip(modes, metadata): + if mode.strip().lower() == "train" and resample: + raise NotImplementedError("Resampling is currently not supported.") + dset = LazyDataset( + func=proc_path_label, + items=(data["path"], data["class"]), + cache=cache, + cache_workers=cache_workers, + pin_batches=device.type == "cuda" and cache is CACHE_MODE.CPU and num_workers == 0, + ) + datasets.append(dset) + + # Main-process CPU cache gathers are already pinned. Decoded and worker + # batches are pinned by DataLoader before asynchronous H2D. pin_memory = device.type == "cuda" and cache is not CACHE_MODE.CUDA loaders = [ get_dataloader( diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index 0a9624d..a82ff7a 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -342,3 +342,40 @@ def __getitem__(self, index): for index, output in enumerate(outputs): assert output["name"] == [str(2 * index), str(2 * index + 1)] torch.testing.assert_close(output["value"].mean(1), output["label"].float()) + + +def test_pinned_cache_gather_storage_and_indices(metadata): + _require_prefetch_cuda() + datasets, loaders = get_dataset_dataloader( + metadata, + modes=("val",), + resize_size=4, + batch_size=2, + num_workers=0, + device="cuda:0", + cache="CPU", + cache_workers=0, + ) + dataset = datasets[0] + assert dataset._pin_batches + fetched = dataset.__getitems__([4, 1, 1, -1]) + images, labels = data_loader._collate_batch(fetched) + assert images.is_pinned() and labels.is_pinned() + assert images.pin_memory().data_ptr() == images.data_ptr() + assert labels.tolist() == [4, 1, 1, 4] + before = dataset[1][0].clone() + images[1].zero_() + torch.testing.assert_close(dataset[1][0], before) + retained = next(iter(loaders[0]))[0] + copied = retained.clone() + list(loaders[0]) + torch.testing.assert_close(retained, copied) + + +def test_pinned_cache_gather_never_pins_inside_worker(monkeypatch): + dataset = data_io.LazyDataset(lambda item: torch.tensor(item[0]), ([0, 1, 2],), cache="CPU", cache_workers=0, pin_batches=True) + monkeypatch.setattr(torch.utils.data, "get_worker_info", lambda: object()) + # CPU-only execution also proves this branch does not need a pin allocator. + result = data_loader._collate_batch(dataset.__getitems__([2, 0])) + assert not result.is_pinned() + assert result.tolist() == [2, 0] From f5a8c35ab630ff2d42ae7b5138f9a1f643ac9914 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 23:34:36 +0200 Subject: [PATCH 015/155] fix: save portable checkpoints from compiled training models --- mini_trainer/trainer.py | 6 ++++- tests/test_checkpoint_contract.py | 37 +++++++++++++++++++++++++++++++ 2 files changed, 42 insertions(+), 1 deletion(-) diff --git a/mini_trainer/trainer.py b/mini_trainer/trainer.py index f076507..25e91bc 100644 --- a/mini_trainer/trainer.py +++ b/mini_trainer/trainer.py @@ -353,7 +353,11 @@ def train( if is_best_eval: best_epoch = epoch if output_dir is not None: - raw_model = model.module if hasattr(model, "module") else model + raw_model = model + # Serialize architecture state, not compile/distribution wrappers. + # This also leaves the QT recipe at its original module path. + while isinstance(raw_model, (nn.DataParallel, DDP)) or hasattr(raw_model, "_orig_mod"): + raw_model = raw_model._orig_mod if hasattr(raw_model, "_orig_mod") else raw_model.module assert isinstance(raw_model, nn.Module) checkpoint = { "model": raw_model.state_dict(), diff --git a/tests/test_checkpoint_contract.py b/tests/test_checkpoint_contract.py index 72f1343..a0348ed 100644 --- a/tests/test_checkpoint_contract.py +++ b/tests/test_checkpoint_contract.py @@ -189,3 +189,40 @@ def test_ema_status_warning_only_when_enabled(): assert not caught with pytest.warns(RuntimeWarning, match="temporarily nonfunctional"): EMATeacher(enable=True, total_steps=1, model=torch.nn.Linear(2, 2)) + + +def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatch): + # Exercise Dynamo's real OptimizedModule wrapper without testing CPU + # Inductor performance. Its state_dict adds _orig_mod unless unwrapped. + compile_model = torch.compile + monkeypatch.setattr(torch, "compile", lambda model: compile_model(model, backend="eager")) + for label in ("class_a", "class_b"): + (tmp_path / "data" / label).mkdir(parents=True) + args = { + "input": str(tmp_path / "data"), + "output": str(tmp_path), + "name": "compiled", + "epochs": 1, + "device": "cpu", + "dtype": "float32", + "seed": 42, + "compile": True, + "ema": False, + "builder": DeterministicBuilder, + "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": False}, + "dataloader_builder_kwargs": {"batch_size": 4}, + "lr_schedule_builder_kwargs": {"warmup_epochs": 0}, + "logger_builder_kwargs": {"verbose": False}, + } + train_module.main(**args) + path = tmp_path / "compiled/weights/checkpoint_last.pth" + state = torch.load(path, weights_only=True) + assert not any(key.startswith("_orig_mod.") for key in state["model"]) + eager = torch.load(tmp_path / "compiled/weights/last.pt", weights_only=True) + assert_state_equal(state["model"], eager) + args.update(name="resumed", epochs=2, checkpoint=str(path)) + args["model_builder_kwargs"]["model_type"] = TinyMockModel() + train_module.main(**args) + resumed = torch.load(tmp_path / "resumed/weights/checkpoint_last.pth", weights_only=True) + assert resumed["epoch"] == 1 + assert not any(key.startswith("_orig_mod.") for key in resumed["model"]) From 7ca72e245037068f0799ff91031a6012523976e4 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Tue, 8 Sep 2026 23:45:33 +0200 Subject: [PATCH 016/155] bench: preserve CUDA peaks and compare dense MNIST quantized training --- .github/workflows/benchmarks.yml | 6 ++- dev/benchmarks/README.md | 30 +++++++++++++ dev/benchmarks/models.py | 28 ++++++++++++ dev/benchmarks/performance.py | 73 +++++++++++++++++++++++++++++++ dev/benchmarks/run.py | 56 ++++++++++++++++++++++-- dev/benchmarks/summarize.py | 11 ++++- dev/check-benchmarks.sh | 11 ++++- docs/benchmarks.md | 52 +++++++++++++++++++--- docs/quantized-training.md | 8 ++++ tests/test_benchmark_datasets.py | 56 ++++++++++++++++++++++++ tests/test_benchmark_synthetic.py | 4 ++ 11 files changed, 321 insertions(+), 14 deletions(-) create mode 100644 dev/benchmarks/performance.py diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index f49890d..ca84134 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -95,6 +95,9 @@ jobs: - name: Paired real-data quantized training if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() run: bash dev/check-benchmarks.sh qt-real benchmark-qt-real + - name: Paired dense MNIST quantized training + if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() + run: bash dev/check-benchmarks.sh qt-dense benchmark-qt-dense - name: Synthetic GPU precision profiles run: bash dev/check-benchmarks.sh gpu benchmark-gpu - name: Real-data progression @@ -104,7 +107,7 @@ jobs: if: always() run: | python3 -m dev.benchmarks.summarize benchmark-gpu >> "$GITHUB_STEP_SUMMARY" - for qt_results in benchmark-qt benchmark-qt-real; do + for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense; do if [[ -d "$qt_results" ]]; then python3 -m dev.benchmarks.summarize "$qt_results" >> "$GITHUB_STEP_SUMMARY" fi @@ -124,6 +127,7 @@ jobs: benchmark-real/ benchmark-qt/ benchmark-qt-real/ + benchmark-qt-dense/ !benchmark-qt/**/data/**/*.png !benchmark-gpu/**/data/**/*.png - name: Remove the disposable GPU environment diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index a4b6455..daa0045 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -373,3 +373,33 @@ pinned gathering plus prefetch. FP16 forward/backward measured 1,040, 1,009 and 1,028 images/s respectively. These compute-inclusive differences are small and include regressions despite the clear loader-only improvement. Eliminating the CPU copy does not establish a training speedup for compute-bound models. + +## Dense real-data QT comparison + +```bash +CUDA_VISIBLE_DEVICES=0 TORCHINDUCTOR_COMPILE_THREADS=1 BENCHMARK_DATA_ROOT=/path/to/datasets \ + bash dev/check-benchmarks.sh qt-dense /tmp/benchmarks-qt-dense +``` + +This profile uses MNIST with a dense spatial image MLP: input pooling to 28x28, +three 2048-wide Linear/ReLU layers, and the repository's classifier head. It is +an explicit compute profile, not a proposed CNN replacement. Both paths use +15 epochs, batch size 128, seed 42, SGD with momentum 0.9, head LR 0.3 (backbone +LR 0.1), zero weight decay, FP16 AMP and a CPU cache with zero workers. The model +is compiled; optimizer updates are currently eager. No pretrained weights, +augmentation or test-set selection is used. The ordinary benchmark defaults +remain unchanged. `--model-profile`, `--optimizer` and `--learning-rate` also +allow explicit additional recipes whose configuration is retained in reports. + +QT plus real-data Actions profiles now include this pair and retain its reports +and summaries. This wiring has not been dispatched from this development session. + +The benchmark logger preserves CUDA peaks before every batch/phase reset and records +synchronized train/evaluation batch-loop timing and allocation peaks. The report +also retains total training-call wall time, including setup, figures, checkpoint +writes and first-use compilation. Per-phase timing excludes figures/checkpoints; +first-epoch compilation remains visible. Earlier dataset reports read the peak +only after the logger's final reset; their CUDA readings are now marked +`unverified`, and cannot establish whole-run memory reductions. This correction +does not affect the standalone kernel and transfer probes, which do not use that +logger. It also does not affect recorded accuracy or physical parameter storage. diff --git a/dev/benchmarks/models.py b/dev/benchmarks/models.py index 1a62384..2f0a890 100644 --- a/dev/benchmarks/models.py +++ b/dev/benchmarks/models.py @@ -46,3 +46,31 @@ def __init__(self): def forward(self, images): return self.fc(self.features(images)) + + +class DenseImageMLP(torch.nn.Module): + """Spatial image MLP for measuring QT when Linear dominates the backbone. + + This intentionally dense architecture is a compute profile, not a proposed + replacement for CNNs or a model-quality baseline. Its fixed input pooling + keeps construction and checkpoint restoration independent of dataset size. + """ + + default_transform = Compose([torch.nn.Identity()]) + + def __init__(self): + super().__init__() + self.features = torch.nn.Sequential( + torch.nn.AdaptiveAvgPool2d((28, 28)), + torch.nn.Flatten(), + torch.nn.Linear(3 * 28 * 28, 2048), + torch.nn.ReLU(), + torch.nn.Linear(2048, 2048), + torch.nn.ReLU(), + torch.nn.Linear(2048, 2048), + torch.nn.ReLU(), + ) + self.fc = torch.nn.Linear(2048, 10) + + def forward(self, images): + return self.fc(self.features(images)) diff --git a/dev/benchmarks/performance.py b/dev/benchmarks/performance.py new file mode 100644 index 0000000..503a01e --- /dev/null +++ b/dev/benchmarks/performance.py @@ -0,0 +1,73 @@ +"""Preserve benchmark timing and memory across the logger's phase resets.""" + +import json +import time +from pathlib import Path + +import torch + +from mini_trainer.logging import MultiLogger + + +class BenchmarkLogger(MultiLogger): + def __init__(self, *args, measurement_device="cpu", **kwargs): + self.measurement_device = torch.device(measurement_device) + self.peak_allocated_bytes = 0 + self.phase_measurements = [] + self.phase_active = False + self.phase_peak_allocated_bytes = 0 + super().__init__(*args, **kwargs) + + def _reset_cuda_memory_stats(self): + if self.measurement_device.type == "cuda": + peak = torch.cuda.max_memory_allocated(self.measurement_device) + self.peak_allocated_bytes = max(self.peak_allocated_bytes, peak) + if self.phase_active: + self.phase_peak_allocated_bytes = max(self.phase_peak_allocated_bytes, peak) + super()._reset_cuda_memory_stats() + + def _synchronize(self): + if self.measurement_device.type == "cuda": + torch.cuda.synchronize(self.measurement_device) + + def start_timing(self): + self._synchronize() + super().start_timing() + self.phase_started = time.perf_counter() + self.phase_active = True + self.phase_peak_allocated_bytes = 0 + + def stop_timing(self): + self._synchronize() + self.phase_measurements.append( + { + "epoch": self._epoch, + "phase": self._type, + "seconds": time.perf_counter() - self.phase_started, + "peak_cuda_allocated_bytes": ( + max(self.phase_peak_allocated_bytes, torch.cuda.max_memory_allocated(self.measurement_device)) + if self.measurement_device.type == "cuda" + else None + ), + } + ) + self.phase_active = False + super().stop_timing() + + def finish(self): + super().finish() # Captures the last peak before the final reset. + if self.output_dir is not None: + Path(self.output_dir, "performance.json").write_text( + json.dumps( + { + "phases": self.phase_measurements, + "peak_cuda_allocated_bytes": self.peak_allocated_bytes if self.measurement_device.type == "cuda" else None, + "phase_scope": ( + "synchronized batch loop including loader, preprocessing, compute and batch logging; " + "excludes figures and checkpoints" + ), + }, + indent=2, + ) + + "\n" + ) diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 9b7c4ab..6662ac1 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -24,10 +24,18 @@ from .datasets import prepare_real from .models import NoAugmentationBuilder +from .performance import BenchmarkLogger from .synthetic import generate +class BenchmarkBuilder(NoAugmentationBuilder): + @staticmethod + def build_logger(**kwargs): + return BenchmarkLogger(**kwargs) + + class HierarchicalBenchmarkBuilder(HierarchicalBuilder): + build_logger = staticmethod(BenchmarkBuilder.build_logger) build_augmentation = staticmethod(NoAugmentationBuilder.build_augmentation) @@ -49,7 +57,16 @@ def run( hidden: int = 0, batch_size: int = 32, cache_workers: int | None = None, + model_profile: str = "default", + optimizer: str = "muon", + learning_rate: float | None = None, ): + if model_profile not in ("default", "dense") or optimizer not in ("muon", "adamw", "sgd"): + raise ValueError("Unknown model or optimizer profile.") + if learning_rate is None: + learning_rate = 0.1 if dataset == "synthetic" else 0.01 + if not np.isfinite(learning_rate) or learning_rate <= 0: + raise ValueError("Learning rate must be finite and positive.") matplotlib.use("Agg", force=True) cache = "CPU" if cache == "RAM" else cache target_device = torch.device(device) @@ -99,6 +116,8 @@ def run( spec_path.write_text(json.dumps(spec, indent=2) + "\n") size = 28 if dataset == "mnist" else 64 model_type = "dev.benchmarks.models:TinyConv" + if model_profile == "dense": + model_type = "dev.benchmarks.models:DenseImageMLP" hierarchical = dataset == "blair" records = manifest["records"] train_records = [record for record in records if record["split"] != "test"] @@ -125,7 +144,7 @@ def run( epochs=epochs, size=size, seed=seed, - builder=HierarchicalBenchmarkBuilder if hierarchical else NoAugmentationBuilder, + builder=HierarchicalBenchmarkBuilder if hierarchical else BenchmarkBuilder, ema=False, quantized_training=quantized_training, compile=compile, @@ -143,16 +162,26 @@ def run( "cache_workers": cache_workers, "cuda_prefetch": cuda_prefetch, }, - optimizer_builder_kwargs={"optimizer_cls": MuonAuxAdamW, "lr": 0.1 if dataset == "synthetic" else 0.01, "weight_decay": 0.0}, + optimizer_builder_kwargs={ + "optimizer_cls": {"muon": MuonAuxAdamW, "adamw": torch.optim.AdamW, "sgd": torch.optim.SGD}[optimizer], + "lr": learning_rate, + "weight_decay": 0.0, + **({"momentum": 0.9} if optimizer == "sgd" else {}), + }, criterion_builder_kwargs={"label_smoothing": 0.0}, regularizer_builder_kwargs={"strength": 0.0}, lr_schedule_builder_kwargs={"warmup_epochs": 0.0}, - logger_builder_kwargs={"logger_cls": []}, + logger_builder_kwargs={"logger_cls": [], "measurement_device": device}, ) if target_device.type == "cuda": torch.cuda.synchronize(target_device) elapsed = time.perf_counter() - started - peak_memory = torch.cuda.max_memory_allocated(target_device) if target_device.type == "cuda" else None + performance = json.loads((output / "training/logs/performance.json").read_text()) + peak_memory = ( + max(performance["peak_cuda_allocated_bytes"], torch.cuda.max_memory_allocated(target_device)) + if target_device.type == "cuda" + else None + ) weights = output / "training/weights/last.pt" model, preprocess = Classifier.build(weights=str(weights), device=target_device, dtype=torch.float32) model.eval() @@ -208,6 +237,10 @@ def run( "onnx": False, }, "dataset": dataset, + "model_profile": model_profile, + "optimizer": optimizer, + "learning_rate": learning_rate, + "momentum": 0.9 if optimizer == "sgd" else None, "quantization_recipe": quantization_recipe, "compile": compile, "hidden": hidden, @@ -228,6 +261,12 @@ def run( "chance_accuracy": 1 / scores_by_level[0].shape[1], "test_accuracy": accuracy, "training_wall_seconds": elapsed, + "phase_measurements": performance["phases"], + "phase_measurement_scope": performance["phase_scope"], + "phase_peak_memory_scope": "maximum across all batch resets within each timed phase", + "peak_cuda_memory_scope": ( + "maximum across logger batch/phase resets and post-training allocation; excludes final held-out inference" + ), "training_wall_scope": "setup, training, validation, logging and checkpoints; includes first-use compilation/autotuning", "num_workers_requested": num_workers, "num_workers": 0 if cache == "CUDA" else num_workers, @@ -278,6 +317,9 @@ def main(): parser.add_argument("--quantized-training", action="store_true") parser.add_argument("--cuda-prefetch", action="store_true") parser.add_argument("--compile", action="store_true") + parser.add_argument("--model-profile", choices=["default", "dense"], default="default") + parser.add_argument("--optimizer", choices=["muon", "adamw", "sgd"], default="muon") + parser.add_argument("--learning-rate", type=float) parser.add_argument("--hidden", type=int, default=0) parser.add_argument("--batch-size", type=int, default=32) parser.add_argument( @@ -311,6 +353,9 @@ def main(): hidden=args.hidden, batch_size=args.batch_size, cache_workers=args.cache_workers, + model_profile=args.model_profile, + optimizer=args.optimizer, + learning_rate=args.learning_rate, ) except Exception as error: args.output.mkdir(parents=True, exist_ok=True) @@ -329,6 +374,9 @@ def main(): "hidden": args.hidden, "batch_size": args.batch_size, "cache_workers": args.cache_workers, + "model_profile": args.model_profile, + "optimizer": args.optimizer, + "learning_rate": args.learning_rate, "error": {"type": type(error).__name__, "message": str(error)}, } (args.output / "report.json").write_text(json.dumps(failure, indent=2) + "\n") diff --git a/dev/benchmarks/summarize.py b/dev/benchmarks/summarize.py index 8a4d3ea..16fef0f 100644 --- a/dev/benchmarks/summarize.py +++ b/dev/benchmarks/summarize.py @@ -26,7 +26,13 @@ def summarize(directory: Path) -> str: ) parameter_bytes = report.get("parameter_bytes", "—") peak = report.get("peak_cuda_allocated_bytes") - peak_memory = f"{peak / 2**20:.2f}" if peak is not None else "—" + peak_memory = ( + f"{peak / 2**20:.2f}" + if peak is not None and report.get("peak_cuda_memory_scope") + else "unverified" + if peak is not None + else "—" + ) lines.append( f"| {name} | {report['status']} | {device} | {quantization} | {accuracy} | {parameter_bytes} | {peak_memory} | {duration} |" ) @@ -42,7 +48,8 @@ def summarize(directory: Path) -> str: "only with matching hardware, dataset/configuration and timing scope. See JSON reports", "for provenance, errors and explicit coverage flags. CPU results do not validate GPU behavior.", "QT coverage counts quantized Linear modules; other operations may remain floating point.", - "Parameter bytes describe stored parameters, while peak CUDA memory covers the full run.", + "Parameter bytes describe stored parameters. CUDA peaks cover training, excluding final held-out inference.", + "Older CUDA readings without a scope marker are unverified because logger resets could hide earlier peaks.", ] ) return "\n".join(lines) + "\n" diff --git a/dev/check-benchmarks.sh b/dev/check-benchmarks.sh index e7adc31..9f5892a 100644 --- a/dev/check-benchmarks.sh +++ b/dev/check-benchmarks.sh @@ -9,7 +9,7 @@ if [[ -e "$results" ]]; then echo 'Results directory must be new.' >&2 exit 2 fi -case "$mode" in cpu|gpu|real|qt|qt-real) ;; *) echo 'Mode must be cpu, gpu, real, qt or qt-real.' >&2; exit 2 ;; esac +case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real or qt-dense.' >&2; exit 2 ;; esac mkdir -p -- "$results" status=0 run_profile() { @@ -46,6 +46,15 @@ elif [[ "$mode" == gpu ]]; then elif [[ "$mode" == qt ]]; then run_profile synthetic-float --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 run_profile synthetic-int8 --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --quantized-training +elif [[ "$mode" == qt-dense ]]; then + : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/}" + for precision in float int8; do + quantization=() + if [[ "$precision" == int8 ]]; then quantization=(--quantized-training); fi + run_profile "mnist-dense-$precision" --dataset mnist --data-root "$BENCHMARK_DATA_ROOT/mnist" \ + --model-profile dense --optimizer sgd --learning-rate 0.3 --epochs 15 --batch-size 128 --compile \ + --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --allow-nondeterministic "${quantization[@]}" + done elif [[ "$mode" == qt-real ]]; then : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/ and blair/}" : "${BLAIR_CLASS_SPEC:?Set BLAIR_CLASS_SPEC to a reviewed Blair class specification}" diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 35ffd6c..94b02e1 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -88,7 +88,7 @@ Local observations on the RTX 3080 Ti Laptop GPU, Python 3.13.7, PyTorch zero loader/cache workers and FP16 AMP with float32 optimizer parameters. Synthetic uses 12 epochs; MNIST and Blair use 5. No augmentation or EMA is used. -| Dataset / path | Held-out accuracy | Parameter bytes | Peak CUDA MiB | Training wall seconds | +| Dataset / path | Held-out accuracy | Parameter bytes | Legacy CUDA reading MiB | Training wall seconds | | --- | --- | ---: | ---: | ---: | | Synthetic float | 100% | 88 | 64.04 | 4.22 | | Synthetic INT8, repeat | 100% | 68 | 32.03 | 12.95 | @@ -109,10 +109,12 @@ sequence used by dropout. MNIST quantizes only its final Linear. Blair uses a 64-unit hidden layer in both paths and quantizes that layer; convolutions and normalized hierarchical heads remain floating point. Parameter bytes exclude buffers, gradients, optimizer -state and activations. Whole-run peak allocation includes workspaces: the tiny -synthetic model's roughly 32 MiB difference cannot be explained by its 20-byte -parameter reduction. Blair's parameter reduction barely changes overall peak -allocation. None of these dataset runs demonstrates a training speedup. +state and activations. The legacy CUDA readings in these tables were captured after the logger reset +its peak counters. They do **not** establish whole-run peak allocation or its +reduction, and should not be used for that comparison. The corrected benchmark +now preserves the maximum across every phase reset and the final training call, +and marks the timing/memory scope in each report. Older reports are rendered as +`unverified` in summaries. Parameter storage and accuracy measurements are unaffected. None of these dataset runs demonstrates a training speedup. The initial Blair QT attempt aborted in Tkinter cleanup before reporting; the headless fix allowed the successful rerun above. Reports now preserve skipped @@ -124,7 +126,7 @@ See [the reproduction commands](../dev/benchmarks/README.md#integrated-qt-datase With row-wise weight normalization supported, a further matched Blair pair uses no hidden layer and quantizes the normalized classifier direction directly: -| Blair path, hidden size 0 | Held-out accuracy | Parameter bytes | Peak CUDA MiB | Training wall seconds | +| Blair path, hidden size 0 | Held-out accuracy | Parameter bytes | Legacy CUDA reading MiB | Training wall seconds | | --- | --- | ---: | ---: | ---: | | Float | 64.25% species / 80.45% parent | 75,848 | 64.34 | 11.65 | | INT8 normalized direction | 62.62% species / 76.14% parent | 37,548 | 32.25 | 25.32 | @@ -135,3 +137,41 @@ Convolutions still remain floating point. This establishes real hierarchical training and restored-checkpoint inference with normalized integer weights, with roughly half the parameter storage. Accuracy is lower in this single run and QT is slower; neither convergence parity nor a throughput improvement is established. + +## Dense MNIST profile with corrected peak measurements + +The dense spatial MLP profile quantizes all four Linear weights, including the +backbone. Both paths use 15 epochs, batch size 128, seed 42, SGD momentum 0.9, +head/backbone LR 0.3/0.1, zero weight decay, FP16 AMP, model compilation and a CPU +cache. The 4,000/1,000/5,000 train/validation/test split is unchanged. Optimizer +updates are eager. This is a compute-heavy classification profile, not a proposed +MNIST model-quality baseline or a claim about CNN performance. + +| Path | Test accuracy | Parameter bytes | Whole-training peak CUDA MiB | Total training-call seconds | Median train-loop seconds, epochs 2–15 | +| --- | ---: | ---: | ---: | ---: | ---: | +| Float | 93.16% | 52,944,936 | 236.30 | 12.49 | 0.191 | +| INT8 | 93.36% | 13,291,600 | 373.30 | 99.98 | 0.355 | + +A second INT8 run reproduced the accuracy and peak, with total wall time reduced +to 58.77 seconds after compiler caches were populated. Startup costs and cache +conditions prevent interpreting the total wall-time ratio as a steady-state +speed ratio. Batch-loop times include loading, preprocessing, compute and batch +logging; they exclude figures and checkpoint writes. CUDA measurements now +preserve maxima across **every batch/phase/finish reset**, rather than reading +only the counter remaining after training. Early experimental per-phase memory +fields without `phase_peak_memory_scope` are likewise unverified; the corrected +logger tracks each phase across its batch resets. + +This is a negative performance result: physical parameter storage falls by about +75%, but total peak memory increases and the timed training loop is slower. The +small accuracy difference is one seed, not evidence of a quality improvement. +The standalone fast kernel probe compiled optimizer updates as well as the +model; this trainer profile compiles only the model. Fusing optimizer updates +while preserving overflow detection, scheduler advancement and resume semantics +is therefore a concrete next investigation, together with temporary-allocation +profiling. Kernel-probe speedups do not establish completion of the QT goal. + +Use [`qt-dense`](../dev/benchmarks/README.md#dense-real-data-qt-comparison) to run +this pair and retain the full reports, checkpoints and predictions in the shared +pipeline. The optional GPU Actions job includes it when QT and real-data profiles +are enabled; no self-hosted job was dispatched from this session. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index c476eb3..df4a65c 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -110,3 +110,11 @@ checkpoint restoration, eager/compiled CUDA training and masked inference. Initial zero direction rows are rejected before preparation mutates any weights; normalization is undefined for these rows. The mathematical contract follows [PyTorch weight normalization](https://docs.pytorch.org/docs/2.12/generated/torch.nn.utils.parametrizations.weight_norm.html). + +The [dense MNIST comparison](benchmarks.md#dense-mnist-profile-with-corrected-peak-measurements) +now exercises a quantized backbone through the compiled trainer. Its accuracy is +similar for one seed, but QT is slower and uses a higher whole-training peak than +the floating path despite substantially smaller parameter storage. Corrected +benchmark logging preserves CUDA peaks across all resets; older dataset-run CUDA +readings do not establish whole-run memory reductions. Compiled training now saves +unwrapped model keys so ordinary checkpoint restoration and resume remain valid. diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index 21c576e..b39b67c 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -98,3 +98,59 @@ def test_shared_harness_records_process_failures(tmp_path): assert report["quantized_training"] == profile.endswith("int8") assert "test_accuracy" not in report assert "requested" in (output / "summary.md").read_text() + + +def test_benchmark_retains_cuda_peaks_across_phase_resets(tmp_path): + import os + + import torch + from torch.utils.data import DataLoader + + from dev.benchmarks.performance import BenchmarkLogger + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to validate cross-phase CUDA peaks") + assert torch.cuda.is_available() + torch.cuda.set_device(0) + loader = DataLoader(torch.arange(4), batch_size=2) + logger = BenchmarkLogger( + train_loader=loader, + val_loader=loader, + epochs=1, + output=str(tmp_path), + name="measure", + logger_cls=[], + measurement_device="cuda:0", + ) + large = torch.empty(8 * 1024 * 1024, device="cuda:0") + earlier_peak = torch.cuda.max_memory_allocated() + logger.update(epoch=0, type="train") + del large + logger.start_timing() + later = torch.empty(4 * 1024 * 1024, device="cuda:0") + phase_peak = torch.cuda.max_memory_allocated() + logger.step() + del later + logger.step() + logger.stop_timing() + logger.update(epoch=0, type="eval") + logger.start_timing() + logger.stop_timing() + logger.finish() + report = json.loads((tmp_path / "measure/logs/performance.json").read_text()) + assert report["peak_cuda_allocated_bytes"] >= earlier_peak + assert report["peak_cuda_allocated_bytes"] > torch.cuda.max_memory_allocated() + assert [phase["phase"] for phase in report["phases"]] == ["train", "eval"] + assert report["phases"][0]["peak_cuda_allocated_bytes"] >= phase_peak + assert all(phase["seconds"] >= 0 for phase in report["phases"]) + + +def test_summary_marks_legacy_cuda_readings_unverified(tmp_path): + path = tmp_path / "legacy" + path.mkdir() + report = {"status": "completed", "peak_cuda_allocated_bytes": 64 * 2**20} + (path / "report.json").write_text(json.dumps(report)) + assert "unverified" in summarize(tmp_path).splitlines()[4] + report["peak_cuda_memory_scope"] = "maximum across logger resets" + (path / "report.json").write_text(json.dumps(report)) + assert "64.00" in summarize(tmp_path).splitlines()[4] diff --git a/tests/test_benchmark_synthetic.py b/tests/test_benchmark_synthetic.py index 950bd2b..dc522bc 100644 --- a/tests/test_benchmark_synthetic.py +++ b/tests/test_benchmark_synthetic.py @@ -32,6 +32,10 @@ def test_synthetic_training_matches_oracle_and_repeats(tmp_path): finally: torch.set_num_threads(threads) assert first["cache"] == second["cache"] == "CPU" + assert len(first["phase_measurements"]) == 24 + assert [phase["phase"] for phase in first["phase_measurements"]] == ["train", "eval"] * 12 + assert all(phase["seconds"] >= 0 and phase["peak_cuda_allocated_bytes"] is None for phase in first["phase_measurements"]) + assert first["peak_cuda_allocated_bytes"] is None assert first["test_accuracy"] == second["test_accuracy"] == 1.0 assert first["dataset_manifest_sha256"] == second["dataset_manifest_sha256"] with np.load(tmp_path / "first/predictions.npz") as a, np.load(tmp_path / "second/predictions.npz") as b: From e136ac61d91fa5ab198667d7b40dcdf21abebe6a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 00:19:21 +0200 Subject: [PATCH 017/155] feat: add opt-in optimizer compilation with portable resume state Keep AMP update gating and Muon counters outside compiled child updates. Initialize lazy state on the first real call, use tensor learning rates for scheduler updates, and save scalar rates for eager resume. Preserve Muon's explicit BF16 casts. Expose --compile-optimizer independently of model compilation and record it in benchmark reports. Validate with static checks, the full suite (261 passed, 76 skipped, 1 expected EMA failure), and focused CUDA checks (81 passed). INT8 storage-kernel experiments remain separate. --- dev/README.md | 27 +++++++ dev/benchmarks/run.py | 6 ++ mini_trainer/train.py | 13 ++++ mini_trainer/training/compilation.py | 75 ++++++++++++++++++++ tests/test_checkpoint_contract.py | 3 +- tests/test_optimizer_steps.py | 101 ++++++++++++++++++++++++++- 6 files changed, 223 insertions(+), 2 deletions(-) create mode 100644 mini_trainer/training/compilation.py diff --git a/dev/README.md b/dev/README.md index 1530a26..ca4fe53 100644 --- a/dev/README.md +++ b/dev/README.md @@ -192,3 +192,30 @@ Raw `LazyDataset` users can request `pin_batches=True` for the same behavior. Worker processes always use the ordinary gather path and DataLoader's parent-side pinning; this option never initializes the CUDA pin allocator in a worker. CPU training, CUDA-cached datasets and scalar indexing keep their existing behavior. + +### Optimizer compilation + +`mt_train --compile-optimizer` opts into compiling optimizer updates independently +of model `--compile`. The benchmark runner accepts the same option and records it +in result and failure reports. It defaults off; compilation overhead and graph +breaks can outweigh any steady-state benefit, so measure the intended workload. + +The first real optimizer call initializes lazy state eagerly. Subsequent calls +use compilation with tensor learning rates, allowing scheduler changes without +specializing a graph for every numeric rate. AMP overflow handling and scheduler +advancement remain controlled by the trainer. MuonAuxAdamW compiles its child +optimizers while keeping its outer step counter in Python; Muon's compilation +preserves the explicit BF16 casts in its Newton-Schulz iterations. + +For custom training, call `mini_trainer.training.compilation.compile_optimizer` +after constructing the scheduler and restoring checkpoint state. Saved learning +rates remain ordinary scalars, so a checkpoint can resume without compilation. +Explicit `foreach=True` Adam/AdamW requires `capturable=True`; unsupported +combinations fail before the helper changes the optimizer. Default and explicit +`foreach=False` updates do not need that setting. + +Regression coverage compares eager and compiled SGD, AdamW, their native fused +variants, and MuonAuxAdamW on CUDA, including overflow skips, scheduler changes, +parameter updates and optimizer state. This does not establish support for every +custom optimizer or a real-workload INT8 speedup. Quantized update dispatch and +its performance remain a separate validation boundary. diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 6662ac1..087e1db 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -54,6 +54,7 @@ def run( cuda_prefetch: bool = False, quantized_training: bool = False, compile: bool = False, + compile_optimizer: bool = False, hidden: int = 0, batch_size: int = 32, cache_workers: int | None = None, @@ -148,6 +149,7 @@ def run( ema=False, quantized_training=quantized_training, compile=compile, + compile_optimizer=compile_optimizer, model_builder_kwargs={ "model_type": model_type, "hidden": hidden if hidden else False, @@ -243,6 +245,7 @@ def run( "momentum": 0.9 if optimizer == "sgd" else None, "quantization_recipe": quantization_recipe, "compile": compile, + "compile_optimizer": compile_optimizer, "hidden": hidden, "batch_size": batch_size, "cache_workers": cache_workers, @@ -317,6 +320,7 @@ def main(): parser.add_argument("--quantized-training", action="store_true") parser.add_argument("--cuda-prefetch", action="store_true") parser.add_argument("--compile", action="store_true") + parser.add_argument("--compile-optimizer", action="store_true") parser.add_argument("--model-profile", choices=["default", "dense"], default="default") parser.add_argument("--optimizer", choices=["muon", "adamw", "sgd"], default="muon") parser.add_argument("--learning-rate", type=float) @@ -350,6 +354,7 @@ def main(): cuda_prefetch=args.cuda_prefetch, quantized_training=args.quantized_training, compile=args.compile, + compile_optimizer=args.compile_optimizer, hidden=args.hidden, batch_size=args.batch_size, cache_workers=args.cache_workers, @@ -371,6 +376,7 @@ def main(): "cuda_prefetch": args.cuda_prefetch, "quantized_training": args.quantized_training, "compile": args.compile, + "compile_optimizer": args.compile_optimizer, "hidden": args.hidden, "batch_size": args.batch_size, "cache_workers": args.cache_workers, diff --git a/mini_trainer/train.py b/mini_trainer/train.py index 9fffe3e..68458c2 100644 --- a/mini_trainer/train.py +++ b/mini_trainer/train.py @@ -75,6 +75,7 @@ def main( # noqa: D417 lr_schedule_builder_kwargs: dict[str, Any] = {"warmup_epochs": 2.0}, logger_builder_kwargs: dict[str, Any] = {"verbose": False}, ddp_info: dict | None = None, + compile_optimizer: bool = False, ): """Train a classifier. @@ -290,6 +291,12 @@ def main( # noqa: D417 start_epoch = start_epoch + 1 log.info(f"Training restarted from checkpoint(s): {checkpoint}") + if compile_optimizer: + from mini_trainer.training.compilation import compile_optimizer as prepare_compiled_optimizer + + prepare_compiled_optimizer(optimizer) + log.info("Optimizer updates compiled; scheduler and AMP step gating remain active.") + # Instantiate logger logger_output = None if get_rank() > 0 else output logger = builder.build_logger( @@ -531,6 +538,12 @@ def cli(description="Train a classifier", **extra_kwargs): # noqa: D103 ) cfg_args = parser.add_argument_group("Runtime [optional]") + cfg_args.add_argument( + "--compile-optimizer", + action="store_true", + dest="compile_optimizer", + help="Compile optimizer updates with tensor learning rates; model compilation is controlled separately.", + ) cfg_args.add_argument( "--cuda-prefetch", action="store_true", diff --git a/mini_trainer/training/compilation.py b/mini_trainer/training/compilation.py new file mode 100644 index 0000000..ec5471e --- /dev/null +++ b/mini_trainer/training/compilation.py @@ -0,0 +1,75 @@ +"""Opt-in optimizer compilation with stable learning-rate inputs.""" + +from functools import wraps + +import torch + +from .muon import Muon, MuonAuxAdamW + + +def _tensor_learning_rates(optimizer): + for group in optimizer.param_groups: + if not isinstance(group["lr"], torch.Tensor): + # Python floats are double precision. Keep that value unchanged, + # while letting compilation treat scheduler updates as inputs. + group["lr"] = torch.tensor(group["lr"], dtype=torch.float64) + + +def _portable_learning_rates(optimizer, state): + for group in state["param_groups"]: + if isinstance(group["lr"], torch.Tensor): + group["lr"] = group["lr"].item() + return state + + +def _compile_after_initial_call(optimizer, options): + step = optimizer.step + compiled = torch.compile(step, **options) + initialized = False + + @wraps(step) + def update(*args, **kwargs): + nonlocal initialized + if not initialized: + # Initialize lazy momentum/state during a real call. Tracing that + # mutation across multiple groups can fail in Dynamo; never insert + # a fake update or bypass GradScaler to initialize it. + result = step(*args, **kwargs) + _tensor_learning_rates(optimizer) + optimizer.register_load_state_dict_post_hook(_tensor_learning_rates) + initialized = True + return result + return compiled(*args, **kwargs) + + return update + + +def compile_optimizer(optimizer, *, backend=None): + """Compile updates after scheduler construction and checkpoint restoration. + + Keep hooks, scaler overflow decisions and scheduler calls in their usual + order. Composite Muon counters stay outside compiled child updates. Numeric + learning rates become scalar tensor inputs, but checkpoint groups retain + scalar rates so ordinary eager resume does not require this option. + """ + targets = [getattr(optimizer, name) for name in optimizer.optimizers] if isinstance(optimizer, MuonAuxAdamW) else [optimizer] + for target in targets: + if isinstance(target, (torch.optim.Adam, torch.optim.AdamW)) and any( + group.get("foreach") and not group.get("capturable") for group in target.param_groups + ): + raise ValueError( + "Optimizer compilation with explicit foreach Adam/AdamW requires capturable=True; " + "use foreach=False (or its default) for non-capturable updates." + ) + for target in targets: + if getattr(target, "_mini_trainer_compiled", False): + continue + target.register_state_dict_post_hook(_portable_learning_rates) + options = {} if backend is None else {"backend": backend} + if backend is None and isinstance(target, Muon): + # Newton-Schulz intentionally rounds intermediate values to BF16. + # Preserve those casts instead of silently changing its iteration. + options["options"] = {"emulate_precision_casts": True} + target.step = _compile_after_initial_call(target, options) + target._mini_trainer_compiled = True + return optimizer diff --git a/tests/test_checkpoint_contract.py b/tests/test_checkpoint_contract.py index a0348ed..e344049 100644 --- a/tests/test_checkpoint_contract.py +++ b/tests/test_checkpoint_contract.py @@ -207,6 +207,7 @@ def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatc "dtype": "float32", "seed": 42, "compile": True, + "compile_optimizer": True, "ema": False, "builder": DeterministicBuilder, "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": False}, @@ -220,7 +221,7 @@ def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatc assert not any(key.startswith("_orig_mod.") for key in state["model"]) eager = torch.load(tmp_path / "compiled/weights/last.pt", weights_only=True) assert_state_equal(state["model"], eager) - args.update(name="resumed", epochs=2, checkpoint=str(path)) + args.update(name="resumed", epochs=2, checkpoint=str(path), compile_optimizer=False) args["model_builder_kwargs"]["model_type"] = TinyMockModel() train_module.main(**args) resumed = torch.load(tmp_path / "resumed/weights/checkpoint_last.pth", weights_only=True) diff --git a/tests/test_optimizer_steps.py b/tests/test_optimizer_steps.py index ef7aef0..3e05886 100644 --- a/tests/test_optimizer_steps.py +++ b/tests/test_optimizer_steps.py @@ -35,11 +35,16 @@ def device(request): @pytest.mark.parametrize("kind", KINDS) @pytest.mark.parametrize("scaled", [False, True]) -def test_epoch_only_advances_scheduler_and_ema_after_updates(kind, scaled, device): +@pytest.mark.parametrize("compiled", [False, True]) +def test_epoch_only_advances_scheduler_and_ema_after_updates(kind, scaled, device, compiled): torch.manual_seed(42) model = torch.nn.Sequential(torch.nn.Flatten(), torch.nn.Linear(2, 2)).to(device) optimizer = make_optimizer(kind, model.parameters()) scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.5) + if compiled: + from mini_trainer.training.compilation import compile_optimizer + + compile_optimizer(optimizer, backend="eager" if device.type == "cpu" else None) scaler = torch.amp.GradScaler(device.type, enabled=scaled, init_scale=8, growth_interval=1) images = torch.tensor([[[[1.0, 2.0]]]]).repeat(3, 1, 1, 1) loader = DataLoader(TensorDataset(images, torch.tensor([0, 1, 0])), batch_size=1) @@ -147,3 +152,97 @@ def test_overflow_detected_even_when_scale_cannot_back_off_further(kind): assert scaler.get_scale() == 0.0 torch.testing.assert_close(parameter, torch.ones_like(parameter), rtol=0, atol=0) assert not optimizer._optimizer_step_post_hooks + + +def test_compiled_optimizer_scheduler_does_not_recompile_every_step(): + from torch._dynamo.testing import CompileCounter + + from mini_trainer.training.compilation import compile_optimizer + + torch._dynamo.reset() + parameter = torch.nn.Parameter(torch.ones(4, 4)) + optimizer = torch.optim.SGD([parameter], lr=0.1, momentum=0.9) + scheduler = torch.optim.lr_scheduler.LambdaLR(optimizer, lambda step: 0.93**step) + counter = CompileCounter() + compile_optimizer(optimizer, backend=counter) + scaler = torch.amp.GradScaler("cpu", enabled=False) + frames_after_warmup = None + for step in range(12): + optimizer.zero_grad() + parameter.square().sum().backward() + assert _optimizer_step(optimizer, scaler) + scheduler.step() + if step == 2: + frames_after_warmup = counter.frame_count + assert frames_after_warmup and counter.frame_count == frames_after_warmup + state = optimizer.state_dict() + assert isinstance(state["param_groups"][0]["lr"], float) + assert isinstance(optimizer.param_groups[0]["lr"], torch.Tensor) + optimizer.load_state_dict(state) + assert isinstance(optimizer.param_groups[0]["lr"], torch.Tensor) + fresh = torch.optim.SGD([torch.nn.Parameter(parameter.detach().clone())], lr=1.0, momentum=0.9) + fresh.load_state_dict(state) + assert isinstance(fresh.param_groups[0]["lr"], float) + + +def test_compiled_foreach_adamw_requires_capturable_before_mutation(): + from mini_trainer.training.compilation import compile_optimizer + + parameter = torch.nn.Parameter(torch.ones(4, 4)) + optimizer = torch.optim.AdamW([parameter], lr=0.01, foreach=True) + before = optimizer.step + with pytest.raises(ValueError, match="requires capturable=True"): + compile_optimizer(optimizer, backend="eager") + assert optimizer.step == before + assert isinstance(optimizer.param_groups[0]["lr"], float) + assert not optimizer.state + + +def _assert_compiled_optimizer_state(actual, expected, key=None): + if isinstance(actual, torch.Tensor): + # Dynamo moves Adam's scalar step counter to CUDA. Check its exact + # numeric state; all non-counter tensor devices must remain unchanged. + if key == "step" and actual.numel() == expected.numel() == 1: + expected = expected.to(actual.device) + torch.testing.assert_close(actual, expected, rtol=1e-5, atol=2e-6) + elif isinstance(actual, dict): + assert actual.keys() == expected.keys() + for name in actual: + _assert_compiled_optimizer_state(actual[name], expected[name], name) + elif isinstance(actual, (list, tuple)): + assert type(actual) is type(expected) and len(actual) == len(expected) + for left, right in zip(actual, expected, strict=True): + _assert_compiled_optimizer_state(left, right) + else: + assert actual == expected + + +@pytest.mark.parametrize("kind", KINDS) +def test_cuda_compiled_optimizer_matches_eager_updates(kind): + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to compare compiled CUDA updates") + assert torch.cuda.is_available() + torch.manual_seed(71) + initial = torch.randn(8, 8, device="cuda") + parameters = [torch.nn.Parameter(initial.clone()) for _ in range(2)] + optimizers = [make_optimizer(kind, [parameter]) for parameter in parameters] + schedulers = [torch.optim.lr_scheduler.StepLR(opt, step_size=1, gamma=0.93) for opt in optimizers] + compile_optimizer(optimizers[1]) + scalers = [torch.amp.GradScaler("cuda", init_scale=8, growth_interval=100) for _ in optimizers] + for step in range(6): + gradient = torch.randn_like(initial) + decisions = [] + for parameter, optimizer, scheduler, scaler in zip(parameters, optimizers, schedulers, scalers, strict=True): + optimizer.zero_grad() + scaler.scale((parameter * gradient).sum()).backward() + if step == 2: + parameter.grad.fill_(float("inf")) + updated = _optimizer_step(optimizer, scaler) + decisions.append(updated) + if updated: + scheduler.step() + assert decisions[0] == decisions[1] == (step != 2) + torch.testing.assert_close(parameters[0], parameters[1], rtol=1e-5, atol=2e-6) + _assert_compiled_optimizer_state(optimizers[0].state_dict(), optimizers[1].state_dict()) From 6d5e6695398deed2d622ec051aeafb53337c499f Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 00:29:38 +0200 Subject: [PATCH 018/155] docs: record fused INT8 update measurements and regression limits Record dense MNIST quality, timing and whole-run versus later-phase allocation. Trace the 256 MiB Triton tuning buffer and document fake-tensor and compiler-cache failures that prevent shipping the experimental storage kernel. --- docs/benchmarks.md | 44 ++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 94b02e1..153ad67 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -175,3 +175,47 @@ Use [`qt-dense`](../dev/benchmarks/README.md#dense-real-data-qt-comparison) to r this pair and retain the full reports, checkpoints and predictions in the shared pipeline. The optional GPU Actions job includes it when QT and real-data profiles are enabled; no self-hosted job was dispatched from this session. + +### Fused INT8 storage updates (2026-09-09) + +The same dense MNIST profile was exercised with experimental fused CUDA storage updates. +This implementation has not passed the full CUDA regression set and is not a +delivered feature; the results below guide the next implementation attempt. +These are individual runs on the same RTX 3080 Ti, seed 42, data split, SGD +settings and 15-epoch budget used above. All compile the model. The floating +comparison also compiles the optimizer; INT8 is shown both ways to expose the +effect of compiling the outer update wrapper. Later-epoch figures below cover +epochs 3–15, after first-use compilation and optimizer initialization. + +| Execution | Test accuracy | Whole-run peak MiB | Later training peak MiB | Median later training epoch s | Training wall s | +| --- | ---: | ---: | ---: | ---: | ---: | +| Float, compiled optimizer | 92.44% | 236.30 | 236.30 | 0.185 | 14.91 | +| INT8 fused storage, eager optimizer | 93.32% | 373.30 | 164.58 | 0.273 | 44.26 | +| INT8 fused storage, compiled optimizer | 93.12% | 423.79 | 164.35 | 0.271 | 47.01 | + +Fusing the storage update lowers later-phase allocation and runtime relative to +the earlier INT8 update, but does **not** establish a whole-run memory or speed +win over float. Accuracy differences remain single-seed observations. Wall times +include compilation/autotuning and depend on cache state; they are not controlled +cold-cache comparisons. The corresponding report source hashes begin +`8fce9ea8fe3c`, `4f88ebf5632d`, and `a4e805ebe5c4a` respectively. + +A separate two-epoch CUDA allocation trace identified a 268,435,456-byte buffer +allocated by Triton's `get_empty_cache_for_benchmark` during TorchAO INT8 matrix +kernel autotuning. This occurs during first-use training and evaluation, and +explains why steady-state storage savings do not translate to a lower whole-run +peak. Reducing this tuning overhead remains necessary; excluding it from the +reported peak would conceal a real allocation that users must accommodate. + +The four-weight dense run emitted no recompilation-limit fallback, but a separate +twelve-group SGD stress test did: outer optimizer compilation specialized on +`TrainingWeight` object identities. Keep that option off for general QT workloads +until the wrapper path is fixed. Seven isolated storage tests passed rounding +bounds, CUDA RNG replay and saved-tensor invalidation checks. However, the combined +CUDA suite finished with 40 passing and three failing tests: fake-tensor execution +reached the real storage kernel during normalized Adam updates, and the storage +compiler exhausted its eight-variant cache across dtype/operation combinations, +also preventing the twelve-parameter reuse regression from running successfully. +Raising that global limit would conceal the underlying dispatch/cache design +problem. The next implementation needs a storage kernel that handles these +variants without depending on per-frame Dynamo specialization. From 2cbe56d574e0a6d7831c369b7ee3cd96d6c8ee5e Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 00:40:56 +0200 Subject: [PATCH 019/155] perf: fuse INT8 storage updates and reduce matrix tuning memory Use explicit Triton row updates with a fake-tensor-safe custom operator, preserving integer storage, stochastic rounding, and tensor version invalidation. Tune TorchAO matrix kernels through a local CUDA-graph tuner without its 256 MiB cache-flush buffer. The paired dense MNIST run reduces whole-training peak allocation from 236.30 to 154.00 MiB. Record that runtime and quality parity remain unproven. Keep the many-group outer optimizer fallback as a strict expected-failure regression. Validation: static checks; full suite 261 passed, 85 skipped, 1 expected EMA failure; CUDA 44 passed and 1 expected compiler limitation; minimal installed-wheel smoke passed. --- docs/benchmarks.md | 38 +++++ docs/quantized-training.md | 28 +++ docs/roadmap.md | 7 + mini_trainer/modeling/_quantized_matmul.py | 53 ++++++ mini_trainer/modeling/_quantized_training.py | 70 ++++++-- mini_trainer/modeling/_quantized_update.py | 82 +++++++++ tests/test_quantized_training_model.py | 170 ++++++++++++++++++- 7 files changed, 437 insertions(+), 11 deletions(-) create mode 100644 mini_trainer/modeling/_quantized_matmul.py create mode 100644 mini_trainer/modeling/_quantized_update.py diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 153ad67..76cf59d 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -219,3 +219,41 @@ also preventing the twelve-parameter reuse regression from running successfully. Raising that global limit would conceal the underlying dispatch/cache design problem. The next implementation needs a storage kernel that handles these variants without depending on per-frame Dynamo specialization. + +### Explicit CUDA kernels and local tuning + +The replacement uses a row-wise Triton update behind a custom operator with fake +execution support. It avoids the failed prototype's per-frame compilation cache. +INT8 matrix multiplication retains TorchAO's kernel/configurations, but a separate +local tuner measures kernels with CUDA graphs and caches selected configurations +on disk. This avoids the 256 MiB cache-flushing buffer without changing global +TorchAO or Triton behavior. + +The dense MNIST pair was rerun with the same dataset, seed 42, 15 epochs, batch +128, model compilation and eager optimizers. Both report source hash +`0175e5a8d8e8998e98e57288478b4adc5fd5645e93f0855ab3ef1b9bef55bd2f` +and the same dataset manifest. These are single runs, not statistical estimates. + +| Execution | Test accuracy | Whole-run peak MiB | Median training epoch s, epochs 3–15 | Training wall s | +| --- | ---: | ---: | ---: | ---: | +| Float | 93.16% | 236.30 | 0.237 | 14.13 | +| INT8, explicit kernels | 92.42% | 154.00 | 0.242 | 24.74 | + +This demonstrates approximately 35% lower **whole-run** peak allocation for the +INT8 profile, including first-use tuning. It does not establish a training speed +win: later-epoch times are similar and total wall time remains higher. First-use +compilation and cache state affect the wall comparison, and one seed cannot +establish equivalent model quality. The original QT speed objective remains open. + +The combined CUDA kernel/model regressions now pass the previously failing +normalized checkpoint and dtype/operation cases. Coverage includes stochastic +rounding bounds, CUDA RNG replay, saved-tensor invalidation, and reuse across +twelve distinct weights with changing learning rates. A forced-cold matrix tuning +test compares against integer reference arithmetic and requires less than 16 MiB +temporary allocation for a tiny product, so a cached tuning result cannot mask a +return of the old 256 MiB allocation. + +Outer optimizer compilation still falls back when sufficiently many quantized +groups specialize on wrapper identities. A separate strict expected-failure test +enables hard failure on that fallback; it must be removed when the outer path is +fixed. This limitation is distinct from the now-working compiled storage operator. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index df4a65c..fd45857 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -53,6 +53,34 @@ the developer probe. Eager SGD, AdamW and the repository's MuonAuxAdamW update paths are covered; fused optimizer variants are not established. Quantized regularization uses a differentiable floating view of the represented weights. +CUDA SGD/AdamW-style `add_` and `addcdiv_` updates fuse dequantization, +the weight update and stochastic requantization over the underlying storage +tensors. This kernel compiles on first use even when the outer optimizer is +eager. It retains INT8 codes and row scales, advances tensor version counters, +and preserves explicit intermediate precision casts. Scalar tensor weight decay +rescales rows without another stochastic rounding pass. CPU inspection/update +tests use the ordinary floating calculation and copy path. The fused row kernel +covers matching FP32/FP16/BF16 update tensors and rows up to 16,384 elements; +broadcasting, mixed dtypes and wider rows retain the ordinary update path. +No floating master weight is retained by either path. The new kernel uses CUDA +RNG seeds with Triton stochastic rounding, so exact trajectories differ from the +earlier floating update even when starting from the same seed. + +Matrix products reuse TorchAO's INT8 kernel with a separate local tuner. CUDA +graph timing avoids the default tuner's 256 MiB cache-flushing allocation, and +selected configurations are cached on disk. TorchAO's global operators and tuner +are unchanged. The explicit update and matrix operators provide fake execution +implementations for model compilation. + +Keep `--compile-optimizer` off for general QT workloads for now. Although small +integrated cases pass, compiling an outer optimizer with many quantized parameter +groups can specialize on weight identities and fall back to eager execution. +That outer-optimizer limitation remains a strict CUDA expected-failure regression. +The earlier storage prototype's fake-tensor and dtype-cache failures are resolved +by an explicit Triton kernel and custom-operator boundary; the storage kernel +does not depend on Dynamo's per-frame variant cache. Model `--compile` remains a +separate option. See the [measured results](benchmarks.md#explicit-cuda-kernels-and-local-tuning). + ## Checkpoints and inference Prepared models include their recipe in `state_dict`. Ordinary `mt_train` diff --git a/docs/roadmap.md b/docs/roadmap.md index 87c1a90..de48600 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -102,6 +102,13 @@ Loader hardening, float16/bfloat16 AMP and benchmark infrastructure do not compl that target. The implementation and comparison plan is in [training feature validation](training-feature-validation.md). +The explicit CUDA storage-update kernel and local matrix tuner now reduce +whole-run peak allocation in the dense MNIST comparison by approximately 35%; +[training speed remains unproven](benchmarks.md#explicit-cuda-kernels-and-local-tuning). +A strict CUDA expected-failure regression records outer optimizer compilation +falling back after specializing on many quantized parameter-group identities. +Keep that limit visible while improving optimizer dispatch and throughput. + Deliver quantization-aware training and post-training inference quantization as separate opt-in capabilities, recording actual weight/activation bit widths, calibration data, backend kernels, checkpoint/resume and export/runtime support. diff --git a/mini_trainer/modeling/_quantized_matmul.py b/mini_trainer/modeling/_quantized_matmul.py new file mode 100644 index 0000000..ada2271 --- /dev/null +++ b/mini_trainer/modeling/_quantized_matmul.py @@ -0,0 +1,53 @@ +"""Local INT8 matrix kernel tuning without Triton's large L2-flush allocation.""" + +import torch +import triton +from torchao.prototype.quantized_training.int8_mm import _scaled_int8_mm_kernel as _upstream_kernel +from triton.testing import do_bench_cudagraph + + +def _benchmark(kernel, quantiles): + return do_bench_cudagraph(kernel, rep=5, quantiles=quantiles) + + +# Reuse TorchAO's kernel and candidate configurations, but create a separate +# tuner. Never alter TorchAO's global operator, tuner, or Triton benchmark hooks. +# CUDA graph timing measures repeated device execution without allocating the +# default benchmark's 256 MiB cache-flushing buffer. Persist selected configs so +# a new process need not retune every matrix shape. +_kernel = triton.autotune(configs=_upstream_kernel.configs, key=_upstream_kernel.keys, do_bench=_benchmark, cache_results=True)( + _upstream_kernel.fn +) + + +@torch.library.custom_op("mini_trainer::scaled_int8_mm", mutates_args=()) +def scaled_int8_mm(left: torch.Tensor, right: torch.Tensor, row_scale: torch.Tensor, column_scale: torch.Tensor) -> torch.Tensor: + """Compute a scaled INT8 matrix product using locally tuned CUDA kernels.""" + rows, contraction = left.shape + columns = right.shape[1] + output = torch.empty((rows, columns), dtype=row_scale.dtype, device=left.device) + + def grid(meta): + return (triton.cdiv(rows, meta["BLOCK_M"]) * triton.cdiv(columns, meta["BLOCK_N"]),) + + with torch.cuda.device(left.device): + _kernel[grid]( + left, + right, + output, + row_scale, + column_scale, + rows, + columns, + contraction, + *left.stride(), + *right.stride(), + *output.stride(), + COL_SCALE_SCALAR=column_scale.numel() == 1, + ) + return output + + +@scaled_int8_mm.register_fake +def _fake_scaled_int8_mm(left, right, row_scale, column_scale): + return torch.empty((left.shape[0], right.shape[1]), dtype=row_scale.dtype, device=left.device) diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index df9ba6c..e79190a 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -9,7 +9,9 @@ import torch from torch.utils._python_dispatch import return_and_correct_aliasing from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise -from torchao.prototype.quantized_training.int8_mm import scaled_int8_mm as _native_scaled_int8_mm + +from ._quantized_matmul import scaled_int8_mm as _native_scaled_int8_mm +from ._quantized_update import update_int8_rows_ # Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary # graph key. Include the backend implementation so changing backward math cannot @@ -77,7 +79,7 @@ def multiply_inplace(func, types, args, kwargs): original, multiplier = args # AdamW's decoupled decay is a scalar rescale. Preserve the integer codes # exactly instead of adding a second stochastic rounding to every update. - if isinstance(multiplier, (float, int)): + if isinstance(multiplier, (float, int)) or isinstance(multiplier, torch.Tensor) and multiplier.ndim == 0: original.scale.mul_(multiplier) return original return original.copy_(original.dequantize() * multiplier) @@ -147,14 +149,15 @@ def backward(ctx, grad_output): def scaled_int8_mm(left, right, row_scale, column_scale): - # TorchAO's Python validation squeezes the row scale, rejecting M=1 even - # though its native kernel supports it. Validate that case without squeeze. - if left.shape[0] == 1: - if row_scale.shape != (1,) or column_scale.shape != (right.shape[1],): - raise ValueError("Invalid INT8 matrix scales.") - if left.shape[1] != right.shape[0] or row_scale.dtype != column_scale.dtype: - raise ValueError("Incompatible INT8 matrix shapes or scale dtypes.") - return torch.ops.torchao.scaled_int8_mm(left, right, row_scale, column_scale) + # Validate row scales without squeeze, which would reject a single sample. + if row_scale.shape != (left.shape[0],) or column_scale.numel() not in (1, right.shape[1]): + raise ValueError("Invalid INT8 matrix scales.") + if left.shape[1] != right.shape[0] or row_scale.dtype != column_scale.dtype: + raise ValueError("Incompatible INT8 matrix shapes or scale dtypes.") + if left.dtype != torch.int8 or right.dtype != torch.int8: + raise ValueError("INT8 matrix products require integer inputs.") + if not row_scale.is_contiguous() or not column_scale.is_contiguous(): + raise ValueError("INT8 matrix scales must be contiguous.") return _native_scaled_int8_mm(left, right, row_scale, column_scale) @@ -210,3 +213,50 @@ def backward(ctx, gradient): projection = (gradient.float() * unit).sum(dim=1, keepdim=True) direction_gradient = (gradient.float() - projection * unit) * (magnitude.float() / (scales.float().abs() * norm).unsqueeze(1)) return direction_gradient.to(scales.dtype), projection.to(magnitude.dtype) + + +@TrainingWeight.implements_torch_function(torch.Tensor.add_) +def add_update(func, types, args, kwargs): + original = args[0] + update = args[1] if len(args) > 1 else kwargs["other"] + alpha = kwargs.get("alpha", 1) + # Tensor learning rates must stay tensor inputs. Passing them through the + # aten alpha scalar overload can specialize dispatch on Python objects. + return _apply_weight_update(original, update, alpha) + + +@TrainingWeight.implements_torch_function(torch.Tensor.addcdiv_) +def addcdiv_update(func, types, args, kwargs): + original = args[0] + numerator = args[1] if len(args) > 1 else kwargs["tensor1"] + denominator = args[2] if len(args) > 2 else kwargs["tensor2"] + value = kwargs.get("value", 1) + return _apply_weight_update(original, numerator, value, denominator) + + +def _apply_weight_update(original, update, alpha, denominator=None): + supported = ( + original.device.type == "cuda" + and original.dtype in (torch.float32, torch.float16, torch.bfloat16) + and 0 < original.shape[1] <= 16384 + and isinstance(update, torch.Tensor) + and update.shape == original.shape + and update.dtype == original.dtype + and update.device == original.device + and ( + denominator is None + or denominator.shape == original.shape + and denominator.dtype == original.dtype + and denominator.device == original.device + ) + ) + if not supported or torch.is_grad_enabled() and original.requires_grad: + change = update if denominator is None else update / denominator + return original.copy_(original.dequantize() + change * alpha) + if not isinstance(alpha, torch.Tensor): + alpha = torch.tensor(alpha, dtype=torch.float64) + update_int8_rows_(original.int_data, original.scale, update, alpha, denominator) + # The kernel mutates storage tensors directly; also advance the wrapper's + # version counter for cache invalidation and saved-tensor safety. + torch.autograd.graph.increment_version(original) + return original diff --git a/mini_trainer/modeling/_quantized_update.py b/mini_trainer/modeling/_quantized_update.py new file mode 100644 index 0000000..23cdad1 --- /dev/null +++ b/mini_trainer/modeling/_quantized_update.py @@ -0,0 +1,82 @@ +"""CUDA row-wise INT8 updates without a floating master weight matrix.""" + +import torch +import triton +import triton.language as tl + + +@triton.jit +def _update_rows( + codes, + scales, + update, + denominator, + seed, + alpha, + columns, + code_row_stride, + code_column_stride, + scale_stride, + update_row_stride, + update_column_stride, + denominator_row_stride, + denominator_column_stride, + DIVIDE: tl.constexpr, + BLOCK: tl.constexpr, +): + row = tl.program_id(0) + column = tl.arange(0, BLOCK) + valid = column < columns + code_offset = row * code_row_stride + column * code_column_stride + old_codes = tl.load(codes + code_offset, valid, 0).to(tl.float32) + old_scale = tl.load(scales + row * scale_stride) + dtype = old_scale.dtype + represented = (old_codes * old_scale.to(tl.float32)).to(dtype).to(tl.float32) + change = tl.load(update + row * update_row_stride + column * update_column_stride, valid, 0).to(tl.float32) + if DIVIDE: + divisor = tl.load(denominator + row * denominator_row_stride + column * denominator_column_stride, valid, 1).to(tl.float32) + change = (change / divisor).to(dtype).to(tl.float32) + change = (change * alpha).to(dtype).to(tl.float32) + values = (represented + change).to(dtype).to(tl.float32) + maximum = tl.max(tl.where(valid, tl.abs(values), 0), 0) + next_scale = (maximum / 127).to(dtype) + inverse = 1.0 / tl.maximum(next_scale.to(tl.float32), 1.0e-12) + random = tl.rand(tl.load(seed), row * columns + column) + next_codes = tl.floor(values * inverse + random) + next_codes = tl.minimum(tl.maximum(next_codes, -128), 127).to(tl.int8) + tl.store(codes + code_offset, next_codes, valid) + tl.store(scales + row * scale_stride, next_scale) + + +@torch.library.custom_op("mini_trainer::update_int8_rows_", mutates_args=("codes", "scales")) +def update_int8_rows_( + codes: torch.Tensor, scales: torch.Tensor, update: torch.Tensor, alpha: torch.Tensor, denominator: torch.Tensor | None +) -> None: + """Mutate represented weights; fake execution never launches a CUDA kernel.""" + # Optimizer learning rates normally live on CPU. Pass their numeric value + # as a runtime kernel argument, without a device allocation per weight. + coefficient = alpha.item() + seed = torch.randint(0, 2**31, (), device=codes.device, dtype=torch.int64) + divisor = update if denominator is None else denominator + with torch.cuda.device(codes.device): + _update_rows[(codes.shape[0],)]( + codes, + scales, + update, + divisor, + seed, + coefficient, + codes.shape[1], + *codes.stride(), + scales.stride(0), + *update.stride(), + *divisor.stride(), + DIVIDE=denominator is not None, + BLOCK=triton.next_power_of_2(codes.shape[1]), + enable_fp_fusion=False, + ) + + +@update_int8_rows_.register_fake +def _fake_update_int8_rows_(codes, scales, update, alpha, denominator): + return None diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index d419089..be3c036 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -135,7 +135,8 @@ def test_mixed_muon_adamw_updates_and_counter(): @pytest.mark.parametrize("normalized", [False, True]) -def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized): +@pytest.mark.parametrize("compiled_optimizer", [False, True]) +def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, compiled_optimizer): from mini_trainer.modeling import Classifier from mini_trainer.modeling._quantized_training import TrainingWeight from mini_trainer.train import main @@ -153,6 +154,7 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized): "device": device, "dtype": "float16", "quantized_training": True, + "compile_optimizer": compiled_optimizer, "seed": 42, "builder": DeterministicBuilder, "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": normalized}, @@ -174,6 +176,7 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized): # Resume through the ordinary training entry point. This checks state # plumbing, not identical stochastic continuation with a changed schedule. args["epochs"] = 2 + args["compile_optimizer"] = False args["name"] = "resumed" args["checkpoint"] = str(tmp_path / "quantized/weights/checkpoint_last.pth") args["model_builder_kwargs"]["model_type"] = TinyMockModel() @@ -320,3 +323,168 @@ def test_normalized_classifier_integer_training_and_masked_inference(compiled): model.set_active_features([0, 2, 3]) selected = model(inputs[:1]) torch.testing.assert_close(selected, complete[:, [0, 2, 3]]) + + +def test_tensor_scalar_decay_preserves_integer_codes_and_rng(): + from mini_trainer.modeling._quantized_training import TrainingWeight + + weight = nn.Parameter(TrainingWeight.from_float(torch.randn(5, 17))) + codes = weight.int_data.clone() + scales = weight.scale.clone() + rng = torch.get_rng_state() + with torch.no_grad(): + weight.mul_(torch.tensor(0.97, dtype=torch.float64)) + assert torch.equal(weight.int_data, codes) + assert torch.equal(torch.get_rng_state(), rng) + torch.testing.assert_close(weight.scale, scales * 0.97) + + +@pytest.mark.parametrize("operation", ["add", "addcdiv"]) +@pytest.mark.parametrize("dtype", [torch.float32, torch.float16, torch.bfloat16]) +def test_cuda_storage_update_rounding_versions_and_rng(operation, dtype): + from mini_trainer.modeling._quantized_training import TrainingWeight + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify fused storage updates") + assert torch.cuda.is_available() + torch.manual_seed(29) + weight = nn.Parameter(TrainingWeight.from_float(torch.randn(33, 67, device="cuda", dtype=dtype))) + update = torch.randn_like(weight.dequantize()) + denominator = update.abs() + 1 + + def apply(target): + with torch.no_grad(): + if operation == "add": + return target.add_(update, alpha=-0.03) + return target.addcdiv_(update, denominator, value=-0.03) + + # Warm compilation before saving RNG; compilation itself is outside the + # update's reproducibility contract. + apply(nn.Parameter(weight.detach().clone())) + represented = weight.dequantize().detach() + expected = represented.add(update, alpha=-0.03) if operation == "add" else represented.addcdiv(update, denominator, value=-0.03) + replica = nn.Parameter(weight.detach().clone()) + before_version = weight._version + before_codes_version = weight.int_data._version + rng = torch.cuda.get_rng_state() + assert apply(weight) is weight + torch.cuda.set_rng_state(rng) + apply(replica) + assert weight._version > before_version + assert weight.int_data._version > before_codes_version + assert torch.equal(weight.int_data, replica.int_data) + assert torch.equal(weight.scale, replica.scale) + # Stochastic rounding differs by at most one code interval, plus the + # floating arithmetic's precision. There is no floating master parameter. + bound = expected.abs().amax(1, keepdim=True) / 127 + 4 * torch.finfo(dtype).eps + assert torch.all((weight.dequantize() - expected).abs() <= bound) + assert weight.int_data.dtype == torch.int8 + assert weight.scale.shape == (33,) + + +def test_cuda_storage_update_invalidates_saved_weight(): + from mini_trainer.modeling._quantized_training import TrainingWeight + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify saved-tensor invalidation") + assert torch.cuda.is_available() + weight = nn.Parameter(TrainingWeight.from_float(torch.randn(16, 32, device="cuda"))) + inputs = torch.randn(8, 32, device="cuda", requires_grad=True) + output = nn.functional.linear(inputs, weight) + with torch.no_grad(): + weight.add_(torch.ones_like(weight.dequantize()), alpha=-0.01) + with pytest.raises(RuntimeError, match="modified by an inplace operation"): + output.sum().backward() + + +def test_cuda_storage_kernel_reused_across_parameter_objects_and_rates(monkeypatch): + from torch._dynamo.testing import CompileCounterWithBackend + + from mini_trainer.modeling import _quantized_training as backend + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify storage-kernel reuse") + assert torch.cuda.is_available() + counter = CompileCounterWithBackend("inductor") + operation = backend.update_int8_rows_ + + def update_rows(codes, scales, update, alpha, denominator): + operation(codes, scales, update, alpha, denominator) + + kernel = torch.compile(update_rows, backend=counter, fullgraph=True, dynamic=True) + monkeypatch.setattr(backend, "update_int8_rows_", kernel) + weights = [nn.Parameter(backend.TrainingWeight.from_float(torch.randn(33, 67, device="cuda"))) for _ in range(12)] + optimizer = torch.optim.SGD([{"params": [weight], "lr": 0.01 / (index + 1)} for index, weight in enumerate(weights)], momentum=0.9) + scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.93) + after_warmup = None + for step in range(4): + for weight in weights: + weight.grad = torch.ones(weight.shape, device="cuda") + optimizer.step() + scheduler.step() + if step == 0: + after_warmup = counter.frame_count + assert after_warmup and counter.frame_count == after_warmup + + +def test_cuda_local_matmul_tuning_bounds_temporary_memory(monkeypatch): + from torchao.prototype.quantized_training.int8_mm import _scaled_int8_mm_kernel as upstream + + from mini_trainer.modeling._quantized_matmul import _kernel + from mini_trainer.modeling._quantized_training import scaled_int8_mm + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify first-use tuning memory") + assert torch.cuda.is_available() + assert _kernel is not upstream and _kernel.fn is upstream.fn + # Force actual tuning rather than accepting an earlier process's disk cache. + monkeypatch.setattr(_kernel, "cache", {}) + monkeypatch.setattr(_kernel, "cache_results", False) + left = torch.randint(-127, 128, (17, 129), dtype=torch.int8, device="cuda") + right = torch.randint(-127, 128, (19, 129), dtype=torch.int8, device="cuda").T + rows = torch.rand(17, device="cuda") + columns = torch.rand(19, device="cuda") + torch.cuda.synchronize() + torch.cuda.reset_peak_memory_stats() + before = torch.cuda.memory_allocated() + result = scaled_int8_mm(left, right, rows, columns) + torch.cuda.synchronize() + extra_peak = torch.cuda.max_memory_allocated() - before + # This tiny product must not trigger the old tuner's 256 MiB flush buffer. + assert extra_peak < 16 * 1024**2 + product = left.cpu().to(torch.int64) @ right.cpu().to(torch.int64) + expected = product.float().to("cuda") * rows[:, None] * columns[None, :] + torch.testing.assert_close(result, expected) + + +@pytest.mark.xfail( + strict=True, + raises=torch._dynamo.exc.FailOnRecompileLimitHit, + reason="Dynamo still specializes outer SGD groups on TrainingWeight identities", +) +def test_cuda_compiled_optimizer_handles_many_quantized_groups(): + from torch._dynamo.testing import CompileCounterWithBackend + + from mini_trainer.modeling._quantized_training import TrainingWeight + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify many-group optimizer compilation") + assert torch.cuda.is_available() + weights = [nn.Parameter(TrainingWeight.from_float(torch.randn(64, 128, device="cuda"))) for _ in range(12)] + optimizer = torch.optim.SGD([{"params": [weight], "lr": 0.01 / (index + 1)} for index, weight in enumerate(weights)], momentum=0.9) + scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.93) + counter = CompileCounterWithBackend("inductor") + compile_optimizer(optimizer, backend=counter) + after_warmup = None + # Fail on fallback; some compiled frames alone would not establish that + # every parameter group can execute without exhausting Dynamo's cache. + with torch._dynamo.config.patch(fail_on_recompile_limit_hit=True): + for step in range(6): + for weight in weights: + weight.grad = torch.ones(weight.shape, device="cuda") + optimizer.step() + scheduler.step() + if step == 2: + after_warmup = counter.frame_count + assert after_warmup and counter.frame_count == after_warmup From 9b0e1f4953803b8dbe36c4d014544eabdb026c11 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 00:51:24 +0200 Subject: [PATCH 020/155] fix: support compiled optimizer FMA updates for quantized weights Handle the out-of-place primitive that Dynamo emits for tensor learning rates. This prevents optimizer-loop graph breaks and per-weight cache exhaustion without changing global compiler limits. Restore the many-group regression and extend it to AdamW. Verify one-rounding update bounds, optimizer state, and unbiased sub-code updates with advancing CUDA RNG. Document the paired MNIST result: memory remains lower, but speed and quality parity are not established. Validation: static checks; full suite 264 passed, 90 skipped, 1 expected EMA failure; CUDA regression set 108 passed plus focused arithmetic and RNG checks passed. --- docs/benchmarks.md | 36 +++++++ docs/quantized-training.md | 13 ++- docs/roadmap.md | 8 +- mini_trainer/modeling/_quantized_training.py | 9 ++ tests/test_quantized_training_model.py | 103 +++++++++++++++++-- 5 files changed, 154 insertions(+), 15 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 76cf59d..ceebc94 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -257,3 +257,39 @@ Outer optimizer compilation still falls back when sufficiently many quantized groups specialize on wrapper identities. A separate strict expected-failure test enables hard failure on that fallback; it must be removed when the outer path is fixed. This limitation is distinct from the now-working compiled storage operator. + +### Optimizer FMA dispatch + +Tracing the outer optimizer failure identified missing `prims.fma` dispatch. +Dynamo rewrites tensor-learning-rate `add_`/`addcdiv_` updates as an out-of-place +fused multiply-add followed by `copy_`. The missing operation broke the optimizer +loop into per-weight frames, eventually exhausting the compilation cache. +The quantized weight now supplies its represented floating values for this +out-of-place primitive; the following copy performs stochastic requantization. +No floating master weight is retained. + +The twelve-group SGD probe now stabilizes at two compiled graphs after warmup. +SGD and AdamW regressions pass with hard failure enabled for compiler-cache +fallback, replacing the prior strict expected failure. Additional checks compare +optimizer state and update error against floating arithmetic, verify that FMA +itself neither mutates weights nor consumes RNG, and confirm that quarter-code +updates retain their expected average while CUDA RNG and rounding masks advance. + +Both dense MNIST paths were rerun with model **and optimizer** compilation, the +same seed/split and 15-epoch, batch-128 budget. Both report source hash +`b31568d70b4d48a60e3d03a0c1016a06e7f49d31bbedeb134f4b50489db6ddc7`. + +| Execution | Test accuracy | Whole-run peak MiB | Median training epoch s, epochs 3–15 | Training wall s | +| --- | ---: | ---: | ---: | ---: | +| Float | 92.44% | 236.30 | 0.167 | 11.49 | +| INT8 | 87.38% | 182.97 | 0.227 | 25.11 | + +Compilation compatibility is fixed, but this is still a negative speed/quality +result. Whole-run allocation remains lower than float, but rises relative to +the explicit eager-optimizer row kernel. The INT8 checkpoint records 464 completed +updates versus 465 for float, reflecting one AMP overflow skip; gating remains +active. The accuracy drop is not explained by the passed single-update checks. +Different stochastic rounding trajectories and accumulation over training require +further investigation. Neither this single seed nor compilation success establishes +model-quality parity or completion of the QT goal. Wall times still include +first-use compilation and depend on cache state. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index fd45857..b4b7e2c 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -72,14 +72,17 @@ selected configurations are cached on disk. TorchAO's global operators and tuner are unchanged. The explicit update and matrix operators provide fake execution implementations for model compilation. -Keep `--compile-optimizer` off for general QT workloads for now. Although small -integrated cases pass, compiling an outer optimizer with many quantized parameter -groups can specialize on weight identities and fall back to eager execution. -That outer-optimizer limitation remains a strict CUDA expected-failure regression. +`--compile-optimizer` now supports the FMA primitive that Dynamo uses for tensor +learning-rate updates. The formerly failing twelve-group SGD regression passes, +as does twelve-group AdamW, with hard failure enabled on compiler-cache fallback. +This establishes compilation compatibility, not a performance recommendation: +the current dense MNIST comparison is slower and less accurate with QT than float. +Compiled stochastic requantization can follow a different random trajectory from +the eager row kernel, so a matching seed does not establish identical training. The earlier storage prototype's fake-tensor and dtype-cache failures are resolved by an explicit Triton kernel and custom-operator boundary; the storage kernel does not depend on Dynamo's per-frame variant cache. Model `--compile` remains a -separate option. See the [measured results](benchmarks.md#explicit-cuda-kernels-and-local-tuning). +separate option. See the [measured results](benchmarks.md#optimizer-fma-dispatch). ## Checkpoints and inference diff --git a/docs/roadmap.md b/docs/roadmap.md index de48600..6134c3b 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -105,9 +105,11 @@ that target. The implementation and comparison plan is in The explicit CUDA storage-update kernel and local matrix tuner now reduce whole-run peak allocation in the dense MNIST comparison by approximately 35%; [training speed remains unproven](benchmarks.md#explicit-cuda-kernels-and-local-tuning). -A strict CUDA expected-failure regression records outer optimizer compilation -falling back after specializing on many quantized parameter-group identities. -Keep that limit visible while improving optimizer dispatch and throughput. +The outer optimizer fallback was traced to missing FMA dispatch for tensor +learning rates. Twelve-group SGD and AdamW now compile without cache fallback, +and the former expected-failure marker is removed. The [paired compiled-optimizer +result](benchmarks.md#optimizer-fma-dispatch) is still slower and less accurate +under QT; throughput and convergence work remain required. Deliver quantization-aware training and post-training inference quantization as separate opt-in capabilities, recording actual weight/activation bit widths, diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index e79190a..47e3fbc 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -74,6 +74,15 @@ def add(func, types, args, kwargs): return func(*(value.dequantize() if isinstance(value, TrainingWeight) else value for value in args), **kwargs) +@TrainingWeight.implements(torch.ops.prims.fma.default) +def fused_multiply_add(func, types, args, kwargs): + # Dynamo lowers add_/addcdiv_ with tensor learning rates to fma + copy_. + # The out-of-place result is floating; copy_ performs the ordinary single + # stochastic requantization. Leaving fma unsupported breaks optimizer loops + # into per-weight frames and eventually exhausts the compilation cache. + return func(*(value.dequantize() if isinstance(value, TrainingWeight) else value for value in args), **kwargs) + + @TrainingWeight.implements(torch.ops.aten.mul_.Tensor) def multiply_inplace(func, types, args, kwargs): original, multiplier = args diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index be3c036..4d9c91c 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -457,12 +457,8 @@ def test_cuda_local_matmul_tuning_bounds_temporary_memory(monkeypatch): torch.testing.assert_close(result, expected) -@pytest.mark.xfail( - strict=True, - raises=torch._dynamo.exc.FailOnRecompileLimitHit, - reason="Dynamo still specializes outer SGD groups on TrainingWeight identities", -) -def test_cuda_compiled_optimizer_handles_many_quantized_groups(): +@pytest.mark.parametrize("kind", ["sgd", "adamw"]) +def test_cuda_compiled_optimizer_handles_many_quantized_groups(kind): from torch._dynamo.testing import CompileCounterWithBackend from mini_trainer.modeling._quantized_training import TrainingWeight @@ -472,7 +468,8 @@ def test_cuda_compiled_optimizer_handles_many_quantized_groups(): pytest.skip("Set RUN_CUDA_TESTS=1 to verify many-group optimizer compilation") assert torch.cuda.is_available() weights = [nn.Parameter(TrainingWeight.from_float(torch.randn(64, 128, device="cuda"))) for _ in range(12)] - optimizer = torch.optim.SGD([{"params": [weight], "lr": 0.01 / (index + 1)} for index, weight in enumerate(weights)], momentum=0.9) + groups = [{"params": [weight], "lr": 0.01 / (index + 1)} for index, weight in enumerate(weights)] + optimizer = torch.optim.SGD(groups, momentum=0.9, weight_decay=0.1) if kind == "sgd" else torch.optim.AdamW(groups, foreach=False) scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.93) counter = CompileCounterWithBackend("inductor") compile_optimizer(optimizer, backend=counter) @@ -488,3 +485,95 @@ def test_cuda_compiled_optimizer_handles_many_quantized_groups(): if step == 2: after_warmup = counter.frame_count assert after_warmup and counter.frame_count == after_warmup + + +@pytest.mark.parametrize("position", [0, 1, 2]) +def test_fma_uses_represented_weights_without_mutation_or_rounding(position): + from mini_trainer.modeling._quantized_training import TrainingWeight + + values = [torch.randn(4, 7) for _ in range(3)] + weight = nn.Parameter(TrainingWeight.from_float(values[position])) + values[position] = weight + codes = weight.int_data.clone() + scales = weight.scale.clone() + version = weight._version + rng = torch.get_rng_state() + with torch.no_grad(): + expected = torch.ops.prims.fma(*(value.dequantize() if value is weight else value for value in values)) + actual = torch.ops.prims.fma(*values) + assert not isinstance(actual, TrainingWeight) + torch.testing.assert_close(actual, expected, rtol=0, atol=0) + assert torch.equal(weight.int_data, codes) and torch.equal(weight.scale, scales) + assert weight._version == version + assert torch.equal(torch.get_rng_state(), rng) + + +@pytest.mark.parametrize("kind", ["sgd", "adamw"]) +def test_cuda_compiled_quantized_update_matches_float_before_rounding(kind): + from mini_trainer.modeling._quantized_training import TrainingWeight + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify compiled QT update arithmetic") + assert torch.cuda.is_available() + torch.manual_seed(107) + weight = nn.Parameter(TrainingWeight.from_float(torch.randn(32, 64, device="cuda"))) + reference = nn.Parameter(weight.dequantize().detach().clone()) + cls = torch.optim.SGD if kind == "sgd" else torch.optim.AdamW + options = {"lr": 0.03, "weight_decay": 0.1, "foreach": False} + if kind == "sgd": + options.update(momentum=0.9, nesterov=True) + optimizer, eager = cls([weight], **options), cls([reference], **options) + schedulers = [torch.optim.lr_scheduler.StepLR(opt, step_size=1, gamma=0.93) for opt in (optimizer, eager)] + compile_optimizer(optimizer) + for _ in range(6): + with torch.no_grad(): + reference.copy_(weight.dequantize()) + gradient = torch.randn_like(reference) + weight.grad, reference.grad = gradient.clone(), gradient.clone() + optimizer.step() + eager.step() + for scheduler in schedulers: + scheduler.step() + bound = reference.detach().abs().amax(1, keepdim=True) / 127 + 1e-6 + assert torch.all((weight.dequantize() - reference).abs() <= bound) + for key, value in optimizer.state[weight].items(): + expected = eager.state[reference][key] + if isinstance(value, torch.Tensor): + torch.testing.assert_close(value, expected.to(value.device), rtol=1e-5, atol=2e-6) + else: + assert value == expected + + +def test_cuda_compiled_stochastic_rounding_preserves_sub_code_updates(): + from mini_trainer.modeling._quantized_training import TrainingWeight + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify sub-code compiled QT updates") + assert torch.cuda.is_available() + torch.manual_seed(109) + weight = nn.Parameter(TrainingWeight.from_float(torch.ones(128, 1024, device="cuda"))) + optimizer = torch.optim.SGD([weight], lr=1, foreach=False) + compile_optimizer(optimizer) + gradient = torch.full(weight.shape, -0.25 / 127, device="cuda") + gradient[:, 0] = 0 + previous_codes = None + for _ in range(4): + # Hold each row's maximum fixed, then request a quarter-code increase + # elsewhere. Rounding must retain that signal statistically rather than + # deterministically dropping it or biasing it upward. + with torch.no_grad(): + weight.int_data.zero_() + weight.int_data[:, 0] = 127 + weight.scale.fill_(1 / 127) + weight.grad = gradient + rng = torch.cuda.get_rng_state() + optimizer.step() + codes = weight.int_data[:, 1:] + assert torch.all((codes == 0) | (codes == 1)) + assert abs(codes.float().mean().item() - 0.25) < 0.01 + assert not torch.equal(rng, torch.cuda.get_rng_state()) + if previous_codes is not None: + assert not torch.equal(codes, previous_codes) + previous_codes = codes.clone() From 2bcc32f6c38cb0755559e681ae25054aa7adc14e Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 01:11:37 +0200 Subject: [PATCH 021/155] perf: keep stacked batches intact through loader collation Tag repository batch indices so LazyDataset can defer redundant per-sample views. Preserve ordinary external batched fetches, default collation, dataset identity, spawned workers, shuffle RNG, and CUDA prefetch. The small-image cache probe improves from 0.52M to 2.22M samples/s. Record larger-batch MNIST runs showing QT memory and warm-runtime gains, while retaining cold-start and accuracy-variation limits. Validation: static checks; full suite 274 passed, 90 skipped, 1 expected EMA failure; CUDA loader and QT suite 108 passed. --- dev/README.md | 19 +++++++++++ docs/benchmarks.md | 40 ++++++++++++++++++++++ docs/quantized-training.md | 5 ++- docs/roadmap.md | 8 ++++- mini_trainer/data/io.py | 45 +++++++++++++++++++++---- mini_trainer/data/loader.py | 12 ++++++- tests/utils/test_loader.py | 67 +++++++++++++++++++++++++++++++++++++ 7 files changed, 187 insertions(+), 9 deletions(-) diff --git a/dev/README.md b/dev/README.md index ca4fe53..2c40a6a 100644 --- a/dev/README.md +++ b/dev/README.md @@ -193,6 +193,25 @@ Worker processes always use the ordinary gather path and DataLoader's parent-sid pinning; this option never initializes the CUDA pin allocator in a worker. CPU training, CUDA-cached datasets and scalar indexing keep their existing behavior. +### Direct collation of stacked batches + +Repository loaders now retain the gathered batch tensors through collation, +avoiding creation of one image/label view per sample. Their batch sampler tags +index lists for this internal path; dataset identity, shuffle/drop-last behavior, +distributed sampler access and RNG consumption are preserved. + +Ordinary external `LazyDataset.__getitems__` calls still return actual sample +lists, including direct `torch.stack` compatibility. An external DataLoader that +reuses the repository batch sampler with its default collator materializes sample +views on demand. Tests cover image-only and image/label batches with both direct +loading and CPU caches, including spawned workers and CUDA prefetch. + +A one-thread cache benchmark with 4,096 uint8 RGB 28×28 images, batch size 128, +and seven alternating trials measured approximately 0.52 million samples/s before +this change and 2.22 million afterward. This isolates cached iteration; larger +images, decoding, transfer and model compute change the overall benefit. See the +[integrated measurements](../docs/benchmarks.md#larger-batches-and-direct-collation). + ### Optimizer compilation `mt_train --compile-optimizer` opts into compiling optimizer updates independently diff --git a/docs/benchmarks.md b/docs/benchmarks.md index ceebc94..f9d6413 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -293,3 +293,43 @@ Different stochastic rounding trajectories and accumulation over training requir further investigation. Neither this single seed nor compilation success establishes model-quality parity or completion of the QT goal. Wall times still include first-use compilation and depend on cache state. + +### Larger batches and direct collation + +The dense MNIST profile was also evaluated at batch size 512 for 60 epochs, +using the same data split, seed 42, model, SGD learning rate and model/optimizer +compilation. This is a separate workload, not a replacement for the unfavorable +batch-128 results. Before the loader change, the float/INT8 pair reached +93.06%/92.84% accuracy, with median later training epochs of 0.0809/0.0756 seconds +and whole-run peaks of 254.26/190.86 MiB. Initial wall times were 33.00/48.02 seconds, +including INT8 first-use compilation and tuning. + +A warmed-up epoch trace showed the cached loader creating per-sample views of +already-stacked tensors before the repository collator returned those same batch +tensors. Direct collation removes this work. In the separate one-thread cache +probe (4,096 RGB uint8 28×28 images, batch 128, seven trials), cached iteration +increased from 520,193 to 2,219,771 samples/s; the scalar-fetch control measured +281,768 and 293,087 samples/s respectively. These are loader-only measurements. + +The integrated pair with direct collation and an unchanged INT8 repeat produced: + +| Execution | Test accuracy | Whole-run peak MiB | Median training epoch s, epochs 3–60 | Training wall s | +| --- | ---: | ---: | ---: | ---: | +| Float | 93.06% | 254.26 | 0.0802 | 30.61 | +| INT8 | 93.00% | 184.32 | 0.0711 | 27.13 | +| INT8, unchanged repeat | 92.54% | 184.32 | 0.0684 | 26.02 | + +These runs share source hash +`60773b54432fcbe6a9f8f5ee6b9f40186d1a4d2b3d33760136a5187205ef6a13` +and the same dataset manifest. They provide evidence of lower whole-run memory +and faster training for this workload, including total wall time with previously +populated compiler/tuner caches. They do not establish a cold-start advantage or +a speedup for other batch sizes, architectures or hardware. + +Float predictions matched the pre-loader-change run exactly. INT8 predictions +varied both across the loader change and between two runs of identical code and +seed; these CUDA profiles explicitly allow nondeterministic execution. All runs +retained the same held-out labels and paths. Independent shuffled-loader tests +require exact batches and identical CPU RNG consumption over multiple epochs. +The accuracy variation means these single-seed observations do not establish +quality parity; multi-seed convergence and broader workload validation remain open. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index b4b7e2c..778883f 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -76,7 +76,10 @@ implementations for model compilation. learning-rate updates. The formerly failing twelve-group SGD regression passes, as does twelve-group AdamW, with hard failure enabled on compiler-cache fallback. This establishes compilation compatibility, not a performance recommendation: -the current dense MNIST comparison is slower and less accurate with QT than float. +the batch-128 dense MNIST comparison is slower and less accurate with QT than float. +The [larger-batch profile](benchmarks.md#larger-batches-and-direct-collation) shows +lower memory and faster training after compiler caches are populated, but still +needs multi-seed quality validation. Compiled stochastic requantization can follow a different random trajectory from the eager row kernel, so a matching seed does not establish identical training. The earlier storage prototype's fake-tensor and dtype-cache failures are resolved diff --git a/docs/roadmap.md b/docs/roadmap.md index 6134c3b..f3a0b01 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -104,13 +104,19 @@ that target. The implementation and comparison plan is in The explicit CUDA storage-update kernel and local matrix tuner now reduce whole-run peak allocation in the dense MNIST comparison by approximately 35%; -[training speed remains unproven](benchmarks.md#explicit-cuda-kernels-and-local-tuning). +[the batch-128 speed result remains unfavorable](benchmarks.md#explicit-cuda-kernels-and-local-tuning). The outer optimizer fallback was traced to missing FMA dispatch for tensor learning rates. Twelve-group SGD and AdamW now compile without cache fallback, and the former expected-failure marker is removed. The [paired compiled-optimizer result](benchmarks.md#optimizer-fma-dispatch) is still slower and less accurate under QT; throughput and convergence work remain required. +Direct collation removes redundant per-sample views from repository loaders while +preserving external default collation and shuffle RNG. A [larger-batch MNIST +profile](benchmarks.md#larger-batches-and-direct-collation) now shows lower memory +and faster training with populated compiler caches. Multi-seed quality comparisons, +cold-start cost, and broader workloads still need validation. + Deliver quantization-aware training and post-training inference quantization as separate opt-in capabilities, recording actual weight/activation bit widths, calibration data, backend kernels, checkpoint/resume and export/runtime support. diff --git a/mini_trainer/data/io.py b/mini_trainer/data/io.py index 9012e58..9e45410 100644 --- a/mini_trainer/data/io.py +++ b/mini_trainer/data/io.py @@ -340,14 +340,40 @@ def _infer_numeric_dtype(seq) -> Any: return object +class _DirectBatchIndices(list): + """Index list for the repository collator, which accepts stacked tensors.""" + + class _FetchedBatch(list): """Sample views for standard collators, with the already-stacked batch attached.""" - def __init__(self, data): + def __init__(self, data, *, sample_views=True): self.data = data - # unbind produces views, not sample copies. An actual list preserves - # torch.stack compatibility in external DataLoaders' default collators. - super().__init__(data.unbind(0) if isinstance(data, torch.Tensor) else zip(*(value.unbind(0) for value in data))) + self._sample_views = False + super().__init__() + if sample_views: + self._materialize() + + def _materialize(self): + if not self._sample_views: + data = self.data + super().extend(data.unbind(0) if isinstance(data, torch.Tensor) else zip(*(value.unbind(0) for value in data))) + self._sample_views = True + + def __getitem__(self, index): + # An external default collator starts with batch[0]. Materialize here + # if it reuses our tagged batch sampler with a different collator. + self._materialize() + return super().__getitem__(index) + + def __iter__(self): + self._materialize() + return super().__iter__() + + def __len__(self): + if self._sample_views: + return super().__len__() + return len(self.data) if isinstance(self.data, torch.Tensor) else len(self.data[0]) class LazyDataset(torch.utils.data.Dataset): @@ -361,6 +387,8 @@ class LazyDataset(torch.utils.data.Dataset): * "guess" : Select a caching strategy via heuristic. """ + _supports_direct_batches = True + def __init__( # noqa: D107 self, func: Callable[[Any], torch.Tensor | tuple[torch.Tensor, ...] | list[torch.Tensor]], @@ -523,8 +551,13 @@ def __getitems__(self, indices): ) else: data = tuple(tensor.index_select(0, index) for tensor in tensors) - return _FetchedBatch(data[0] if self._ram_was_single_tensor else data) - return _FetchedBatch(self[indices]) + data = data[0] if self._ram_was_single_tensor else data + else: + data = self[indices] + # Ordinary external batched fetches still return actual sample lists, + # including direct torch.stack compatibility. Our sampler marks only + # batches whose collator can consume the stacked storage directly. + return _FetchedBatch(data, sample_views=not isinstance(indices, _DirectBatchIndices)) def __getitem__(self, index): match self._cache_mode: diff --git a/mini_trainer/data/loader.py b/mini_trainer/data/loader.py index 2c7654d..5ecce45 100644 --- a/mini_trainer/data/loader.py +++ b/mini_trainer/data/loader.py @@ -13,6 +13,7 @@ from .io import ( CACHE_MODE, LazyDataset, + _DirectBatchIndices, _FetchedBatch, guess_cache_mode, make_read_and_resize_fn, @@ -69,6 +70,14 @@ def __call__(self, x: str) -> torch.Tensor: return self.hook(self.reader(x)) +class _DirectBatchSampler(BatchSampler): + """Retain normal sampling while requesting stacked batches from LazyDataset.""" + + def __iter__(self): + for indices in super().__iter__(): + yield _DirectBatchIndices(indices) + + def _collate_batch(samples): if isinstance(samples, _FetchedBatch): # Match default_collate's tuple-to-list convention for (image, label). @@ -105,7 +114,8 @@ def get_dataloader( # noqa: D103 if num_workers > 0 and multiprocessing_context is not None: mp_context = multiprocessing_context - sampler = BatchSampler(base_sampler, batch_size=batch_size, drop_last=drop_last) + sampler_cls = _DirectBatchSampler if getattr(dataset, "_supports_direct_batches", False) else BatchSampler + sampler = sampler_cls(base_sampler, batch_size=batch_size, drop_last=drop_last) loader_cls = CUDAPrefetchLoader if cuda_prefetch else DataLoader transfer_kwargs = {"device": device} if cuda_prefetch else {} diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index a82ff7a..ea366b3 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -379,3 +379,70 @@ def test_pinned_cache_gather_never_pins_inside_worker(monkeypatch): result = data_loader._collate_batch(dataset.__getitems__([2, 0])) assert not result.is_pinned() assert result.tolist() == [2, 0] + + +def test_cached_repository_loader_does_not_unpack_sample_views(): + from torch.utils._python_dispatch import TorchDispatchMode + + images = torch.arange(96).reshape(8, 3, 2, 2) + dataset = data_io.LazyDataset(lambda item: (images[item[0]], torch.tensor(item[0])), (list(range(8)),), cache="cpu", cache_workers=0) + loader = data_loader.get_dataloader(dataset, "val", 4, 0, False, torch.device("cpu")) + unpacked = [] + + class ObserveUnpacking(TorchDispatchMode): + def __torch_dispatch__(self, func, types, args=(), kwargs=None): + if func in (torch.ops.aten.unbind.int, torch.ops.aten.select.int): + unpacked.append(func) + return func(*args, **(kwargs or {})) + + with ObserveUnpacking(): + actual = list(loader) + assert not unpacked + assert loader.dataset is dataset + torch.testing.assert_close(torch.cat([batch[0] for batch in actual]), images) + assert torch.cat([batch[1] for batch in actual]).tolist() == list(range(8)) + + +def test_direct_batches_preserve_shuffled_sampling_and_rng(): + dataset = data_io.LazyDataset(lambda item: torch.tensor(item[0]), (list(range(19)),), cache="cpu", cache_workers=0) + optimized = data_loader.get_dataloader(dataset, "train", 4, 0, False, torch.device("cpu")) + reference = torch.utils.data.DataLoader( + dataset, batch_sampler=torch.utils.data.BatchSampler(RandomSampler(dataset), batch_size=4, drop_last=True) + ) + torch.manual_seed(31) + for _ in range(3): + rng = torch.get_rng_state() + expected = list(reference) + expected_rng = torch.get_rng_state() + torch.set_rng_state(rng) + actual = list(optimized) + torch.testing.assert_close(actual, expected, rtol=0, atol=0) + assert torch.equal(torch.get_rng_state(), expected_rng) + + +@pytest.mark.parametrize("cache", ["none", "cpu"]) +@pytest.mark.parametrize("workers", [0, 1]) +@pytest.mark.parametrize("labels", [False, True]) +def test_default_collator_can_reuse_repository_batch_sampler(metadata, cache, workers, labels): + if labels: + datasets, loaders = get_dataset_dataloader( + metadata, resize_size=4, modes=("val",), cache=cache, cache_workers=0, batch_size=2, num_workers=0 + ) + dataset, optimized = datasets[0], loaders[0] + else: + dataset, optimized = get_inference_dataloader( + metadata["path"], resize_size=4, cache=cache, cache_workers=0, batch_size=2, num_workers=0 + ) + external = torch.utils.data.DataLoader( + dataset, + batch_sampler=optimized.batch_sampler, + num_workers=workers, + multiprocessing_context="spawn" if workers else None, + ) + for expected, actual in zip(optimized, external, strict=True): + torch.testing.assert_close(actual, expected, rtol=0, atol=0) + if not labels: + # Ordinary direct batched fetches remain real Tensor lists, including + # torch.stack's C-level sequence handling (which bypasses list methods). + direct = dataset.__getitems__([4, 1, 1]) + torch.testing.assert_close(torch.stack(direct), dataset[[4, 1, 1]]) From a9ccb3c3f2b28230cdc988d3c9f1a9cdcfb8838c Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 01:35:21 +0200 Subject: [PATCH 022/155] fix: invalidate quantized graphs when kernel implementations change Include matrix and update kernel sources in the training tensor's compiler fingerprint, preventing old opaque graphs from hiding operator changes during validation. Retain the opaque matrix path after the graph-visible experiment showed excessive startup cost and unstable real-data accuracy despite passing numerical checks. Validation: static checks; full suite 274 passed, 90 skipped, 1 expected EMA failure; retained CUDA backend 52 passed. --- docs/benchmarks.md | 30 ++++++++++++++++++++ mini_trainer/modeling/_quantized_training.py | 11 +++++-- 2 files changed, 38 insertions(+), 3 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index f9d6413..464c23d 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -333,3 +333,33 @@ retained the same held-out labels and paths. Independent shuffled-loader tests require exact batches and identical CPU RNG consumption over multiple epochs. The accuracy variation means these single-seed observations do not establish quality parity; multi-seed convergence and broader workload validation remain open. + +### Graph-visible matmul experiment (not retained) + +An experimental replacement of the opaque INT8 matrix operator with +`torch.library.triton_op` made its launch visible inside compiled graphs. A +profiler regression confirmed that Python custom-operator dispatch disappeared, +and 53 CUDA numerical/model regressions passed. Graph capture initially lost the +upstream kernel's default `GROUP_M` argument; passing it explicitly fixed that +compilation error. These results were insufficient to establish a usable change. + +The batch-128, 15-epoch dense MNIST control ran from an isolated copy of commit +`2bcc32f`, preserving the direct loader and all training settings. It reached +92.16% accuracy with a 0.221-second median later training epoch and 12.33-second +training call. The graph-visible experiment initially measured 0.211 seconds per +later epoch, but required 78.44 seconds overall. With the final experiment source +hash `335c9dd82b854dd199f7d838b5595a312d46a6b57580e86dff2835b378e7474b`, +two runs produced 76.18% and 62.82% accuracy; their training-call times were +75.39 and 11.32 seconds, and later-epoch medians 0.209 and 0.166 seconds. All +retained the same 182.97 MiB whole-run allocation peak. + +The cached speed result cannot justify the quality degradation, and the numerical +tests did not identify its cause. The graph-visible path was therefore removed; +the opaque operator remains in use. This is an unresolved experimental result, +not evidence that the operator API itself is incorrect. + +One necessary safeguard is retained: the training tensor's compiler fingerprint +now includes the matrix and update kernel source files as well as the backend +file. The previous fingerprint could allow an old opaque graph to hide a changed +operator implementation during validation. First-use compilation must be measured +again after any of these source files change. diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index 47e3fbc..35cd2c2 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -14,9 +14,14 @@ from ._quantized_update import update_int8_rows_ # Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary -# graph key. Include the backend implementation so changing backward math cannot -# reuse a graph compiled for an earlier version. Compute once when importing. -_IMPLEMENTATION_HASH = hashlib.sha256(Path(__file__).read_bytes()).hexdigest() +# graph key. Include the backend and kernel implementations so changing hidden +# backward math or operator decomposition cannot reuse an earlier graph. +# Compute once when importing. +_IMPLEMENTATION_HASH = hashlib.sha256( + b"".join( + Path(__file__).with_name(name).read_bytes() for name in ("_quantized_training.py", "_quantized_matmul.py", "_quantized_update.py") + ) +).hexdigest() class TrainingWeight(Int8QuantizedTrainingLinearWeight): From 1ab482f22260fbe6249e8263296f5dcf6351d6db Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 01:48:20 +0200 Subject: [PATCH 023/155] bench: track multi-seed quantized training speed and quality --- .github/workflows/benchmarks.yml | 6 +++- dev/benchmarks/README.md | 22 ++++++++++++++ dev/benchmarks/run.py | 1 + dev/benchmarks/summarize.py | 23 ++++++++++++--- dev/check-benchmarks.sh | 19 +++++++++++-- docs/benchmarks.md | 49 ++++++++++++++++++++++++++++++++ docs/roadmap.md | 7 +++-- tests/test_benchmark_datasets.py | 39 +++++++++++++++++++++---- 8 files changed, 152 insertions(+), 14 deletions(-) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index ca84134..555fe78 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -98,6 +98,9 @@ jobs: - name: Paired dense MNIST quantized training if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() run: bash dev/check-benchmarks.sh qt-dense benchmark-qt-dense + - name: Multi-seed large-batch quantized training + if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() + run: bash dev/check-benchmarks.sh qt-large-batch benchmark-qt-large-batch - name: Synthetic GPU precision profiles run: bash dev/check-benchmarks.sh gpu benchmark-gpu - name: Real-data progression @@ -107,7 +110,7 @@ jobs: if: always() run: | python3 -m dev.benchmarks.summarize benchmark-gpu >> "$GITHUB_STEP_SUMMARY" - for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense; do + for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense benchmark-qt-large-batch; do if [[ -d "$qt_results" ]]; then python3 -m dev.benchmarks.summarize "$qt_results" >> "$GITHUB_STEP_SUMMARY" fi @@ -128,6 +131,7 @@ jobs: benchmark-qt/ benchmark-qt-real/ benchmark-qt-dense/ + benchmark-qt-large-batch/ !benchmark-qt/**/data/**/*.png !benchmark-gpu/**/data/**/*.png - name: Remove the disposable GPU environment diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index daa0045..dbecfdd 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -403,3 +403,25 @@ only after the logger's final reset; their CUDA readings are now marked `unverified`, and cannot establish whole-run memory reductions. This correction does not affect the standalone kernel and transfer probes, which do not use that logger. It also does not affect recorded accuracy or physical parameter storage. + +### Multi-seed large-batch comparison + +```bash +CUDA_VISIBLE_DEVICES=0 BENCHMARK_DATA_ROOT=/path/to/datasets \ + bash dev/check-benchmarks.sh qt-large-batch /tmp/benchmarks-qt-large-batch +``` + +This additional profile uses the same dense model, optimizer, learning rates and +data preparation with batch size 512, 60 epochs and both model and optimizer +compilation. It runs matched float/INT8 pairs for seeds 42, 43 and 44, alternating +their order. All six reports and failures are retained. The shared runner defaults +to one Inductor compiler worker; an explicit `TORCHINDUCTOR_COMPILE_THREADS` +overrides this, and reports record the environment setting. + +QT plus real-data Actions runs include this profile alongside the original +small-batch pair. Summaries show median training-phase time from epoch 3 onward +separately from total training-call wall time. The former includes loading, +preprocessing and batch logging, excludes validation/figures/checkpoints, and +may still contain later compilation. Compiler caches are not cleared between +runs: neither column establishes fresh-cache performance. Real-data completion +still has no quality acceptance threshold; inspect accuracy for every seed. diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 087e1db..b283299 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -281,6 +281,7 @@ def run( "amp_exercised": precision != torch.float32, "deterministic_algorithms": torch.are_deterministic_algorithms_enabled(), "torch_threads": torch.get_num_threads(), + "torchinductor_compile_threads_env": os.environ.get("TORCHINDUCTOR_COMPILE_THREADS"), "python": platform.python_version(), "platform": platform.platform(), "cuda_version": torch.version.cuda, diff --git a/dev/benchmarks/summarize.py b/dev/benchmarks/summarize.py index 16fef0f..5886d0f 100644 --- a/dev/benchmarks/summarize.py +++ b/dev/benchmarks/summarize.py @@ -1,6 +1,7 @@ """Render retained JSON benchmark reports as a repository/Actions summary.""" import json +import statistics from argparse import ArgumentParser from pathlib import Path @@ -9,8 +10,9 @@ def summarize(directory: Path) -> str: lines = [ "# Dataset benchmark results", "", - "| Run | Status | Device / precision | QT coverage | Accuracy by level | Parameter bytes | Peak CUDA MiB | Training wall time |", - "| --- | --- | --- | --- | --- | --- | --- | --- |", + "| Run | Status | Device / precision | QT coverage | Accuracy by level | Parameter bytes | Peak CUDA MiB | " + "Median train epoch 3+ | Training wall time |", + "| --- | --- | --- | --- | --- | --- | --- | --- | --- |", ] reports = sorted(directory.rglob("report.json")) for path in reports: @@ -18,6 +20,16 @@ def summarize(directory: Path) -> str: accuracy = ", ".join(f"{value:.2%}" for value in report.get("level_accuracies", [])) or "—" seconds = report.get("training_wall_seconds") duration = f"{seconds:.2f}s" if seconds is not None else "—" + later_epochs = [ + phase["seconds"] for phase in report.get("phase_measurements", []) if phase["phase"] == "train" and phase["epoch"] >= 2 + ] + later_duration = ( + f"{statistics.median(later_epochs):.3f}s" + if later_epochs and report.get("phase_measurement_scope") + else "unverified" + if later_epochs + else "—" + ) name = path.parent.relative_to(directory).as_posix() device = f"{report.get('device', '?')} / {report.get('dtype', '?')}" recipe = report.get("quantization_recipe") @@ -34,10 +46,11 @@ def summarize(directory: Path) -> str: else "—" ) lines.append( - f"| {name} | {report['status']} | {device} | {quantization} | {accuracy} | {parameter_bytes} | {peak_memory} | {duration} |" + f"| {name} | {report['status']} | {device} | {quantization} | {accuracy} | {parameter_bytes} | " + f"{peak_memory} | {later_duration} | {duration} |" ) if not reports: - lines.append("| No reports produced | incomplete | — | — | — | — | — | — |") + lines.append("| No reports produced | incomplete | — | — | — | — | — | — | — |") lines.extend( [ "", @@ -50,6 +63,8 @@ def summarize(directory: Path) -> str: "QT coverage counts quantized Linear modules; other operations may remain floating point.", "Parameter bytes describe stored parameters. CUDA peaks cover training, excluding final held-out inference.", "Older CUDA readings without a scope marker are unverified because logger resets could hide earlier peaks.", + "Later-epoch medians use timed training phases from epoch 3 onward, including loading, preprocessing and batch logging.", + "They exclude validation/figures/checkpoints, but may still include later compilation; they do not replace total wall time.", ] ) return "\n".join(lines) + "\n" diff --git a/dev/check-benchmarks.sh b/dev/check-benchmarks.sh index 9f5892a..3c4e1f8 100644 --- a/dev/check-benchmarks.sh +++ b/dev/check-benchmarks.sh @@ -9,13 +9,14 @@ if [[ -e "$results" ]]; then echo 'Results directory must be new.' >&2 exit 2 fi -case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real or qt-dense.' >&2; exit 2 ;; esac +case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense or qt-large-batch.' >&2; exit 2 ;; esac mkdir -p -- "$results" status=0 run_profile() { local profile="$1" shift - if OMP_NUM_THREADS=1 MPLBACKEND=Agg "$benchmark_python" -m dev.benchmarks.run --output "$results/$profile" "$@" > "$results/$profile.log" 2>&1; then + if OMP_NUM_THREADS=1 MPLBACKEND=Agg TORCHINDUCTOR_COMPILE_THREADS="${TORCHINDUCTOR_COMPILE_THREADS:-1}" \ + "$benchmark_python" -m dev.benchmarks.run --output "$results/$profile" "$@" > "$results/$profile.log" 2>&1; then return else local exit_code="$?" @@ -55,6 +56,20 @@ elif [[ "$mode" == qt-dense ]]; then --model-profile dense --optimizer sgd --learning-rate 0.3 --epochs 15 --batch-size 128 --compile \ --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --allow-nondeterministic "${quantization[@]}" done +elif [[ "$mode" == qt-large-batch ]]; then + : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/}" + for seed in 42 43 44; do + precisions=(float int8) + if (( seed % 2 )); then precisions=(int8 float); fi + for precision in "${precisions[@]}"; do + quantization=() + if [[ "$precision" == int8 ]]; then quantization=(--quantized-training); fi + run_profile "mnist-large-batch-$precision-seed$seed" --dataset mnist --data-root "$BENCHMARK_DATA_ROOT/mnist" \ + --seed "$seed" --model-profile dense --optimizer sgd --learning-rate 0.3 --epochs 60 --batch-size 512 \ + --compile --compile-optimizer --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 \ + --allow-nondeterministic "${quantization[@]}" + done + done elif [[ "$mode" == qt-real ]]; then : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/ and blair/}" : "${BLAIR_CLASS_SPEC:?Set BLAIR_CLASS_SPEC to a reviewed Blair class specification}" diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 464c23d..c7e49f3 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -363,3 +363,52 @@ now includes the matrix and update kernel source files as well as the backend file. The previous fingerprint could allow an old opaque graph to hide a changed operator implementation during validation. First-use compilation must be measured again after any of these source files change. + +### Continuous multi-seed large-batch profile + +The shared [`qt-large-batch` profile](../dev/benchmarks/README.md#multi-seed-large-batch-comparison) +adds seeds 42, 43 and 44 to the dense MNIST batch-512, 60-epoch comparison, with +both model and optimizer compilation. Run order alternates float/INT8 between +seeds. The optional QT plus real-data Actions job retains all six reports and +shows later-epoch timing alongside accuracy, peak memory and whole training-call +time. The original small-batch comparison remains in the pipeline. + +Each seed controls both initialization and the training/validation split; paired +float/INT8 runs use the same manifest. The test set is fixed and evaluates the +final checkpoint. These runs allow nondeterministic CUDA execution, and compiler +caches are not cleared between runs. Three seeds are a useful regression signal, +not a quality-equivalence test or a controlled cold-start benchmark. + +The first local run on the RTX 3080 Ti Laptop GPU produced the following results. +All six reports share runtime source hash +`27184f27cd4158db5ad94cdd10776b6d94b2c2dd8ec0248a4712120374298deb` +and lock hash `43ad5c7df81212b3bcd536220201f8888666723507c317e8c86dd538fb595745`. +Each pair's manifest, held-out labels and paths were verified equal. These are +sequential runs with one compiler worker and no concurrent GPU tests. + +| Seed | Execution | Test accuracy | Whole-run peak MiB | Median training epoch s, epochs 3–60 | Training wall s | +| --- | --- | ---: | ---: | ---: | ---: | +| 42 | Float | 93.06% | 254.26 | 0.074 | 28.65 | +| 42 | INT8 | 92.52% | 184.32 | 0.071 | 34.72 | +| 43 | Float | 92.76% | 249.13 | 0.076 | 35.44 | +| 43 | INT8 | 92.42% | 184.32 | 0.071 | 28.49 | +| 44 | Float | 92.74% | 249.13 | 0.074 | 33.22 | +| 44 | INT8 | 92.64% | 184.32 | 0.078 | 28.98 | + +INT8 reduces whole-run peak allocation by 26–28% in every pair, while held-out +accuracy is lower by 0.10–0.54 percentage points (mean difference −0.33 points). +Later training phases are faster in two pairs and slower in one; whole training +calls likewise show mixed results. The first INT8 run follows a kernel fingerprint +change, which can trigger recompilation. This supports the memory improvement, +but does not establish a reliable speed win or quality parity. Convergence and +startup profiling remain necessary before recommending QT for this workload. + +Local reports, logs, checkpoints and predictions were retained under +`/tmp/mini-trainer-qt-large-batch-multiseed`; these temporary artifacts are not +committed. Future enabled Actions runs retain the corresponding artifacts for +90 days and publish the table in the job summary. No remote job was dispatched +for this local validation. + +The synthetic CUDA float/INT8 oracle pair was rerun with the retained kernels; +both reached the required 100% accuracy after training and checkpoint reload. +This verifies the simple task, not convergence equivalence on MNIST or Blair. diff --git a/docs/roadmap.md b/docs/roadmap.md index f3a0b01..103260c 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -114,8 +114,11 @@ under QT; throughput and convergence work remain required. Direct collation removes redundant per-sample views from repository loaders while preserving external default collation and shuffle RNG. A [larger-batch MNIST profile](benchmarks.md#larger-batches-and-direct-collation) now shows lower memory -and faster training with populated compiler caches. Multi-seed quality comparisons, -cold-start cost, and broader workloads still need validation. +and faster training with populated compiler caches in individual runs. The first +[three-seed comparison](benchmarks.md#continuous-multi-seed-large-batch-profile) +confirms 26–28% lower peak memory but mixed speed results and 0.10–0.54 percentage +points lower accuracy. Quality parity, reliable speed gains, cold-start cost, and +broader workloads still need validation. Deliver quantization-aware training and post-training inference quantization as separate opt-in capabilities, recording actual weight/activation bit widths, diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index b39b67c..e1903bd 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -70,7 +70,8 @@ def test_summary_preserves_failures_and_unmeasured_fields(tmp_path): assert "No reports produced" in summarize(Path(tmp_path / "missing")) -def test_shared_harness_records_process_failures(tmp_path): +@pytest.mark.parametrize("mode", ["qt", "qt-large-batch"]) +def test_shared_harness_records_process_failures(tmp_path, mode): import os import shlex import subprocess @@ -84,18 +85,23 @@ def test_shared_harness_records_process_failures(tmp_path): runner.chmod(0o755) output = tmp_path / "reports" result = subprocess.run( - ["bash", "dev/check-benchmarks.sh", "qt", str(output)], - env={**os.environ, "BENCHMARK_PYTHON": str(runner)}, + ["bash", "dev/check-benchmarks.sh", mode, str(output)], + env={**os.environ, "BENCHMARK_PYTHON": str(runner), "BENCHMARK_DATA_ROOT": str(tmp_path)}, capture_output=True, text=True, timeout=30, ) assert result.returncode == 1 - for profile in ("synthetic-float", "synthetic-int8"): + profiles = ( + ("synthetic-float", "synthetic-int8") + if mode == "qt" + else tuple(f"mnist-large-batch-{precision}-seed{seed}" for seed in (42, 43, 44) for precision in ("float", "int8")) + ) + for profile in profiles: report = json.loads((output / profile / "report.json").read_text()) assert report["status"] == "failed" assert report["error"]["exit_code"] == 134 - assert report["quantized_training"] == profile.endswith("int8") + assert report["quantized_training"] == ("int8" in profile) assert "test_accuracy" not in report assert "requested" in (output / "summary.md").read_text() @@ -154,3 +160,26 @@ def test_summary_marks_legacy_cuda_readings_unverified(tmp_path): report["peak_cuda_memory_scope"] = "maximum across logger resets" (path / "report.json").write_text(json.dumps(report)) assert "64.00" in summarize(tmp_path).splitlines()[4] + + +def test_summary_later_epoch_median_requires_timing_scope(tmp_path): + report = { + "status": "completed", + "training_wall_seconds": 123, + "phase_measurements": [ + {"epoch": 0, "phase": "train", "seconds": 80}, + {"epoch": 1, "phase": "train", "seconds": 20}, + {"epoch": 2, "phase": "train", "seconds": 2}, + {"epoch": 3, "phase": "train", "seconds": 4}, + {"epoch": 3, "phase": "eval", "seconds": 10}, + ], + } + path = tmp_path / "report.json" + path.write_text(json.dumps(report)) + assert "| unverified | 123.00s |" in summarize(tmp_path) + report["phase_measurement_scope"] = "synchronized batch loop" + path.write_text(json.dumps(report)) + assert "| 3.000s | 123.00s |" in summarize(tmp_path) + report["phase_measurements"] = report["phase_measurements"][:2] + path.write_text(json.dumps(report)) + assert "| — | 123.00s |" in summarize(tmp_path) From a7007ac928dcfe2c9bef943fd6335370503b0a92 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 02:03:29 +0200 Subject: [PATCH 024/155] perf: fuse compiled INT8 weight requantization --- docs/benchmarks.md | 45 ++++++++++++++++ docs/quantized-training.md | 14 ++++- docs/roadmap.md | 6 ++- mini_trainer/modeling/_quantized_training.py | 28 +++++++++- mini_trainer/modeling/_quantized_update.py | 43 +++++++++++---- tests/test_quantized_training_model.py | 55 ++++++++++++++++++++ 6 files changed, 177 insertions(+), 14 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index c7e49f3..c479db6 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -412,3 +412,48 @@ for this local validation. The synthetic CUDA float/INT8 oracle pair was rerun with the retained kernels; both reached the required 100% accuracy after training and checkpoint reload. This verifies the simple task, not convergence equivalence on MNIST or Blair. + +### Functional fused requantization + +The compiled optimizer's floating-to-INT8 copy now uses a fused row kernel that +returns fresh codes and scales, followed by ordinary tensor copies. A first +in-place custom-operator experiment was rejected: generated SGD code updated +weight storage before calculating momentum that still depended on the old +weights. The existing floating-reference momentum regression caught the error. +The functional version passes that regression, independent stochastic-rounding +and RNG-replay checks, and checkpoint tests. It retains no floating master weight. + +The same three-seed profile was rerun, sequentially with no concurrent GPU tests. +Runtime source hash is +`ae6843b04efb8697fd2b830b3be729cbf61a061d3439abc8673c1d1d752d4542`; +the lock hash is unchanged. Paired manifests, held-out labels and paths match, +and all floating accuracies match the preceding run. + +| Seed | Execution | Test accuracy | Whole-run peak MiB | Median training epoch s, epochs 3–60 | Training wall s | +| --- | --- | ---: | ---: | ---: | ---: | +| 42 | Float | 93.06% | 254.26 | 0.076 | 30.77 | +| 42 | INT8 | 93.22% | 175.37 | 0.074 | 30.57 | +| 43 | Float | 92.76% | 249.13 | 0.074 | 31.54 | +| 43 | INT8 | 92.98% | 175.37 | 0.076 | 29.20 | +| 44 | Float | 92.74% | 249.13 | 0.071 | 32.40 | +| 44 | INT8 | 93.00% | 175.37 | 0.081 | 28.41 | + +Peak allocation falls another 4.9% relative to the preceding INT8 run, and is now +30–31% below float. Accuracy is 0.16–0.26 percentage points above the paired +floating runs; the changed rounding stream means this is not proof of an +intrinsically better learning algorithm. Whole training calls are shorter in +all three pairs, but later training phases are slower in two. Startup, validation +and logging costs remain part of the whole-run measurement, and caches were not +cleared. These results support retaining the memory improvement; they do not +establish a consistent compute speedup. Repeated timing and broader workloads +remain necessary. The synthetic CUDA pair again reached 100% oracle accuracy. + +Reports and predictions are retained locally under +`/tmp/mini-trainer-qt-large-batch-functional-requant`, with the oracle pair under +`/tmp/mini-trainer-qt-oracle-functional-requant`. The shared pipeline will exercise +the new backend without changing its recipes or acceptance thresholds. + +A separate eight-trial alternating dispatch probe compared the cached autotuner +with direct launches of its identical selected kernel. Skipping tuner bookkeeping +saved about 4 microseconds for the batch-128 square contraction, but less than +2% for the three larger training shapes. A second launch cache was not added. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 778883f..f7bd377 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -78,10 +78,20 @@ as does twelve-group AdamW, with hard failure enabled on compiler-cache fallback This establishes compilation compatibility, not a performance recommendation: the batch-128 dense MNIST comparison is slower and less accurate with QT than float. The [larger-batch profile](benchmarks.md#larger-batches-and-direct-collation) shows -lower memory and faster training after compiler caches are populated, but still -needs multi-seed quality validation. +lower memory and faster training in individual runs after compiler caches are +populated. The [three-seed requantization comparison](benchmarks.md#functional-fused-requantization) +shows lower memory and slightly higher accuracy than float, but later-phase +speed remains mixed and broader workload validation is still required. Compiled stochastic requantization can follow a different random trajectory from the eager row kernel, so a matching seed does not establish identical training. +Floating-to-INT8 copies now use a fused row-quantization kernel for matching CUDA +FP32/FP16/BF16 tensors with at most 16,384 columns. It returns fresh codes and +scales; ordinary tensor copies perform the final storage mutation so compiled +optimizer calculations that need the old weight remain correctly ordered. This +keeps no floating master weight, though transient floating updates still exist. +The operator is marked as seeded randomness, and compiled regressions check +independent rounding, generator replay, sub-code updates and optimizer state. +The random trajectory can differ from earlier compiled requantization. The earlier storage prototype's fake-tensor and dtype-cache failures are resolved by an explicit Triton kernel and custom-operator boundary; the storage kernel does not depend on Dynamo's per-frame variant cache. Model `--compile` remains a diff --git a/docs/roadmap.md b/docs/roadmap.md index 103260c..e4be429 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -117,8 +117,10 @@ profile](benchmarks.md#larger-batches-and-direct-collation) now shows lower memo and faster training with populated compiler caches in individual runs. The first [three-seed comparison](benchmarks.md#continuous-multi-seed-large-batch-profile) confirms 26–28% lower peak memory but mixed speed results and 0.10–0.54 percentage -points lower accuracy. Quality parity, reliable speed gains, cold-start cost, and -broader workloads still need validation. +points lower accuracy. [Functional fused requantization](benchmarks.md#functional-fused-requantization) +then reduced peak memory to 30–31% below float, with slightly higher accuracy in +all three pairs. Whole-run times improved, but later-phase speed remained mixed. +Reliable speed gains, cold-start cost, and broader workload validation remain open. Deliver quantization-aware training and post-training inference quantization as separate opt-in capabilities, recording actual weight/activation bit widths, diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index 35cd2c2..30295bc 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -11,7 +11,7 @@ from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise from ._quantized_matmul import scaled_int8_mm as _native_scaled_int8_mm -from ._quantized_update import update_int8_rows_ +from ._quantized_update import quantize_int8_rows, update_int8_rows_ # Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary # graph key. Include the backend and kernel implementations so changing hidden @@ -79,6 +79,32 @@ def add(func, types, args, kwargs): return func(*(value.dequantize() if isinstance(value, TrainingWeight) else value for value in args), **kwargs) +@TrainingWeight.implements(torch.ops.aten.copy_.default) +def copy_weight(func, types, args, kwargs): + destination, source = args[:2] + if not isinstance(destination, Int8QuantizedTrainingLinearWeight): + destination.copy_(source.dequantize(), **kwargs) + elif isinstance(source, Int8QuantizedTrainingLinearWeight): + destination.int_data.copy_(source.int_data, **kwargs) + destination.scale.copy_(source.scale, **kwargs) + elif ( + destination.device.type == "cuda" + and source.device == destination.device + and source.shape == destination.shape + and source.dtype == destination.dtype + and source.dtype in (torch.float32, torch.float16, torch.bfloat16) + and 0 < source.shape[1] <= 16384 + ): + codes, scales = quantize_int8_rows(source) + destination.int_data.copy_(codes, **kwargs) + destination.scale.copy_(scales, **kwargs) + else: + codes, scales = quantize_int8_rowwise(source, stochastic_rounding=True) + destination.int_data.copy_(codes, **kwargs) + destination.scale.copy_(scales, **kwargs) + return destination + + @TrainingWeight.implements(torch.ops.prims.fma.default) def fused_multiply_add(func, types, args, kwargs): # Dynamo lowers add_/addcdiv_ with tensor learning rates to fma + copy_. diff --git a/mini_trainer/modeling/_quantized_update.py b/mini_trainer/modeling/_quantized_update.py index 23cdad1..75b2c80 100644 --- a/mini_trainer/modeling/_quantized_update.py +++ b/mini_trainer/modeling/_quantized_update.py @@ -22,22 +22,26 @@ def _update_rows( denominator_row_stride, denominator_column_stride, DIVIDE: tl.constexpr, + COPY: tl.constexpr, BLOCK: tl.constexpr, ): row = tl.program_id(0) column = tl.arange(0, BLOCK) valid = column < columns code_offset = row * code_row_stride + column * code_column_stride - old_codes = tl.load(codes + code_offset, valid, 0).to(tl.float32) - old_scale = tl.load(scales + row * scale_stride) - dtype = old_scale.dtype - represented = (old_codes * old_scale.to(tl.float32)).to(dtype).to(tl.float32) + dtype = scales.dtype.element_ty change = tl.load(update + row * update_row_stride + column * update_column_stride, valid, 0).to(tl.float32) - if DIVIDE: - divisor = tl.load(denominator + row * denominator_row_stride + column * denominator_column_stride, valid, 1).to(tl.float32) - change = (change / divisor).to(dtype).to(tl.float32) - change = (change * alpha).to(dtype).to(tl.float32) - values = (represented + change).to(dtype).to(tl.float32) + if COPY: + values = change + else: + old_codes = tl.load(codes + code_offset, valid, 0).to(tl.float32) + old_scale = tl.load(scales + row * scale_stride) + represented = (old_codes * old_scale.to(tl.float32)).to(dtype).to(tl.float32) + if DIVIDE: + divisor = tl.load(denominator + row * denominator_row_stride + column * denominator_column_stride, valid, 1).to(tl.float32) + change = (change / divisor).to(dtype).to(tl.float32) + change = (change * alpha).to(dtype).to(tl.float32) + values = (represented + change).to(dtype).to(tl.float32) maximum = tl.max(tl.where(valid, tl.abs(values), 0), 0) next_scale = (maximum / 127).to(dtype) inverse = 1.0 / tl.maximum(next_scale.to(tl.float32), 1.0e-12) @@ -56,6 +60,10 @@ def update_int8_rows_( # Optimizer learning rates normally live on CPU. Pass their numeric value # as a runtime kernel argument, without a device allocation per weight. coefficient = alpha.item() + _launch(codes, scales, update, coefficient, denominator, copy=False) + + +def _launch(codes, scales, update, coefficient, denominator, *, copy): seed = torch.randint(0, 2**31, (), device=codes.device, dtype=torch.int64) divisor = update if denominator is None else denominator with torch.cuda.device(codes.device): @@ -72,6 +80,7 @@ def update_int8_rows_( *update.stride(), *divisor.stride(), DIVIDE=denominator is not None, + COPY=copy, BLOCK=triton.next_power_of_2(codes.shape[1]), enable_fp_fusion=False, ) @@ -80,3 +89,19 @@ def update_int8_rows_( @update_int8_rows_.register_fake def _fake_update_int8_rows_(codes, scales, update, alpha, denominator): return None + + +@torch.library.custom_op("mini_trainer::quantize_int8_rows", mutates_args=(), tags=(torch.Tag.nondeterministic_seeded,)) +def quantize_int8_rows(values: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: + """Requantize updates without hiding mutations from the graph scheduler.""" + codes = torch.empty(values.shape, device=values.device, dtype=torch.int8) + scales = torch.empty(values.shape[0], device=values.device, dtype=values.dtype) + _launch(codes, scales, values, 0.0, None, copy=True) + return codes, scales + + +@quantize_int8_rows.register_fake +def _fake_quantize_int8_rows(values): + return torch.empty(values.shape, device=values.device, dtype=torch.int8), torch.empty( + values.shape[0], device=values.device, dtype=values.dtype + ) diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index 4d9c91c..e0f3b5c 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -577,3 +577,58 @@ def test_cuda_compiled_stochastic_rounding_preserves_sub_code_updates(): if previous_codes is not None: assert not torch.equal(codes, previous_codes) previous_codes = codes.clone() + + +@pytest.mark.parametrize("dtype", [torch.float32, torch.float16, torch.bfloat16]) +@pytest.mark.parametrize("compiled", [False, True]) +def test_cuda_functional_requantization_preserves_independent_rounding(dtype, compiled): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify functional CUDA requantization") + from mini_trainer.modeling._quantized_update import quantize_int8_rows + + values = torch.full((128, 2048), 0.25, device="cuda", dtype=dtype)[:, ::2] + values[:, 0] = 127 + + def twice(values): + return quantize_int8_rows(values), quantize_int8_rows(values) + + run = torch.compile(twice, fullgraph=True) if compiled else twice + run(values) # Initialize compiler state before checking generator replay. + rng = torch.cuda.get_rng_state() + first, second = run(values) + after = torch.cuda.get_rng_state() + assert not torch.equal(rng, after) + assert not torch.equal(first[0], second[0]) + for codes, scales in (first, second): + assert codes.dtype == torch.int8 and scales.dtype == dtype + assert torch.all(scales == 1) and torch.all(codes[:, 0] == 127) + assert torch.all((codes[:, 1:] == 0) | (codes[:, 1:] == 1)) + assert abs(codes[:, 1:].float().mean().item() - 0.25) < 0.01 + torch.cuda.set_rng_state(rng) + replay = run(values) + for expected, actual in zip((first, second), replay, strict=True): + for left, right in zip(expected, actual, strict=True): + torch.testing.assert_close(left, right, rtol=0, atol=0) + + +def test_cuda_weight_copy_preserves_storage_aliases_and_source(): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify CUDA weight copy semantics") + from mini_trainer.modeling._quantized_training import TrainingWeight + + codes = torch.zeros((4, 14), device="cuda", dtype=torch.int8)[:, ::2] + scales = torch.ones(8, device="cuda")[::2] + weight = nn.Parameter(TrainingWeight(codes, scales)) + alias = weight.detach() + values = torch.randn((4, 14), device="cuda")[:, ::2] + before = values.clone() + version = weight._version + with torch.no_grad(): + assert weight.copy_(values) is weight + assert weight._version > version and alias._version == weight._version + assert weight.int_data.data_ptr() == codes.data_ptr() + assert weight.scale.data_ptr() == scales.data_ptr() + torch.testing.assert_close(values, before, rtol=0, atol=0) + torch.testing.assert_close(alias.dequantize(), weight.dequantize(), rtol=0, atol=0) + bound = values.abs().amax(1, keepdim=True) / 127 + 1e-6 + assert torch.all((weight.dequantize() - values).abs() <= bound) From d2f98cdf8c56ae589eb5de97862b300fb4d2cd90 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 02:15:12 +0200 Subject: [PATCH 025/155] test: isolate quantized optimizer compilation counters --- tests/test_quantized_training_model.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index e0f3b5c..5cb81b2 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -467,6 +467,9 @@ def test_cuda_compiled_optimizer_handles_many_quantized_groups(kind): if os.environ.get("RUN_CUDA_TESTS") != "1": pytest.skip("Set RUN_CUDA_TESTS=1 to verify many-group optimizer compilation") assert torch.cuda.is_available() + # Measure this optimizer's frames, independently of earlier tests' Dynamo + # caches/skip decisions. Never reset between groups or measured updates. + torch._dynamo.reset() weights = [nn.Parameter(TrainingWeight.from_float(torch.randn(64, 128, device="cuda"))) for _ in range(12)] groups = [{"params": [weight], "lr": 0.01 / (index + 1)} for index, weight in enumerate(weights)] optimizer = torch.optim.SGD(groups, momentum=0.9, weight_decay=0.1) if kind == "sgd" else torch.optim.AdamW(groups, foreach=False) From 052012903cf37676ff109b60a51f7a3f4936e5f6 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 02:15:12 +0200 Subject: [PATCH 026/155] perf: skip unused preprocessing when EMA is disabled --- docs/benchmarks.md | 42 +++++++++++++++++++++++++++++++++++ mini_trainer/trainer.py | 2 +- tests/test_optimizer_steps.py | 41 ++++++++++++++++++++++++++++++++++ 3 files changed, 84 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index c479db6..bcfcee4 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -457,3 +457,45 @@ A separate eight-trial alternating dispatch probe compared the cached autotuner with direct launches of its identical selected kernel. Skipping tuner bookkeeping saved about 4 microseconds for the batch-128 square contraction, but less than 2% for the three larger training shapes. A second launch cache was not added. + +### Avoiding disabled-EMA work + +The ordinary training loop previously preprocessed a second batch for the teacher +even with EMA disabled. The disabled teacher returned a CUDA zero scalar, adding +allocation and synchronization work despite contributing no loss. The loop now +skips that teacher call and retains a floating zero. Enabled teacher behavior, +loss checks, optimizer overflow handling and scheduler/update gating are unchanged. +EMA itself remains temporarily unsupported. + +Custom preprocessing functions now run once per training batch when EMA is +disabled. The former unused call could consume random numbers or cause other +side effects; eliminating those effects is intentional. A regression checks +call counts and exact student updates against direct SGD training. + +A warmed 31-batch INT8 training trace shows 62 CUDA stream synchronizations, +down from 124, and 1,519 runtime kernel launches, down from 1,705. Separate +unprofiled MNIST runs used the dense batch-128, 15-epoch, seed-42 recipe with +both model and optimizer compilation. Before/after order alternated across +two trials for float and INT8, with no concurrent GPU tests. + +| Precision | Trial | Accuracy, both paths | Median train epoch s before | After | Training wall s before | After | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| Float | 1 | 92.44% | 0.179 | 0.181 | 11.50 | 11.48 | +| Float | 2 | 92.44% | 0.196 | 0.187 | 12.79 | 12.14 | +| INT8 | 1 | 93.26% | 0.239 | 0.226 | 12.07 | 11.71 | +| INT8 | 2 | 93.26% | 0.275 | 0.255 | 13.26 | 12.99 | + +All paired prediction arrays, including every score, label and path, match +exactly. INT8 later-phase time falls by 5–7% in these trials; float timing is +mixed. Whole training calls are shorter in all pairs, but the smallest difference +is within ordinary timing noise. This removes measured overhead from both paths; +INT8 remains slower than float on this small-batch workload. + +The control ran from an isolated copy of `a7007ac`, with source hash +`ae6843b04efb8697fd2b830b3be729cbf61a061d3439abc8673c1d1d752d4542`. +The changed source hash is +`f6069758fdf01810bb343c1042b6b24fbae7422a71b1f1d065957f6949241635`. +Reports, logs and predictions are retained locally under +`/tmp/mini-trainer-disabled-ema-comparison`; traces are +`/tmp/mini-trainer-requant-epoch-trace.json` and +`/tmp/mini-trainer-disabled-ema-epoch-trace.json`. diff --git a/mini_trainer/trainer.py b/mini_trainer/trainer.py index 25e91bc..d875ffa 100644 --- a/mini_trainer/trainer.py +++ b/mini_trainer/trainer.py @@ -131,7 +131,7 @@ def train_one_epoch( # TODO: Add optional contrastive path # ctr_loss = contrastive_criterion() # If EMA is disabled ``distill_loss`` is ``0.0`` - distill_loss = model_ema.teach(step=step, input=preprocess(batch), student=logits) + distill_loss = model_ema.teach(step=step, input=preprocess(batch), student=logits) if model_ema else 0.0 reg = regularizer(model) if isinstance(loss, torch.Tensor) and loss.numel() == 1: diff --git a/tests/test_optimizer_steps.py b/tests/test_optimizer_steps.py index 3e05886..e6d12f8 100644 --- a/tests/test_optimizer_steps.py +++ b/tests/test_optimizer_steps.py @@ -15,6 +15,46 @@ KINDS = ["muon", "adamw", "sgd", "fused_adamw", "fused_sgd"] +def test_disabled_ema_does_not_preprocess_an_unused_teacher_batch(): + from mini_trainer.modeling.ema import EMATeacher + + torch.manual_seed(119) + model = torch.nn.Sequential(torch.nn.Flatten(), torch.nn.Linear(2, 2)) + reference = copy.deepcopy(model) + optimizer = torch.optim.SGD(model.parameters(), lr=0.1) + reference_optimizer = torch.optim.SGD(reference.parameters(), lr=0.1) + images = torch.tensor([[[[1.0, 2.0]]], [[[3.0, 4.0]]]]) + targets = torch.tensor([0, 1]) + loader = DataLoader(TensorDataset(images, targets), batch_size=1) + teacher = EMATeacher(enable=False, total_steps=2) + teacher.teach = Mock(side_effect=AssertionError("Disabled teacher must not be invoked")) + preprocess = Mock(side_effect=lambda batch: batch / 4) + logger = Mock() + logger.status.return_value = "disabled teacher regression" + criterion = torch.nn.CrossEntropyLoss() + train_one_epoch( + model, + teacher, + criterion, + optimizer, + torch.amp.GradScaler("cpu", enabled=False), + torch.optim.lr_scheduler.LambdaLR(optimizer, lambda _: 1), + loader, + 0, + logger, + preprocess=preprocess, + clip_grad_norm=None, + ) + for batch, target in loader: + reference_optimizer.zero_grad() + criterion(reference(batch / 4), target).backward() + reference_optimizer.step() + teacher.teach.assert_not_called() + assert preprocess.call_count == len(loader) + assert all(call.kwargs["distillation_loss"] == 0.0 for call in logger.consume.call_args_list) + assert_state_equal(model.state_dict(), reference.state_dict()) + + def make_optimizer(kind, parameters, lr=0.01): if kind == "muon": return MuonAuxAdamW([{"params": list(parameters), "name": "head"}], lr=lr, weight_decay=0.01) @@ -86,6 +126,7 @@ def overflow_once(grad): external_hook.remove() expected_steps = [6, 8] if scaled else [6, 7, 8] assert [call.args[0] for call in teacher.update_parameters.call_args_list] == expected_steps + assert teacher.teach.call_count == len(loader) assert all(call.args[1] is model for call in teacher.update_parameters.call_args_list) assert scheduler.last_epoch == len(expected_steps) assert optimizer.param_groups[0]["lr"] == pytest.approx(0.01 * 0.5 ** len(expected_steps)) From 40ed7d08849e80f1ddf78c8d5600adc0ace59523 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 02:32:49 +0200 Subject: [PATCH 027/155] perf: gather cached worker batches directly into shared memory --- dev/benchmarks/README.md | 19 +++++++++++++++ dev/benchmarks/loader.py | 46 ++++++++++++++++++++++++++++++------- docs/benchmarks.md | 43 ++++++++++++++++++++++++++++++++++ mini_trainer/data/io.py | 20 +++++++++++++++- tests/utils/test_loader.py | 47 ++++++++++++++++++++++++++++++++++++++ 5 files changed, 166 insertions(+), 9 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index dbecfdd..a2162e1 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -347,6 +347,25 @@ training/checkpoint/inference profile passed its 100% oracle gate with prefetch and reproduced the earlier QT held-out scores bit for bit; that establishes compatibility, not a training speedup. +### Worker batch assembly + +```bash +CUDA_VISIBLE_DEVICES='' OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.loader \ + --workers 1 --cache cpu --samples 512 --size 224 --batch-size 32 --repeats 7 +``` + +Use `--cache none` to probe uncached tensor assembly. These are synthetic tensor +readers, so neither mode measures image decoding. The probe verifies identical +batches, warms persistent spawn workers and then alternates scalar/batched passes. +Timing includes worker IPC but excludes startup, cache construction, preprocessing, +H2D and model compute. Worker count is explicit and defaults to zero; cached +loading does not automatically benefit from additional workers. + +Repository CPU-cache gathers now write directly into shared storage inside +workers. External collators retain their own allocation path. Main-process +pinning and CUDA-cache behavior remain separate. See the [measured worker +results](../../docs/benchmarks.md#shared-storage-for-cached-worker-batches). + ### Direct pinned gathering ```bash diff --git a/dev/benchmarks/loader.py b/dev/benchmarks/loader.py index 6f604d5..9e46ac6 100644 --- a/dev/benchmarks/loader.py +++ b/dev/benchmarks/loader.py @@ -1,4 +1,4 @@ -"""Measure scalar versus batched cached loading without changing data or sampling.""" +"""Measure scalar versus batched loading without changing data or sampling.""" import json import statistics @@ -25,19 +25,39 @@ def __getitem__(self, index): return self.dataset[index] -def run(samples=2048, size=64, batch_size=64, repeats=5, pin_batches=False): +class TensorReader: + """Picklable synthetic reader for spawn-worker batch assembly probes.""" + + def __init__(self, images): + self.images = images + + def __call__(self, item): + return self.images[item[0]], torch.tensor(item[0]) + + +def run(samples=2048, size=64, batch_size=64, repeats=5, pin_batches=False, workers=0, cache="cpu"): + if pin_batches and (workers or cache != "cpu"): + raise ValueError("Pinned gathering comparison requires CPU caching and zero workers.") generator = torch.Generator().manual_seed(42) images = torch.randint(0, 256, (samples, 3, size, size), dtype=torch.uint8, generator=generator) - dataset = LazyDataset(lambda item: (images[item[0]], torch.tensor(item[0])), (list(range(samples)),), cache="cpu") + reader = TensorReader(images) + dataset = LazyDataset(reader, (list(range(samples)),), cache=cache, cache_workers=0) + context = "spawn" if workers else None loaders = { - "scalar": DataLoader(ScalarFetch(dataset), batch_size=batch_size, num_workers=0), - "batched": get_dataloader(dataset, "val", batch_size, 0, False, torch.device("cpu")), + "scalar": DataLoader( + ScalarFetch(dataset), + batch_size=batch_size, + num_workers=workers, + persistent_workers=workers > 0, + multiprocessing_context=context, + ), + "batched": get_dataloader(dataset, "val", batch_size, workers, False, torch.device("cpu"), multiprocessing_context=context), } if pin_batches: if not torch.cuda.is_available(): raise RuntimeError("Pinned batch comparison requires an accessible CUDA device.") direct = LazyDataset( - lambda item: (images[item[0]], torch.tensor(item[0])), + reader, (list(range(samples)),), cache="cpu", cache_workers=0, @@ -65,11 +85,15 @@ def run(samples=2048, size=64, batch_size=64, repeats=5, pin_batches=False): "shape": [3, size, size], "dtype": "uint8", "batch_size": batch_size, - "workers": 0, + "workers": workers, + "cache": cache, "threads": torch.get_num_threads(), "identical_batches": True, "pin_batches": pin_batches, - "scope": "cached CPU loader iteration; excludes cache construction, preprocessing, H2D and model compute", + "scope": ( + "synthetic tensor loader iteration including worker IPC; " + "excludes worker startup, cache construction, image decoding, preprocessing, H2D and model compute" + ), "seconds": timings, "median_samples_per_second": {name: samples / statistics.median(values) for name, values in timings.items()}, "speedup": statistics.median(timings[baseline]) / statistics.median(timings[candidate]), @@ -83,9 +107,15 @@ def main(): parser.add_argument("--batch-size", type=int, default=64) parser.add_argument("--repeats", type=int, default=5) parser.add_argument("--pin-batches", action="store_true") + parser.add_argument("--workers", type=int, default=0) + parser.add_argument("--cache", choices=["cpu", "none"], default="cpu") args = parser.parse_args() if min(args.samples, args.size, args.batch_size, args.repeats) < 1: parser.error("All sizes and repeat counts must be positive") + if args.workers < 0: + parser.error("Worker count must be nonnegative") + if args.pin_batches and (args.workers or args.cache != "cpu"): + parser.error("--pin-batches requires --cache cpu and --workers 0") torch.set_num_threads(1) print(json.dumps(run(**vars(args)), indent=2)) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index bcfcee4..aeb8248 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -499,3 +499,46 @@ Reports, logs and predictions are retained locally under `/tmp/mini-trainer-disabled-ema-comparison`; traces are `/tmp/mini-trainer-requant-epoch-trace.json` and `/tmp/mini-trainer-disabled-ema-epoch-trace.json`. + +### Shared storage for cached worker batches + +CPU-cache gathers inside repository DataLoader workers now allocate their final +shared buffer before `index_select`. Previously, the gather produced an ordinary +tensor and the multiprocessing queue copied its storage again. This follows the +allocation approach used by PyTorch's worker collator. A spawned-worker regression +checks storage before it reaches the queue, along with exact values, partial +batches, retained-batch ownership and absence of CUDA initialization in workers. + +The optimization applies to the repository sampler/collator pair. External +collators retain their ordinary input allocation so they do not receive an extra +shared buffer. Worker-count defaults, parent-side pinning, CUDA caching and +main-process gathers are unchanged. + +Two alternating before/after trials used 512 synthetic uint8 RGB tensors at +224×224, batch size 32, one spawn worker per loader and one Torch thread. Each +trial reports the median of seven measured passes after warmup. The control used +an isolated copy of `0520129` with the same updated probe script. + +| Trial | Cached batched images/s before | After | Observed gain | +| --- | ---: | ---: | ---: | +| 1 | 13,425 | 14,395 | 7.2% | +| 2 | 13,226 | 15,367 | 16.2% | + +The separate scalar reference varied between 13,388 and 14,555 images/s across +these processes, so these are host-specific observations rather than a universal +speedup. The probe verifies identical batch tensors. Timing includes IPC but +excludes worker startup, cache construction, image decoding, H2D and model compute; +it does not establish an end-to-end training or inference gain. + +An uncached shared-stack candidate was also tested. It changed throughput from +11,668 to 11,503 and from 12,129 to 11,298 images/s in the two trials. That path +was removed; uncached assembly retains its previous implementation. Fewer copies +did not establish a speed benefit, and changing where work occurs may affect +overlap between decoding/assembly and the queue's sharing work. Real image-decoding +and uncached throughput remain separate optimization targets. In particular, +`get_inference_dataloader` currently streams uncached data; this cached-worker +change does not establish a speed gain for that helper. + +The shared [loader probe](../dev/benchmarks/README.md#worker-batch-assembly) now +accepts explicit `--workers` and `--cache` options with picklable readers. +Detailed trial results are retained under `/tmp/mini-trainer-shared-batch-final`. diff --git a/mini_trainer/data/io.py b/mini_trainer/data/io.py index 9e45410..31bd36d 100644 --- a/mini_trainer/data/io.py +++ b/mini_trainer/data/io.py @@ -344,6 +344,13 @@ class _DirectBatchIndices(list): """Index list for the repository collator, which accepts stacked tensors.""" +def _shared_batch_buffer(template, shape): + # Match PyTorch's worker collator: allocate the final IPC storage directly, + # rather than stack/gather locally and copy it when the queue shares it. + storage = template._typed_storage()._new_shared(math.prod(shape), device=template.device) + return template.new(storage).resize_(shape) + + class _FetchedBatch(list): """Sample views for standard collators, with the already-stacked batch attached.""" @@ -550,7 +557,18 @@ def __getitems__(self, indices): for tensor in tensors ) else: - data = tuple(tensor.index_select(0, index) for tensor in tensors) + worker = isinstance(indices, _DirectBatchIndices) and torch.utils.data.get_worker_info() is not None + data = tuple( + torch.index_select( + tensor, + 0, + index, + out=_shared_batch_buffer(tensor, (len(index), *tensor.shape[1:])) + if worker and tensor.device.type == "cpu" and not tensor.requires_grad + else None, + ) + for tensor in tensors + ) data = data[0] if self._ram_was_single_tensor else data else: data = self[indices] diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index ea366b3..dfae4d5 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -10,6 +10,16 @@ from mini_trainer.data.loader import PathLabelProcessor, get_dataset_dataloader, get_inference_dataloader +def _assert_shared_worker_batch(samples): + assert torch.utils.data.get_worker_info() is not None + assert not torch.cuda.is_initialized() + values = samples.data if isinstance(samples.data, (tuple, list)) else (samples.data,) + # Check before the multiprocessing queue can copy ordinary storage into + # shared memory; checking only the parent's received batch would miss it. + assert all(value.is_shared() for value in values) + return data_loader._collate_batch(samples) + + @pytest.fixture def metadata(tmp_path): paths = [] @@ -381,6 +391,43 @@ def test_pinned_cache_gather_never_pins_inside_worker(monkeypatch): assert result.tolist() == [2, 0] +@pytest.mark.parametrize("labels", [False, True]) +def test_cached_worker_batches_are_built_in_shared_storage(metadata, labels): + if labels: + datasets, loaders = get_dataset_dataloader( + metadata, resize_size=4, modes=("val",), cache="CPU", cache_workers=0, batch_size=2, num_workers=0 + ) + dataset, base = datasets[0], loaders[0] + else: + uncached, _ = get_inference_dataloader(metadata["path"], resize_size=4, batch_size=2, num_workers=0) + dataset = data_io.LazyDataset(uncached.func, uncached.items, cache="CPU", cache_workers=0) + base = data_loader.get_dataloader(dataset, "val", 2, 0, False, torch.device("cpu")) + loader = torch.utils.data.DataLoader( + dataset, + batch_sampler=base.batch_sampler, + collate_fn=_assert_shared_worker_batch, + num_workers=1, + multiprocessing_context="spawn", + ) + batches = list(loader) + expected = [data_loader._collate_batch(dataset.__getitems__(indices)) for indices in base.batch_sampler] + torch.testing.assert_close(batches, expected, rtol=0, atol=0) + first = batches[0][0] if labels else batches[0] + preserved = first.clone() + last = batches[-1][0] if labels else batches[-1] + last.zero_() + torch.testing.assert_close(first, preserved, rtol=0, atol=0) + + +def test_external_collation_does_not_allocate_an_extra_shared_batch(monkeypatch): + dataset = data_io.LazyDataset(lambda item: torch.tensor(item[0]), ([0, 1],), cache="CPU", cache_workers=0) + monkeypatch.setattr(torch.utils.data, "get_worker_info", lambda: object()) + ordinary = dataset.__getitems__([0, 1]).data + direct = dataset.__getitems__(data_io._DirectBatchIndices([0, 1])).data + assert not ordinary.is_shared() and direct.is_shared() + torch.testing.assert_close(ordinary, direct, rtol=0, atol=0) + + def test_cached_repository_loader_does_not_unpack_sample_views(): from torch.utils._python_dispatch import TorchDispatchMode From 6f10d5af23e86bd11377e3ee966f567c89e93a0f Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 02:54:39 +0200 Subject: [PATCH 028/155] perf: resize streamed uint8 images without float intermediates --- dev/benchmarks/README.md | 20 ++++++++ dev/benchmarks/reader.py | 94 ++++++++++++++++++++++++++++++++++++++ docs/benchmarks.md | 39 ++++++++++++++++ mini_trainer/data/io.py | 35 +++++++++++++- tests/utils/test_loader.py | 42 +++++++++++++++++ 5 files changed, 229 insertions(+), 1 deletion(-) create mode 100644 dev/benchmarks/reader.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index a2162e1..2272843 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -347,6 +347,26 @@ training/checkpoint/inference profile passed its 100% oracle gate with prefetch and reproduced the earlier QT held-out scores bit for bit; that establishes compatibility, not a training speedup. +### Streaming image-reader comparison + +```bash +CUDA_VISIBLE_DEVICES='' OMP_NUM_THREADS=1 .venv/bin/python -m dev.benchmarks.reader \ + --data-root examples/blair/test --samples 128 --size 224 --batch-size 16 \ + --workers 1 --repeats 7 > /tmp/blair-reader.json +``` + +Repeat with `--workers 0` or `--data-root examples/mnist/test`. This uses the +actual streaming inference loader and compares the former torchvision resize +path with the current reader. Every batch must match exactly before timing. +JSON includes relative file names, content hashes, dependency versions and every +timing sample. Paths are sorted and the first requested number of JPEG/PNG files +is used; files are never modified. + +Timing includes file reads, decoding, nearest resize, assembly and worker IPC. +The equivalence pass warms the filesystem cache and persistent workers, so these +are not cold-disk or worker-startup measurements. H2D and model compute are excluded. +See the [reader findings](../../docs/benchmarks.md#uint8-nearest-resize-in-the-streaming-reader). + ### Worker batch assembly ```bash diff --git a/dev/benchmarks/reader.py b/dev/benchmarks/reader.py new file mode 100644 index 0000000..b847df7 --- /dev/null +++ b/dev/benchmarks/reader.py @@ -0,0 +1,94 @@ +"""Compare streaming image readers with exact batches and file provenance.""" + +import hashlib +import json +import statistics +import time +from argparse import ArgumentParser +from importlib.metadata import version +from pathlib import Path + +import torch +from torchvision.io import ImageReadMode, decode_image +from torchvision.transforms import functional as transforms + +from mini_trainer.data.io import ReadAndResize +from mini_trainer.data.loader import get_inference_dataloader + + +class TorchvisionReader(ReadAndResize): + """The former decode/float-resize/convert path, retained as a reference.""" + + def __call__(self, path): + if not isinstance(path, str): + path = path[0] + image = decode_image(path, mode=ImageReadMode.RGB, apply_exif_orientation=False) + image = transforms.resize(image, [self.h, self.w], interpolation=self.interp, antialias=self.antialias) + return (image if image.dtype == self.dtype else self.converter(image)).to(self.device) + + +def run(data_root, samples=128, size=224, batch_size=16, repeats=7, workers=0): + root = Path(data_root).resolve() + paths = sorted(path for path in root.rglob("*") if path.suffix.lower() in (".jpg", ".jpeg", ".png") and path.is_file())[:samples] + if not paths: + raise ValueError("No JPEG/PNG images found under the supplied data root.") + manifest = [{"path": str(path.relative_to(root)), "sha256": hashlib.sha256(path.read_bytes()).hexdigest()} for path in paths] + loaders = {} + for name in ("torchvision", "gather"): + dataset, loaders[name] = get_inference_dataloader( + list(map(str, paths)), + resize_size=size, + batch_size=batch_size, + num_workers=workers, + multiprocessing_context="spawn" if workers else None, + ) + if name == "torchvision": + dataset.func = TorchvisionReader((size, size), torch.device("cpu"), torch.uint8) + for expected, actual in zip(loaders["torchvision"], loaders["gather"], strict=True): + torch.testing.assert_close(actual, expected, rtol=0, atol=0) + timings = {name: [] for name in loaders} + for trial in range(repeats + 1): + for name in list(loaders) if trial % 2 else list(reversed(loaders)): + started = time.perf_counter() + count = sum(len(batch) for batch in loaders[name]) + elapsed = time.perf_counter() - started + assert count == len(paths) + if trial: + timings[name].append(elapsed) + return { + "files": manifest, + "samples": len(paths), + "shape": [3, size, size], + "dtype": "uint8", + "batch_size": batch_size, + "workers": workers, + "threads": torch.get_num_threads(), + "versions": {name: version(name) for name in ("torch", "torchvision", "numpy")}, + "identical_batches": True, + "scope": ( + "uncached image loading, decoding, nearest resize, batch assembly and IPC with a warm filesystem cache; " + "excludes worker startup, H2D and model compute" + ), + "seconds": timings, + "median_samples_per_second": {name: len(paths) / statistics.median(values) for name, values in timings.items()}, + "speedup": statistics.median(timings["torchvision"]) / statistics.median(timings["gather"]), + } + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--data-root", type=Path, required=True) + parser.add_argument("--samples", type=int, default=128) + parser.add_argument("--size", type=int, default=224) + parser.add_argument("--batch-size", type=int, default=16) + parser.add_argument("--repeats", type=int, default=7) + parser.add_argument("--workers", type=int, default=0) + args = parser.parse_args() + if min(args.samples, args.size, args.batch_size, args.repeats) < 1 or args.workers < 0: + parser.error("Sizes/repeats must be positive and workers nonnegative.") + torch.set_num_threads(1) + print(json.dumps(run(**vars(args)), indent=2)) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index aeb8248..34de54f 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -542,3 +542,42 @@ change does not establish a speed gain for that helper. The shared [loader probe](../dev/benchmarks/README.md#worker-batch-assembly) now accepts explicit `--workers` and `--cache` options with picklable readers. Detailed trial results are retained under `/tmp/mini-trainer-shared-batch-final`. + +### Uint8 nearest resize in the streaming reader + +The default nearest-neighbor reader now gathers decoded uint8 image rows and +columns directly, avoiding torchvision's temporary float image and conversion +back to bytes. It preserves the legacy nearest coordinate mapping, unchanged-size +identity, output layout and dtype conversion. Other interpolation modes and +outputs larger than 4096 pixels on either axis retain the existing path. Coordinate +caching is limited to 32 one-dimensional arrays, at most about 1 MiB per process. + +The replacement was checked against legacy float resizing across thousands of +source/target length combinations and randomized images, including non-square and +singleton dimensions. Layout checks caught and corrected a singleton-stride +difference before measurement. Direct uint8 `interpolate` was also measured but +was slower; that candidate was not adopted. + +The [file-backed reader probe](../dev/benchmarks/README.md#streaming-image-reader-comparison) +uses the actual uncached inference loader. Each profile reads the first 128 sorted +JPEG/PNG paths under the dataset's test directory, resizes to 224×224, and batches +16 images. One Torch thread is used, with either zero workers or one spawn worker. +Every batch matches the former reader exactly. The table reports medians of seven +alternating measured passes after equivalence checks and warmup. + +| Dataset | Workers | Former reader images/s | Gather reader images/s | Observed gain | +| --- | ---: | ---: | ---: | ---: | +| MNIST | 0 | 4,071 | 4,889 | 20.1% | +| MNIST | 1 | 3,123 | 3,685 | 18.0% | +| Blair | 0 | 2,890 | 3,197 | 10.6% | +| Blair | 1 | 2,049 | 2,653 | 29.5% | + +These measurements include decoding, resizing, batch assembly and IPC with a warm +filesystem cache. They exclude worker startup, H2D and model compute, and do not +establish model-inference or training speedups of the same size. The MNIST profile +deliberately resizes to 224×224; it is not the 28×28 MNIST training recipe. Worker +defaults are unchanged, and adding a worker was slower on both datasets here. + +Results, per-file content hashes, versions and timing samples are retained under +`/tmp/mini-trainer-reader-comparison`. The standalone CPU command can be reused in +continuous validation wherever the corresponding dataset is available. diff --git a/mini_trainer/data/io.py b/mini_trainer/data/io.py index 31bd36d..d4cad28 100644 --- a/mini_trainer/data/io.py +++ b/mini_trainer/data/io.py @@ -8,6 +8,7 @@ from concurrent.futures import ThreadPoolExecutor from contextlib import closing from enum import Enum +from functools import lru_cache from itertools import batched from typing import Any, TypeVar, cast @@ -111,6 +112,28 @@ def _pil_to_torch_interp(interp: int) -> InterpolationMode: return m.get(interp, InterpolationMode.BILINEAR) # type: ignore +@lru_cache(maxsize=32) +def _nearest_indices(source_size, target_size): + # Match ATen's legacy nearest mapping: float32 scale, floor, then clamp. + # Cache only one-dimensional coordinates, never full image-sized maps. + scale = np.float32(source_size) / np.float32(target_size) + indices = (np.arange(target_size, dtype=np.float32) * scale).astype(np.int64) + np.minimum(indices, source_size - 1, out=indices) + return torch.from_numpy(indices) + + +def _resize_nearest_uint8(image, height, width): + if image.shape[-2:] == (height, width): + return image + rows = _nearest_indices(image.shape[1], height) + columns = _nearest_indices(image.shape[2], width) + # Decoders return interleaved RGB storage. Gather whole rows/pixels before + # restoring the contiguous CHW layout produced by torchvision resizing. + return ( + image.permute(1, 2, 0).index_select(0, rows).index_select(1, columns).permute(2, 0, 1).clone(memory_format=torch.contiguous_format) + ) + + class ReadAndResize: """Callable class to read and resize images from paths.""" @@ -137,7 +160,17 @@ def __call__(self, path: str | tuple[str, ...]) -> torch.Tensor: except Exception as e: e.add_note(f"Image path: {path}") raise - img = TF.resize(img, size=[self.h, self.w], interpolation=self.interp, antialias=self.antialias) + if ( + self.interp == InterpolationMode.NEAREST + and img.device.type == "cpu" + and img.dtype == torch.uint8 + and img.ndim == 3 + and 0 < min(self.h, self.w) + and max(self.h, self.w) <= 4096 + ): + img = _resize_nearest_uint8(img, self.h, self.w) + else: + img = TF.resize(img, size=[self.h, self.w], interpolation=self.interp, antialias=self.antialias) if img.dtype != self.dtype: img = self.converter(img) return img.to(self.device) diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index dfae4d5..ec454b3 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -20,6 +20,48 @@ def _assert_shared_worker_batch(samples): return data_loader._collate_batch(samples) +def test_nearest_coordinates_match_legacy_float_resize(): + sources = [*range(1, 64), 127, 255, 257, 1023, 2049] + targets = [*range(1, 64), 127, 224, 257, 4096] + for source in sources: + values = torch.arange(source, dtype=torch.float32).reshape(1, 1, source, 1) + for target in targets: + expected = torch.nn.functional.interpolate(values, size=(target, 1), mode="nearest").flatten().long() + torch.testing.assert_close(data_io._nearest_indices(source, target), expected, rtol=0, atol=0) + + +@pytest.mark.parametrize("source", [(3, 7), (17, 31), (128, 256)]) +@pytest.mark.parametrize("target", [(1, 1), (1, 19), (11, 1), (11, 19), (224, 224)]) +@pytest.mark.parametrize("dtype", [torch.uint8, torch.float32]) +def test_reader_nearest_resize_matches_pixels_layout_and_conversion(monkeypatch, source, target, dtype): + from torchvision.transforms import InterpolationMode + from torchvision.transforms import functional as transforms + + image = torch.randint(0, 256, (*source, 3), dtype=torch.uint8, generator=torch.Generator().manual_seed(127)).permute(2, 0, 1) + monkeypatch.setattr(data_io, "decode_image", lambda *args, **kwargs: image) + reader = data_io.make_read_and_resize_fn((target[1], target[0]), torch.device("cpu"), dtype) + expected = transforms.resize(image, list(target), interpolation=InterpolationMode.NEAREST) + if dtype != torch.uint8: + expected = data_io.make_convert_dtype(dtype)(expected) + torch.testing.assert_close(reader("test-image"), expected, rtol=0, atol=0, check_stride=True) + + +def test_reader_identity_and_non_nearest_paths_remain_unchanged(monkeypatch): + from torchvision.transforms import InterpolationMode + from torchvision.transforms import functional as transforms + + image = torch.randint(0, 256, (7, 11, 3), dtype=torch.uint8).permute(2, 0, 1) + monkeypatch.setattr(data_io, "decode_image", lambda *args, **kwargs: image) + identity = data_io.make_read_and_resize_fn((11, 7), torch.device("cpu"), torch.uint8) + assert identity("test-image") is image + bilinear = data_io.make_read_and_resize_fn((19, 5), torch.device("cpu"), torch.uint8, interpolation=Image.Resampling.BILINEAR) + expected = transforms.resize(image, [5, 19], interpolation=InterpolationMode.BILINEAR) + torch.testing.assert_close(bilinear("test-image"), expected, rtol=0, atol=0, check_stride=True) + wide = data_io.make_read_and_resize_fn((4097, 1), torch.device("cpu"), torch.uint8) + expected = transforms.resize(image, [1, 4097], interpolation=InterpolationMode.NEAREST) + torch.testing.assert_close(wide("test-image"), expected, rtol=0, atol=0, check_stride=True) + + @pytest.fixture def metadata(tmp_path): paths = [] From 95ca909a2d0214d690f41448512495f66b9f160e Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 03:19:39 +0200 Subject: [PATCH 029/155] feat: expose model compilation modes for quantized training Record explicit modes in benchmark reports and preserve existing defaults. Keep normalized weights beside their Linear consumer so embedding publication does not split the INT8 autograd boundary. Validate portable checkpoints and eager resume across normalized heads and optimizer modes; compare embedding-loss gradients in FP32. Full CPU suite and focused CUDA regressions pass. --- dev/README.md | 15 +++++++++++ dev/benchmarks/run.py | 8 ++++++ docs/quantized-training.md | 19 ++++++++++++++ mini_trainer/modeling/classifier.py | 5 +++- mini_trainer/train.py | 9 +++++++ mini_trainer/trainer.py | 7 ++++- mini_trainer/training/compilation.py | 15 ++++++++++- tests/test_benchmark_synthetic.py | 24 ++++++++++++++++- tests/test_checkpoint_contract.py | 14 ++++++++-- tests/test_quantized_training_model.py | 36 +++++++++++++++++++++++++- 10 files changed, 145 insertions(+), 7 deletions(-) diff --git a/dev/README.md b/dev/README.md index 2c40a6a..cd30512 100644 --- a/dev/README.md +++ b/dev/README.md @@ -212,6 +212,21 @@ this change and 2.22 million afterward. This isolates cached iteration; larger images, decoding, transfer and model compute change the overall benefit. See the [integrated measurements](../docs/benchmarks.md#larger-batches-and-direct-collation). +### Model compilation + +`mt_train --compile --compile-mode reduce-overhead` selects a PyTorch model +compilation mode. The Python training entry points accept `compile_mode`, and +`dev.benchmarks.run` accepts the same CLI flag and records it in success and +failure reports. An explicit mode requires `--compile`; omitting it preserves +ordinary `torch.compile(model)` behavior. Optimizer compilation remains separate. + +Supported modes are `default`, `reduce-overhead`, `max-autotune`, and +`max-autotune-no-cudagraphs`. PyTorch's CUDA graph modes can reduce launch overhead +for eligible graphs, but capture is not guaranteed and workspace caching can +increase memory. Measure both float and INT8 with the same mode, including +compilation time, later training phases, peak allocation and held-out quality. +See the [PyTorch compilation modes](https://docs.pytorch.org/docs/2.12/generated/torch.compile.html). + ### Optimizer compilation `mt_train --compile-optimizer` opts into compiling optimizer updates independently diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index b283299..1664115 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -21,6 +21,7 @@ from mini_trainer.modeling import Classifier from mini_trainer.train import main as train from mini_trainer.training import MuonAuxAdamW +from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options from .datasets import prepare_real from .models import NoAugmentationBuilder @@ -61,7 +62,9 @@ def run( model_profile: str = "default", optimizer: str = "muon", learning_rate: float | None = None, + compile_mode: str | None = None, ): + model_compile_options(compile, compile_mode) if model_profile not in ("default", "dense") or optimizer not in ("muon", "adamw", "sgd"): raise ValueError("Unknown model or optimizer profile.") if learning_rate is None: @@ -149,6 +152,7 @@ def run( ema=False, quantized_training=quantized_training, compile=compile, + compile_mode=compile_mode, compile_optimizer=compile_optimizer, model_builder_kwargs={ "model_type": model_type, @@ -245,6 +249,7 @@ def run( "momentum": 0.9 if optimizer == "sgd" else None, "quantization_recipe": quantization_recipe, "compile": compile, + "compile_mode": compile_mode, "compile_optimizer": compile_optimizer, "hidden": hidden, "batch_size": batch_size, @@ -321,6 +326,7 @@ def main(): parser.add_argument("--quantized-training", action="store_true") parser.add_argument("--cuda-prefetch", action="store_true") parser.add_argument("--compile", action="store_true") + parser.add_argument("--compile-mode", choices=MODEL_COMPILE_MODES) parser.add_argument("--compile-optimizer", action="store_true") parser.add_argument("--model-profile", choices=["default", "dense"], default="default") parser.add_argument("--optimizer", choices=["muon", "adamw", "sgd"], default="muon") @@ -355,6 +361,7 @@ def main(): cuda_prefetch=args.cuda_prefetch, quantized_training=args.quantized_training, compile=args.compile, + compile_mode=args.compile_mode, compile_optimizer=args.compile_optimizer, hidden=args.hidden, batch_size=args.batch_size, @@ -377,6 +384,7 @@ def main(): "cuda_prefetch": args.cuda_prefetch, "quantized_training": args.quantized_training, "compile": args.compile, + "compile_mode": args.compile_mode, "compile_optimizer": args.compile_optimizer, "hidden": args.hidden, "batch_size": args.batch_size, diff --git a/docs/quantized-training.md b/docs/quantized-training.md index f7bd377..41299f4 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -124,6 +124,25 @@ ONNX export, checkpoint averaging, DDP/FSDP, quantized activation normalization and integer convolution training are not established for this path. Distributed training and EMA are rejected by the training entry point. +## Model compilation modes + +Model compilation accepts `--compile --compile-mode reduce-overhead` (or +`compile=True, compile_mode="reduce-overhead"` in Python). This is opt-in and +independent of optimizer compilation. The benchmark runner records the selected +mode. See [compilation guidance](../dev/README.md#model-compilation) for the other +modes and measurement requirements; selecting a mode does not establish a speedup. + +Normalized heads resolve their parametrized weights after publishing embeddings, +keeping weight normalization and the integer Linear operation in the same graph. +Previously the embedding publication could split them and cause AOTAutograd to +expect an INT8 tensor subclass gradient where the backward supplies a float tensor. +This affected both ordinary and CUDA graph compilation. Regression coverage now +includes normalized and ordinary heads, eager/default/reduce-overhead execution, +AMP training, compiled/eager optimizers, checkpoint loading and eager resume. +A separate FP32 test compares input and parameter gradients with an embedding +auxiliary loss. AMP compilation can change rounding and INT8 activation bins; +these checks do not promise bitwise-identical eager and compiled trajectories. + ## Evidence and remaining work The [developer probes](../dev/benchmarks/README.md#quantized-training-and-loader-performance) diff --git a/mini_trainer/modeling/classifier.py b/mini_trainer/modeling/classifier.py index 5fbc589..b20b429 100644 --- a/mini_trainer/modeling/classifier.py +++ b/mini_trainer/modeling/classifier.py @@ -232,10 +232,13 @@ def preclassification(self, x: torch.Tensor) -> torch.Tensor: return self.batch_norm(x) def forward(self, x: torch.Tensor) -> torch.Tensor: - weight, bias = self._weight_bias() embeddings = self.preclassification(x) if EmbeddingContext.active(): EmbeddingContext.set(embeddings) + # Resolve parametrized weights beside their consumer. Publishing the + # embeddings can break a compiled graph; carrying a normalized INT8 + # weight across that boundary gives AOTAutograd the wrong tangent type. + weight, bias = self._weight_bias() if self.normalized: return cosine_to_zscore(F.linear(embeddings, weight=weight), self.preclassification_size) + bias else: diff --git a/mini_trainer/train.py b/mini_trainer/train.py index 68458c2..bb37d37 100644 --- a/mini_trainer/train.py +++ b/mini_trainer/train.py @@ -23,6 +23,7 @@ from mini_trainer.modeling import average_checkpoints, classification_module from mini_trainer.trainer import train from mini_trainer.training import MuonAuxAdamW +from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options from mini_trainer.utils import ( broadcast_from_master, ddp_train_wrapper, @@ -76,6 +77,7 @@ def main( # noqa: D417 logger_builder_kwargs: dict[str, Any] = {"verbose": False}, ddp_info: dict | None = None, compile_optimizer: bool = False, + compile_mode: str | None = None, ): """Train a classifier. @@ -111,6 +113,7 @@ def main( # noqa: D417 See ``mini_trainer.builders.BaseBuilder`` for details. """ orig_args = locals() + model_compile_options(compile, compile_mode) # Prepare state if seed is not None: random.seed(seed) @@ -324,6 +327,7 @@ def main( # noqa: D417 output_dir=weight_output_dir, weight_store_rate=5, compile=compile, + compile_mode=compile_mode, ) del train_loader @@ -589,6 +593,11 @@ def cli(description="Train a classifier", **extra_kwargs): # noqa: D103 required=False, help="Compile the model using torch.compile for faster execution (default=False).", ) + cfg_args.add_argument( + "--compile-mode", + choices=MODEL_COMPILE_MODES, + help="Model compilation mode; requires --compile. Optimizer compilation is configured separately.", + ) cfg_args.add_argument( "--dtype", type=str, diff --git a/mini_trainer/trainer.py b/mini_trainer/trainer.py index d875ffa..1aad88f 100644 --- a/mini_trainer/trainer.py +++ b/mini_trainer/trainer.py @@ -19,6 +19,7 @@ from mini_trainer.builders import EMATeacher from mini_trainer.logging import MultiLogger from mini_trainer.modeling import EmbeddingContext, SupervisionContext +from mini_trainer.training.compilation import model_compile_options from mini_trainer.utils import ( TERMINAL_WIDTH, TQDM, @@ -277,6 +278,7 @@ def train( output_dir: str | None = None, weight_store_rate: int | None = None, compile: bool = False, + compile_mode: str | None = None, **kwargs, ): """Full training loop across epochs with periodic evaluation and checkpointing. @@ -300,10 +302,13 @@ def train( dtype: AMP/autocast data type for forward/eval passes. output_dir: If provided, checkpoints are written here. weight_store_rate: Store a snapshot every ``weight_store_rate`` epochs if set. + compile: Compile the model with PyTorch. + compile_mode: Optional PyTorch model compilation mode; requires compile=True. **kwargs: Forwarded to lower-level helpers. """ log = get_logger() + compile_options = model_compile_options(compile, compile_mode) log.info("Start training") start_time = time.time() @@ -326,7 +331,7 @@ def train( # Disable DDPOptimizer graph splitting — it deadlocks on models with # find_unused_parameters or custom scatter ops, causing NCCL timeouts. torch._dynamo.config.optimize_ddp = False - model = torch.compile(model) + model = torch.compile(model, **compile_options) best_eval_metric = -float("inf") best_epoch = -1 diff --git a/mini_trainer/training/compilation.py b/mini_trainer/training/compilation.py index ec5471e..45b9c93 100644 --- a/mini_trainer/training/compilation.py +++ b/mini_trainer/training/compilation.py @@ -1,4 +1,4 @@ -"""Opt-in optimizer compilation with stable learning-rate inputs.""" +"""Opt-in model and optimizer compilation.""" from functools import wraps @@ -6,6 +6,19 @@ from .muon import Muon, MuonAuxAdamW +MODEL_COMPILE_MODES = ("default", "reduce-overhead", "max-autotune", "max-autotune-no-cudagraphs") + + +def model_compile_options(enabled: bool, mode: str | None) -> dict: + """Validate explicit model modes while preserving ordinary compile defaults.""" + if mode is None: + return {} + if not enabled: + raise ValueError("compile_mode requires compile=True (--compile).") + if mode not in MODEL_COMPILE_MODES: + raise ValueError(f"Unknown model compile mode {mode!r}; choose from {MODEL_COMPILE_MODES}.") + return {"mode": mode} + def _tensor_learning_rates(optimizer): for group in optimizer.param_groups: diff --git a/tests/test_benchmark_synthetic.py b/tests/test_benchmark_synthetic.py index dc522bc..b5b95f6 100644 --- a/tests/test_benchmark_synthetic.py +++ b/tests/test_benchmark_synthetic.py @@ -67,7 +67,11 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): from dev.benchmarks.run import main - monkeypatch.setattr(sys, "argv", ["benchmark", "--output", str(tmp_path / "failed"), "--device", "cuda:0"]) + monkeypatch.setattr( + sys, + "argv", + ["benchmark", "--output", str(tmp_path / "failed"), "--device", "cuda:0", "--compile", "--compile-mode", "reduce-overhead"], + ) monkeypatch.setattr(torch.cuda, "is_available", lambda: False) monkeypatch.setattr(torch, "set_num_threads", lambda threads: None) monkeypatch.setattr(torch, "use_deterministic_algorithms", lambda enabled: None) @@ -76,6 +80,7 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): report = json.loads((tmp_path / "failed/report.json").read_text()) assert report["status"] == "failed" assert report["device"] == "cuda:0" + assert report["compile_mode"] == "reduce-overhead" assert report["error"]["type"] == "RuntimeError" assert "test_accuracy" not in report @@ -88,3 +93,20 @@ def test_qt_profile_requires_cuda_before_creating_output(tmp_path): with pytest.raises(ValueError, match="require CUDA"): run(tmp_path / "qt", quantized_training=True) assert not (tmp_path / "qt").exists() + + +def test_compile_mode_requires_compilation_before_creating_outputs(tmp_path): + import pytest + + from dev.benchmarks.run import run + from mini_trainer.train import main + from mini_trainer.training.compilation import model_compile_options + + for mode in ("reduce-overhead", "invalid"): + with pytest.raises(ValueError, match="requires compile=True"): + run(tmp_path / "benchmark", compile_mode=mode) + with pytest.raises(ValueError, match="requires compile=True"): + main(input=str(tmp_path / "missing"), output=str(tmp_path / "train"), compile_mode=mode) + with pytest.raises(ValueError, match="Unknown model compile mode"): + model_compile_options(True, "invalid") + assert not list(tmp_path.iterdir()) diff --git a/tests/test_checkpoint_contract.py b/tests/test_checkpoint_contract.py index e344049..c83f43a 100644 --- a/tests/test_checkpoint_contract.py +++ b/tests/test_checkpoint_contract.py @@ -191,11 +191,19 @@ def test_ema_status_warning_only_when_enabled(): EMATeacher(enable=True, total_steps=1, model=torch.nn.Linear(2, 2)) -def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatch): +@pytest.mark.parametrize("mode", [None, "reduce-overhead"]) +def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatch, mode): # Exercise Dynamo's real OptimizedModule wrapper without testing CPU # Inductor performance. Its state_dict adds _orig_mod unless unwrapped. compile_model = torch.compile - monkeypatch.setattr(torch, "compile", lambda model: compile_model(model, backend="eager")) + model_options = [] + + def compile_with_eager_backend(model, **options): + if isinstance(model, torch.nn.Module): + model_options.append(options) + return compile_model(model, backend="eager") + + monkeypatch.setattr(torch, "compile", compile_with_eager_backend) for label in ("class_a", "class_b"): (tmp_path / "data" / label).mkdir(parents=True) args = { @@ -207,6 +215,7 @@ def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatc "dtype": "float32", "seed": 42, "compile": True, + "compile_mode": mode, "compile_optimizer": True, "ema": False, "builder": DeterministicBuilder, @@ -216,6 +225,7 @@ def test_compiled_checkpoint_uses_portable_keys_and_resumes(tmp_path, monkeypatc "logger_builder_kwargs": {"verbose": False}, } train_module.main(**args) + assert model_options == [{} if mode is None else {"mode": mode}] path = tmp_path / "compiled/weights/checkpoint_last.pth" state = torch.load(path, weights_only=True) assert not any(key.startswith("_orig_mod.") for key in state["model"]) diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index 5cb81b2..1437e25 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -136,7 +136,8 @@ def test_mixed_muon_adamw_updates_and_counter(): @pytest.mark.parametrize("normalized", [False, True]) @pytest.mark.parametrize("compiled_optimizer", [False, True]) -def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, compiled_optimizer): +@pytest.mark.parametrize("compile_mode", [None, "default", "reduce-overhead"]) +def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, compiled_optimizer, compile_mode): from mini_trainer.modeling import Classifier from mini_trainer.modeling._quantized_training import TrainingWeight from mini_trainer.train import main @@ -155,6 +156,8 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, comp "dtype": "float16", "quantized_training": True, "compile_optimizer": compiled_optimizer, + "compile": compile_mode is not None, + "compile_mode": compile_mode, "seed": 42, "builder": DeterministicBuilder, "model_builder_kwargs": {"model_type": TinyMockModel(), "hidden": False, "droprate": 0, "normalized": normalized}, @@ -177,6 +180,8 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, comp # plumbing, not identical stochastic continuation with a changed schedule. args["epochs"] = 2 args["compile_optimizer"] = False + args["compile"] = False + args["compile_mode"] = None args["name"] = "resumed" args["checkpoint"] = str(tmp_path / "quantized/weights/checkpoint_last.pth") args["model_builder_kwargs"]["model_type"] = TinyMockModel() @@ -635,3 +640,32 @@ def test_cuda_weight_copy_preserves_storage_aliases_and_source(): torch.testing.assert_close(alias.dequantize(), weight.dequantize(), rtol=0, atol=0) bound = values.abs().amax(1, keepdim=True) / 127 + 1e-6 assert torch.all((weight.dequantize() - values).abs() <= bound) + + +@pytest.mark.parametrize("mode", ["default", "reduce-overhead"]) +def test_normalized_compilation_preserves_embedding_loss_gradients(mode): + from mini_trainer.modeling import Classifier, EmbeddingContext + + torch.manual_seed(19) + model = Classifier(64, 4, hidden=False, normalized=True).to(cuda()) + prepare_quantized_training(model) + forward = torch.compile(model, mode=mode) + data = torch.randn(8, 64, device=cuda()) + target = torch.arange(8, device=cuda()) % 4 + results = [] + for call in (model, forward): + model.zero_grad(set_to_none=True) + inputs = data.clone().requires_grad_() + # Isolate graph-boundary gradients from AMP fusion rounding, which can + # move normalized activations across an INT8 quantization threshold. + with EmbeddingContext(): + scores = call(inputs) + embeddings = EmbeddingContext.get() + assert embeddings is not None and embeddings.requires_grad + loss = torch.nn.functional.cross_entropy(scores, target) + embeddings[:, 0].sum() * 0.1 + loss.backward() + results.append( + (scores.detach().clone(), inputs.grad.clone(), [None if p.grad is None else p.grad.clone() for p in model.parameters()]) + ) + assert model.linear.parametrizations.weight.original1.grad.norm() > 0 + torch.testing.assert_close(results[0], results[1], rtol=1e-3, atol=1e-4) From 8639c246079d2a11efd31b7cdbeb6ee77c113c53 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 03:26:34 +0200 Subject: [PATCH 030/155] bench: retain paired CUDA graph training results in CI Add a separate three-seed MNIST profile with matched float and INT8 compilation modes, preserving ordinary compilation baselines. Retain failures, arguments, summaries and artifacts. Record 31% lower INT8 peak allocation and shorter whole training calls, while explicitly retaining the mixed later-epoch speed result. Validate harness failure paths and both synthetic oracles. --- .github/workflows/benchmarks.yml | 6 +++- dev/benchmarks/README.md | 16 ++++++++++ dev/check-benchmarks.sh | 14 ++++++--- docs/benchmarks.md | 51 ++++++++++++++++++++++++++++++++ tests/test_benchmark_datasets.py | 11 +++++-- 5 files changed, 91 insertions(+), 7 deletions(-) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index 555fe78..41f2519 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -101,6 +101,9 @@ jobs: - name: Multi-seed large-batch quantized training if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() run: bash dev/check-benchmarks.sh qt-large-batch benchmark-qt-large-batch + - name: Multi-seed CUDA graph quantized training + if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() + run: bash dev/check-benchmarks.sh qt-cudagraphs benchmark-qt-cudagraphs - name: Synthetic GPU precision profiles run: bash dev/check-benchmarks.sh gpu benchmark-gpu - name: Real-data progression @@ -110,7 +113,7 @@ jobs: if: always() run: | python3 -m dev.benchmarks.summarize benchmark-gpu >> "$GITHUB_STEP_SUMMARY" - for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense benchmark-qt-large-batch; do + for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense benchmark-qt-large-batch benchmark-qt-cudagraphs; do if [[ -d "$qt_results" ]]; then python3 -m dev.benchmarks.summarize "$qt_results" >> "$GITHUB_STEP_SUMMARY" fi @@ -132,6 +135,7 @@ jobs: benchmark-qt-real/ benchmark-qt-dense/ benchmark-qt-large-batch/ + benchmark-qt-cudagraphs/ !benchmark-qt/**/data/**/*.png !benchmark-gpu/**/data/**/*.png - name: Remove the disposable GPU environment diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 2272843..cc6450b 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -464,3 +464,19 @@ preprocessing and batch logging, excludes validation/figures/checkpoints, and may still contain later compilation. Compiler caches are not cleared between runs: neither column establishes fresh-cache performance. Real-data completion still has no quality acceptance threshold; inspect accuracy for every seed. + +### CUDA graph comparison + +```bash +CUDA_VISIBLE_DEVICES=0 BENCHMARK_DATA_ROOT=examples \ + bash dev/check-benchmarks.sh qt-cudagraphs /tmp/qt-cudagraphs +``` + +This repeats the three-seed, batch-512, 60-epoch MNIST comparison with +`--compile-mode reduce-overhead` for both float and INT8. All other settings and +alternating execution order match `qt-large-batch`. Both profiles remain in the +optional QT plus real-data GPU workflow, with summaries and retained artifacts. +Explicit modes are recorded in JSON reports; process failures retain the exact +arguments. Neither a mode flag nor a completed real-data run guarantees CUDA +graph replay, convergence equivalence or a speedup. Compare all timings and +memory against float under the same mode, rather than an older float baseline. diff --git a/dev/check-benchmarks.sh b/dev/check-benchmarks.sh index 3c4e1f8..52c781b 100644 --- a/dev/check-benchmarks.sh +++ b/dev/check-benchmarks.sh @@ -9,7 +9,7 @@ if [[ -e "$results" ]]; then echo 'Results directory must be new.' >&2 exit 2 fi -case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense or qt-large-batch.' >&2; exit 2 ;; esac +case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch|qt-cudagraphs) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense, qt-large-batch or qt-cudagraphs.' >&2; exit 2 ;; esac mkdir -p -- "$results" status=0 run_profile() { @@ -56,7 +56,13 @@ elif [[ "$mode" == qt-dense ]]; then --model-profile dense --optimizer sgd --learning-rate 0.3 --epochs 15 --batch-size 128 --compile \ --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --allow-nondeterministic "${quantization[@]}" done -elif [[ "$mode" == qt-large-batch ]]; then +elif [[ "$mode" == qt-large-batch || "$mode" == qt-cudagraphs ]]; then + compile_mode=() + profile=mnist-large-batch + if [[ "$mode" == qt-cudagraphs ]]; then + compile_mode=(--compile-mode reduce-overhead) + profile=mnist-cudagraphs + fi : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/}" for seed in 42 43 44; do precisions=(float int8) @@ -64,9 +70,9 @@ elif [[ "$mode" == qt-large-batch ]]; then for precision in "${precisions[@]}"; do quantization=() if [[ "$precision" == int8 ]]; then quantization=(--quantized-training); fi - run_profile "mnist-large-batch-$precision-seed$seed" --dataset mnist --data-root "$BENCHMARK_DATA_ROOT/mnist" \ + run_profile "$profile-$precision-seed$seed" --dataset mnist --data-root "$BENCHMARK_DATA_ROOT/mnist" \ --seed "$seed" --model-profile dense --optimizer sgd --learning-rate 0.3 --epochs 60 --batch-size 512 \ - --compile --compile-optimizer --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 \ + --compile "${compile_mode[@]}" --compile-optimizer --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 \ --allow-nondeterministic "${quantization[@]}" done done diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 34de54f..462e3e5 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -581,3 +581,54 @@ defaults are unchanged, and adding a worker was slower on both datasets here. Results, per-file content hashes, versions and timing samples are retained under `/tmp/mini-trainer-reader-comparison`. The standalone CPU command can be reused in continuous validation wherever the corresponding dataset is available. + +### Explicit CUDA graph compilation + +The opt-in `qt-cudagraphs` profile repeats the batch-512, 60-epoch, three-seed +comparison with `--compile-mode reduce-overhead` on **both** float and INT8. +The preceding profiles retain their ordinary compilation settings. The optional +QT plus real-data workflow runs this additional profile and retains its summaries +and artifacts. Reproduction is documented in the +[benchmark runner guide](../dev/benchmarks/README.md#cuda-graph-comparison). + +A separate warmed, batch-128 MNIST trace recorded 186 `cudaGraphLaunch` calls +across its third training epoch, verifying actual replay with the retained opaque +integer matrix operation. This is distinct from the rejected graph-visible +matrix-operator experiment above. The trace is local at +`/tmp/mini-trainer-cudagraph-epoch-trace.json`; profiling timings are not used below. + +The matched runs below used implementation commit `95ca909`, the same RTX 3080 Ti +Laptop GPU and dependency versions as the preceding comparisons, one CPU and +compiler thread, sequential GPU execution, and alternating float/INT8 order. +Runtime source hash: +`ee66fe6f88471cbbb798366f1ef056478ec569ca79140208d8928149cbbcaca2`. +Lock hash: +`43ad5c7df81212b3bcd536220201f8888666723507c317e8c86dd538fb595745`. +All six reports record the explicit mode; paired manifests, test labels and paths +match. Compiler caches were not cleared, and CUDA nondeterminism was permitted. + +| Seed | Execution | Test accuracy | Whole-run peak MiB | Median training epoch s, epochs 3–60 | Training wall s | +| --- | --- | ---: | ---: | ---: | ---: | +| 42 | Float | 93.06% | 199.15 | 0.069 | 31.44 | +| 42 | INT8 | 93.22% | 138.31 | 0.066 | 28.52 | +| 43 | Float | 92.76% | 199.15 | 0.071 | 33.90 | +| 43 | INT8 | 92.98% | 138.31 | 0.066 | 27.78 | +| 44 | Float | 92.74% | 199.15 | 0.064 | 33.74 | +| 44 | INT8 | 93.00% | 138.31 | 0.070 | 27.21 | + +INT8 uses 30.5% less peak allocation than the equally configured float model. +Whole training calls are 9–19% shorter. Later training phases are 4–8% faster for +two seeds and 9% slower for the third: a consistent steady-state speedup is still +unproven. All six accuracies match their preceding ordinary-compilation runs; +this is not a statistical quality-equivalence result. Validation, logging, +checkpointing and compilation remain part of whole-call time. No cold-start, +universal model-speed or cross-device claim follows from these runs. + +Reports and predictions are local under `/tmp/mini-trainer-mnist-cudagraph-pairs`. +The new continuous profile preserves the same configuration and all unfavorable +results alongside the earlier profiles, rather than replacing their baselines. + +The synthetic oracle also reached 100% after checkpoint reload for float and INT8 +with the same model mode and compiled MuonAuxAdamW updates. Those reports are at +`/tmp/mini-trainer-cudagraph-oracle`. This checks the simple oracle task, not +normalized-head convergence or general CUDA graph eligibility. diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index e1903bd..e5c70ec 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -70,7 +70,7 @@ def test_summary_preserves_failures_and_unmeasured_fields(tmp_path): assert "No reports produced" in summarize(Path(tmp_path / "missing")) -@pytest.mark.parametrize("mode", ["qt", "qt-large-batch"]) +@pytest.mark.parametrize("mode", ["qt", "qt-large-batch", "qt-cudagraphs"]) def test_shared_harness_records_process_failures(tmp_path, mode): import os import shlex @@ -92,10 +92,11 @@ def test_shared_harness_records_process_failures(tmp_path, mode): timeout=30, ) assert result.returncode == 1 + prefix = "mnist-cudagraphs" if mode == "qt-cudagraphs" else "mnist-large-batch" profiles = ( ("synthetic-float", "synthetic-int8") if mode == "qt" - else tuple(f"mnist-large-batch-{precision}-seed{seed}" for seed in (42, 43, 44) for precision in ("float", "int8")) + else tuple(f"{prefix}-{precision}-seed{seed}" for seed in (42, 43, 44) for precision in ("float", "int8")) ) for profile in profiles: report = json.loads((output / profile / "report.json").read_text()) @@ -103,6 +104,12 @@ def test_shared_harness_records_process_failures(tmp_path, mode): assert report["error"]["exit_code"] == 134 assert report["quantized_training"] == ("int8" in profile) assert "test_accuracy" not in report + arguments = report["arguments"] + if mode == "qt-cudagraphs": + assert arguments[arguments.index("--compile-mode") + 1] == "reduce-overhead" + assert "--compile" in arguments and "--compile-optimizer" in arguments + else: + assert "--compile-mode" not in arguments assert "requested" in (output / "summary.md").read_text() From b476340373977894acb94893669226a42431a563 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 03:43:58 +0200 Subject: [PATCH 031/155] perf: keep embedding publication inside compiled model graphs Store published embeddings in traceable dictionary state while preserving context activation, gradients and cleanup. Compare full-graph and eager gradients and prevent repeated compilation after lazy cache initialization. Validate the full CPU suite and all 56 CUDA INT8 model tests. Record matched MNIST improvements and the CUDA graph memory, startup and accuracy tradeoffs without claiming universal speed gains. --- docs/benchmarks.md | 53 ++++++++++++++++++++++++ docs/quantized-training.md | 26 +++++++----- mini_trainer/modeling/classifier.py | 4 +- mini_trainer/modeling/context.py | 10 +++-- tests/test_embedding_context.py | 56 ++++++++++++++++++++++++++ tests/test_quantized_training_model.py | 2 +- 6 files changed, 133 insertions(+), 18 deletions(-) create mode 100644 tests/test_embedding_context.py diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 462e3e5..7348ebf 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -632,3 +632,56 @@ The synthetic oracle also reached 100% after checkpoint reload for float and INT with the same model mode and compiled MuonAuxAdamW updates. Those reports are at `/tmp/mini-trainer-cudagraph-oracle`. This checks the simple oracle task, not normalized-head convergence or general CUDA graph eligibility. + +### Embedding publication without graph breaks + +`EmbeddingContext.set` now publishes its tensor through dictionary state instead +of assigning a tensor-valued class attribute. This lets Dynamo carry the side +effect through a full model graph, preserving the embedding's gradient path. +Activation, retrieval, nesting checks and exception cleanup retain their existing +interface. Tests compare eager/full-graph input and parameter gradients, including +an embedding auxiliary loss, and verify stable compilation after the classifier's +initial lazy-cache guard settles. + +A batch-128 MNIST training trace fell from three compiled forward/backward calls +per batch to one. CUDA graph launches fell from 186 to 62 in the third epoch. +The traces are at `/tmp/mini-trainer-cudagraph-epoch-trace.json` and +`/tmp/mini-trainer-embedding-epoch-trace.json`. Profiling timing is not used below. + +The following unprofiled comparisons used a snapshot of `8639c24` as the control, +the same GPU/dependencies as above, dense MNIST, batch 128, 15 epochs, SGD, +FP16 AMP, model and optimizer compilation, CPU cache, zero cache workers, and +one CPU/compiler thread. Float ran before then after; INT8 ran after then before. +Validation tests had finished before these runs. Dataset manifests, held-out +labels and paths match within every pair. Source hashes are: + +- Before: `ee66fe6f88471cbbb798366f1ef056478ec569ca79140208d8928149cbbcaca2`. +- After: `ccbfc2dfe1e69eae41f52f7b11bfb5c84641cb167ded827b588cad2196f8a023`. + +The lock hash remains `43ad5c7df81212b3bcd536220201f8888666723507c317e8c86dd538fb595745`. +Compiler caches were retained and CUDA nondeterminism was permitted. + +| Mode | Seed | Precision | Accuracy before → after | Peak MiB before → after | Median train epoch 3–15 s before → after | Training wall s before → after | +| --- | --- | --- | ---: | ---: | ---: | ---: | +| Ordinary | 42 | Float | 92.44% → 93.12% | 236.30 → 236.05 | 0.197 → 0.161 | 13.21 → 13.71 | +| Ordinary | 42 | INT8 | 93.26% → 92.84% | 168.95 → 168.95 | 0.274 → 0.228 | 13.85 → 14.94 | +| reduce-overhead | 42 | Float | 92.44% → 93.12% | 193.55 → 217.41 | 0.155 → 0.138 | 11.62 → 11.25 | +| reduce-overhead | 42 | INT8 | 93.26% → 92.84% | 136.96 → 149.60 | 0.174 → 0.184 | 11.62 → 12.38 | +| reduce-overhead | 43 | Float | 92.70% → 93.30% | 193.55 → 217.41 | 0.159 → 0.137 | 12.55 → 11.56 | +| reduce-overhead | 43 | INT8 | 93.10% → 93.02% | 136.96 → 149.60 | 0.179 → 0.155 | 12.19 → 11.74 | + +Ordinary compilation improves later-phase time by about 18% for float and 17% +for INT8 in this pair, with essentially unchanged allocation. Whole-call times +increase, so this is not a startup improvement. Under CUDA graph compilation, +float improves in both seeds, while INT8 is mixed and peak allocation rises by +9% for INT8 and 12% for float. INT8 remains slower than equally configured float +on this small-batch model. The change removes a verified graph break; it does not +establish consistent QT speed superiority or universally lower compiled memory. + +AMP fusion changes the training trajectory: float accuracy rises by 0.60–0.68 +percentage points, while INT8 falls by 0.08–0.42 points. These few short runs +neither establish an intrinsic quality improvement nor prove equivalence. +Reports and predictions are at `/tmp/mini-trainer-embedding-default-pairs` and +`/tmp/mini-trainer-embedding-clean-pairs`. An earlier exploratory comparison at +`/tmp/mini-trainer-embedding-pairs` is retained separately; a short CPU validation +check overlapped that run, so its timings are excluded from this table. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 41299f4..b2d57a0 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -132,16 +132,22 @@ independent of optimizer compilation. The benchmark runner records the selected mode. See [compilation guidance](../dev/README.md#model-compilation) for the other modes and measurement requirements; selecting a mode does not establish a speedup. -Normalized heads resolve their parametrized weights after publishing embeddings, -keeping weight normalization and the integer Linear operation in the same graph. -Previously the embedding publication could split them and cause AOTAutograd to -expect an INT8 tensor subclass gradient where the backward supplies a float tensor. -This affected both ordinary and CUDA graph compilation. Regression coverage now -includes normalized and ordinary heads, eager/default/reduce-overhead execution, -AMP training, compiled/eager optimizers, checkpoint loading and eager resume. -A separate FP32 test compares input and parameter gradients with an embedding -auxiliary loss. AMP compilation can change rounding and INT8 activation bins; -these checks do not promise bitwise-identical eager and compiled trajectories. +Embedding publication uses mutable dictionary state so Dynamo can preserve the +side effect without splitting the model graph. Normalized weights remain beside +their Linear consumer. Previously the class-attribute assignment forced graph +breaks and could make AOTAutograd expect an INT8 tensor subclass gradient where +backward supplies a float tensor. Public context activation, retrieval, nesting +checks and cleanup remain unchanged; this is still a shared context, not thread- +or task-local state. + +Regression coverage includes normalized and ordinary heads, +eager/default/reduce-overhead execution, AMP training, compiled/eager optimizers, +checkpoint loading and eager resume. Full-graph FP32 tests compare input and +parameter gradients with an embedding auxiliary loss and check stable compilation +after lazy classifier metadata initializes. AMP fusion can change rounding and +INT8 activation bins; these checks do not promise bitwise-identical eager and +compiled trajectories. Fewer graphs do not guarantee lower peak memory or shorter +whole training calls; see the [measured tradeoffs](benchmarks.md#embedding-publication-without-graph-breaks). ## Evidence and remaining work diff --git a/mini_trainer/modeling/classifier.py b/mini_trainer/modeling/classifier.py index b20b429..349f8c5 100644 --- a/mini_trainer/modeling/classifier.py +++ b/mini_trainer/modeling/classifier.py @@ -235,9 +235,7 @@ def forward(self, x: torch.Tensor) -> torch.Tensor: embeddings = self.preclassification(x) if EmbeddingContext.active(): EmbeddingContext.set(embeddings) - # Resolve parametrized weights beside their consumer. Publishing the - # embeddings can break a compiled graph; carrying a normalized INT8 - # weight across that boundary gives AOTAutograd the wrong tangent type. + # Resolve parametrized weights beside their Linear consumer. weight, bias = self._weight_bias() if self.normalized: return cosine_to_zscore(F.linear(embeddings, weight=weight), self.preclassification_size) + bias diff --git a/mini_trainer/modeling/context.py b/mini_trainer/modeling/context.py index 47dd258..77f6c0d 100644 --- a/mini_trainer/modeling/context.py +++ b/mini_trainer/modeling/context.py @@ -31,20 +31,22 @@ def __exit__(self, exc_type, exc_val, exc_tb): class EmbeddingContext: """Used for passing embeddings from the classification module to the criterion (or elsewhere).""" - _embeddings: torch.Tensor | None = None + # Dynamo can carry dictionary mutations out of a compiled graph. Assigning + # a Tensor to a class attribute instead forces a graph break at publication. + _state: dict[str, torch.Tensor | None] = {"embeddings": None} _active: bool = False @classmethod def set(cls, embeddings): - cls._embeddings = embeddings + cls._state["embeddings"] = embeddings @classmethod def get(cls): - return cls._embeddings + return cls._state["embeddings"] @classmethod def clear(cls): - cls._embeddings = None + cls._state["embeddings"] = None cls._active = False @classmethod diff --git a/tests/test_embedding_context.py b/tests/test_embedding_context.py new file mode 100644 index 0000000..800ad6a --- /dev/null +++ b/tests/test_embedding_context.py @@ -0,0 +1,56 @@ +"""Embedding publication must retain gradients without splitting model graphs.""" + +import copy + +import pytest +import torch +from torch._dynamo.testing import CompileCounterWithBackend + +from mini_trainer.modeling import Classifier, EmbeddingContext + + +@pytest.mark.parametrize("normalized", [False, True]) +def test_fullgraph_embedding_publication_matches_eager_gradients(normalized): + torch.manual_seed(27) + reference = Classifier(16, 4, hidden=8, droprate=0, normalized=normalized) + model = copy.deepcopy(reference) + counter = CompileCounterWithBackend("aot_eager") + compiled = torch.compile(model, backend=counter, fullgraph=True) + warmed_frames = None + for iteration in range(4): + data = torch.randn(8, 16) + results = [] + for call, parameters in ((reference, reference.parameters()), (compiled, model.parameters())): + call.zero_grad(set_to_none=True) + inputs = data.clone().requires_grad_() + with EmbeddingContext(): + scores = call(inputs) + embeddings = EmbeddingContext.get() + assert EmbeddingContext.active() and embeddings is not None and embeddings.requires_grad + (scores.square().mean() + embeddings[:, 0].sum()).backward() + results.append( + (scores.detach().clone(), inputs.grad.clone(), [None if p.grad is None else p.grad.clone() for p in parameters]) + ) + assert not EmbeddingContext.active() and EmbeddingContext.get() is None + torch.testing.assert_close(results[0], results[1]) + # The classifier initializes its lazy weight-cache metadata on the + # first forward. Once that guard settles, publishing fresh embeddings + # must neither split the graph (fullgraph=True) nor recompile it. + if iteration == 1: + warmed_frames = counter.frame_count + elif iteration > 1: + assert counter.frame_count == warmed_frames + + +def test_embedding_context_rejects_nesting_and_cleans_up_after_errors(): + embeddings = torch.randn(2, 3, requires_grad=True) + with pytest.raises(ValueError, match="body failure"): + with EmbeddingContext(): + EmbeddingContext.set(embeddings) + assert EmbeddingContext.get() is embeddings + with pytest.raises(RuntimeError, match="already active"): + with EmbeddingContext(): + pass + assert EmbeddingContext.active() and EmbeddingContext.get() is embeddings + raise ValueError("body failure") + assert not EmbeddingContext.active() and EmbeddingContext.get() is None diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index 1437e25..03540d9 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -649,7 +649,7 @@ def test_normalized_compilation_preserves_embedding_loss_gradients(mode): torch.manual_seed(19) model = Classifier(64, 4, hidden=False, normalized=True).to(cuda()) prepare_quantized_training(model) - forward = torch.compile(model, mode=mode) + forward = torch.compile(model, mode=mode, fullgraph=True) data = torch.randn(8, 64, device=cuda()) target = torch.arange(8, device=cuda()) % 4 results = [] From 7675a2254b7befcfbad049a4598c87f273b4432d Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 03:48:29 +0200 Subject: [PATCH 032/155] docs: record hierarchical QT graph compilation tradeoffs Retain matched Blair species/parent accuracy, allocation, steady-phase and total timings. Distinguish faster later epochs from slower INT8 startup and document the limited Linear coverage and passing synthetic oracles. --- docs/benchmarks.md | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 7348ebf..133f5ae 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -685,3 +685,35 @@ Reports and predictions are at `/tmp/mini-trainer-embedding-default-pairs` and `/tmp/mini-trainer-embedding-clean-pairs`. An earlier exploratory comparison at `/tmp/mini-trainer-embedding-pairs` is retained separately; a short CPU validation check overlapped that run, so its timings are excluded from this table. + +#### Hierarchical Blair check + +The same source comparison also ran Blair with 3,704 training images, 912 +validation images and the fixed 1,161-image test set. Both sides used TinyConv, +a 64-feature hidden layer, the normalized `HierarchicalClassifier`, the same +reviewed two-level class specification, MuonAuxAdamW, batch 32, five epochs, +FP16 AMP, CPU cache, and model/optimizer compilation with `reduce-overhead`. +INT8 coverage is `fc.hidden` and `fc.linear`; the convolutions remain floating +point. This checks the two-level aggregation path, not every hierarchical head +variant. Manifest, test labels and paths match, and all saved scores are finite. + +| Precision | Revision | Species accuracy | Parent accuracy | Peak MiB | Median train epoch 3–5 s | Training wall s | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| Float | Before | 65.72% | 79.33% | 72.64 | 0.721 | 22.27 | +| Float | After | 64.86% | 78.12% | 66.38 | 0.679 | 18.47 | +| INT8 | Before | 65.81% | 81.05% | 55.96 | 1.060 | 22.95 | +| INT8 | After | 67.53% | 82.95% | 54.62 | 0.845 | 46.54 | + +Later-phase time improves by 6% for float and 20% for INT8, and peak allocation +falls for both. INT8 still takes longer per epoch than float. Its whole-call +time also rises sharply: the first training phase takes 35.60 seconds after the +change versus 13.21 before, consistent with substantial first-use compilation +cost. These caches were not cleared, so this is not a controlled cold-start +comparison. This short single-seed run establishes functional coverage and a +measured steady-phase improvement; it does not establish convergence equivalence, +universal memory behavior, or faster total QT training on Blair. Raw reports, +predictions and the class specification are at `/tmp/mini-trainer-embedding-blair`. + +After this change, both synthetic CUDA oracle runs again reached 100% with model +and MuonAuxAdamW optimizer compilation, `reduce-overhead`, FP16 AMP, training and +checkpoint reload. Reports are at `/tmp/mini-trainer-embedding-oracle`. From a089f4deda68b82b2f0a72cf606c24b8834000dd Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 08:12:47 +0200 Subject: [PATCH 033/155] feat: add opt-in CUDA graph replay for optimizer updates Keep scheduler and AMP gating outside captured updates and preserve Muon counters. Store rates on CUDA for replay and retain explicit rate precision across checkpoint restoration, including the first eager update. Expose INT8 update arithmetic during AOT tracing to avoid opaque CPU-scalar graph boundaries. Validate optimizer replay, fused rate restrictions, update arithmetic and portable resume without changing defaults. --- dev/README.md | 32 +++++ dev/benchmarks/run.py | 9 +- docs/quantized-training.md | 12 +- mini_trainer/modeling/_quantized_training.py | 6 +- mini_trainer/train.py | 11 +- mini_trainer/training/compilation.py | 77 ++++++++++-- tests/test_benchmark_synthetic.py | 42 ++++++- tests/test_optimizer_steps.py | 119 +++++++++++++++++-- tests/test_quantized_training_model.py | 63 +++++++++- 9 files changed, 346 insertions(+), 25 deletions(-) diff --git a/dev/README.md b/dev/README.md index cd30512..c8391f1 100644 --- a/dev/README.md +++ b/dev/README.md @@ -253,3 +253,35 @@ variants, and MuonAuxAdamW on CUDA, including overflow skips, scheduler changes, parameter updates and optimizer state. This does not establish support for every custom optimizer or a real-workload INT8 speedup. Quantized update dispatch and its performance remain a separate validation boundary. + +### Optimizer CUDA graphs + +`mt_train --compile-optimizer --optimizer-cudagraphs` additionally requests CUDA +graph replay for optimizer updates. It requires parameters on one CUDA device +and the default Inductor backend. It is independent of model compilation; +`--compile --compile-mode reduce-overhead` can enable model graphs as well. +The benchmark runner accepts and records the same optimizer option. + +For custom training, call `compile_optimizer(optimizer, cudagraphs=True)` after +scheduler construction and checkpoint restoration. The first real update still +initializes state eagerly, under the trainer's AMP gating. Subsequent updates use +device-resident learning rates. Scheduler updates, overflow decisions and Muon's +outer step counter remain outside graph capture. Changing the graph setting of +an already compiled optimizer is rejected rather than silently ignored. + +Numeric rates retain float64 precision; explicitly supplied tensor rates retain +their dtype. Checkpoints save numeric values plus a `_mini_trainer_lr_dtype` +marker for non-default precision, allowing compiled restoration to recover that +choice. Ordinary eager loading still accepts the numeric rates. Native fused +SGD/Adam/AdamW tensor-rate kernels require an explicitly selected float32 tensor +rate (or its recorded checkpoint marker); unsupported rates fail before mutation. +The existing foreach/capturable restrictions still apply. + +INT8 updates use compiler-visible arithmetic and final storage copies during AOT +fake-tensor tracing. This avoids an opaque in-place operator carrying CPU scalar +inputs into CUDA graph partitions. The native row update remains available in +eager execution. Regression tests check actual optimizer-only replay, arithmetic, +AMP skips, rate precision, same-optimizer restoration and eager checkpoint resume. +Capture eligibility, extra gradient copies, graph workspace memory and first-use +compilation costs still depend on the optimizer and workload. Measure both float +and INT8 with the same options; enabling graphs alone is not evidence of a speedup. diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 1664115..7ef1147 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -21,7 +21,7 @@ from mini_trainer.modeling import Classifier from mini_trainer.train import main as train from mini_trainer.training import MuonAuxAdamW -from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options +from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options, validate_optimizer_compilation from .datasets import prepare_real from .models import NoAugmentationBuilder @@ -63,8 +63,10 @@ def run( optimizer: str = "muon", learning_rate: float | None = None, compile_mode: str | None = None, + optimizer_cudagraphs: bool = False, ): model_compile_options(compile, compile_mode) + validate_optimizer_compilation(compile_optimizer, optimizer_cudagraphs, device) if model_profile not in ("default", "dense") or optimizer not in ("muon", "adamw", "sgd"): raise ValueError("Unknown model or optimizer profile.") if learning_rate is None: @@ -154,6 +156,7 @@ def run( compile=compile, compile_mode=compile_mode, compile_optimizer=compile_optimizer, + optimizer_cudagraphs=optimizer_cudagraphs, model_builder_kwargs={ "model_type": model_type, "hidden": hidden if hidden else False, @@ -251,6 +254,7 @@ def run( "compile": compile, "compile_mode": compile_mode, "compile_optimizer": compile_optimizer, + "optimizer_cudagraphs": optimizer_cudagraphs, "hidden": hidden, "batch_size": batch_size, "cache_workers": cache_workers, @@ -328,6 +332,7 @@ def main(): parser.add_argument("--compile", action="store_true") parser.add_argument("--compile-mode", choices=MODEL_COMPILE_MODES) parser.add_argument("--compile-optimizer", action="store_true") + parser.add_argument("--optimizer-cudagraphs", action="store_true") parser.add_argument("--model-profile", choices=["default", "dense"], default="default") parser.add_argument("--optimizer", choices=["muon", "adamw", "sgd"], default="muon") parser.add_argument("--learning-rate", type=float) @@ -363,6 +368,7 @@ def main(): compile=args.compile, compile_mode=args.compile_mode, compile_optimizer=args.compile_optimizer, + optimizer_cudagraphs=args.optimizer_cudagraphs, hidden=args.hidden, batch_size=args.batch_size, cache_workers=args.cache_workers, @@ -386,6 +392,7 @@ def main(): "compile": args.compile, "compile_mode": args.compile_mode, "compile_optimizer": args.compile_optimizer, + "optimizer_cudagraphs": args.optimizer_cudagraphs, "hidden": args.hidden, "batch_size": args.batch_size, "cache_workers": args.cache_workers, diff --git a/docs/quantized-training.md b/docs/quantized-training.md index b2d57a0..286bf55 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -53,7 +53,7 @@ the developer probe. Eager SGD, AdamW and the repository's MuonAuxAdamW update paths are covered; fused optimizer variants are not established. Quantized regularization uses a differentiable floating view of the represented weights. -CUDA SGD/AdamW-style `add_` and `addcdiv_` updates fuse dequantization, +Eager CUDA SGD/AdamW-style `add_` and `addcdiv_` updates fuse dequantization, the weight update and stochastic requantization over the underlying storage tensors. This kernel compiles on first use even when the outer optimizer is eager. It retains INT8 codes and row scales, advances tensor version counters, @@ -97,6 +97,16 @@ by an explicit Triton kernel and custom-operator boundary; the storage kernel does not depend on Dynamo's per-frame variant cache. Model `--compile` remains a separate option. See the [measured results](benchmarks.md#optimizer-fma-dispatch). +`--compile-optimizer --optimizer-cudagraphs` opts into optimizer graph replay. +During AOT fake-tensor tracing, updates expose floating arithmetic followed by +functional requantization and storage copies. Eager native updates are retained. +Learning rates stay on CUDA during replay; checkpoints retain numeric values and +explicit non-default rate precision. This leaves the AMP gate, scheduler and +MuonAuxAdamW outer counter in their existing roles. See the +[optimizer graph requirements](../dev/README.md#optimizer-cuda-graphs), including +native fused float32-rate restrictions. The native fused optimizer tests use +floating parameters; they do not establish native fused updates of INT8 weights. + ## Checkpoints and inference Prepared models include their recipe in `state_dict`. Ordinary `mt_train` diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index 30295bc..70c32c9 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -7,6 +7,7 @@ from pathlib import Path import torch +from torch._subclasses.fake_tensor import is_fake from torch.utils._python_dispatch import return_and_correct_aliasing from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise @@ -290,7 +291,10 @@ def _apply_weight_update(original, update, alpha, denominator=None): and denominator.device == original.device ) ) - if not supported or torch.is_grad_enabled() and original.requires_grad: + # Expose update math and final copies to compiled graphs. An opaque + # in-place update can carry CPU scalar inputs across CUDA graph partitions + # and hides the storage dependencies the compiler needs to schedule safely. + if is_fake(original) or not supported or torch.is_grad_enabled() and original.requires_grad: change = update if denominator is None else update / denominator return original.copy_(original.dequantize() + change * alpha) if not isinstance(alpha, torch.Tensor): diff --git a/mini_trainer/train.py b/mini_trainer/train.py index bb37d37..3d59454 100644 --- a/mini_trainer/train.py +++ b/mini_trainer/train.py @@ -23,7 +23,7 @@ from mini_trainer.modeling import average_checkpoints, classification_module from mini_trainer.trainer import train from mini_trainer.training import MuonAuxAdamW -from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options +from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options, validate_optimizer_compilation from mini_trainer.utils import ( broadcast_from_master, ddp_train_wrapper, @@ -78,6 +78,7 @@ def main( # noqa: D417 ddp_info: dict | None = None, compile_optimizer: bool = False, compile_mode: str | None = None, + optimizer_cudagraphs: bool = False, ): """Train a classifier. @@ -114,6 +115,7 @@ def main( # noqa: D417 """ orig_args = locals() model_compile_options(compile, compile_mode) + validate_optimizer_compilation(compile_optimizer, optimizer_cudagraphs, device) # Prepare state if seed is not None: random.seed(seed) @@ -297,7 +299,7 @@ def main( # noqa: D417 if compile_optimizer: from mini_trainer.training.compilation import compile_optimizer as prepare_compiled_optimizer - prepare_compiled_optimizer(optimizer) + prepare_compiled_optimizer(optimizer, cudagraphs=optimizer_cudagraphs) log.info("Optimizer updates compiled; scheduler and AMP step gating remain active.") # Instantiate logger @@ -542,6 +544,11 @@ def cli(description="Train a classifier", **extra_kwargs): # noqa: D103 ) cfg_args = parser.add_argument_group("Runtime [optional]") + cfg_args.add_argument( + "--optimizer-cudagraphs", + action="store_true", + help="Opt into CUDA graph replay for optimizer updates; requires --compile-optimizer and CUDA.", + ) cfg_args.add_argument( "--compile-optimizer", action="store_true", diff --git a/mini_trainer/training/compilation.py b/mini_trainer/training/compilation.py index 45b9c93..7e02987 100644 --- a/mini_trainer/training/compilation.py +++ b/mini_trainer/training/compilation.py @@ -1,6 +1,6 @@ """Opt-in model and optimizer compilation.""" -from functools import wraps +from functools import partial, wraps import torch @@ -20,25 +20,50 @@ def model_compile_options(enabled: bool, mode: str | None) -> dict: return {"mode": mode} -def _tensor_learning_rates(optimizer): +def validate_optimizer_compilation(enabled: bool, cudagraphs: bool, device=None) -> None: + if cudagraphs and not enabled: + raise ValueError("optimizer_cudagraphs requires compile_optimizer=True (--compile-optimizer).") + if cudagraphs and device is not None and torch.device(device).type != "cuda": + raise ValueError("Optimizer CUDA graphs require CUDA.") + + +def _saved_rate_dtype(group): + name = group.get("_mini_trainer_lr_dtype", "float64") + dtype = getattr(torch, name, None) if isinstance(name, str) else None + if not isinstance(dtype, torch.dtype): + raise ValueError(f"Invalid saved learning-rate dtype: {name!r}.") + return dtype + + +def _tensor_learning_rates(optimizer, *, cudagraphs=False): for group in optimizer.param_groups: if not isinstance(group["lr"], torch.Tensor): # Python floats are double precision. Keep that value unchanged, # while letting compilation treat scheduler updates as inputs. - group["lr"] = torch.tensor(group["lr"], dtype=torch.float64) + group["lr"] = torch.tensor(group["lr"], dtype=_saved_rate_dtype(group)) + group.pop("_mini_trainer_lr_dtype", None) + if cudagraphs and group["params"]: + # Preserve explicitly supplied tensor precision. Only numeric rates + # need the float64 construction above; graph inputs stay on device. + group["lr"] = group["lr"].to(group["params"][0].device) def _portable_learning_rates(optimizer, state): for group in state["param_groups"]: if isinstance(group["lr"], torch.Tensor): + if group["lr"].dtype != torch.float64: + group["_mini_trainer_lr_dtype"] = str(group["lr"].dtype).removeprefix("torch.") + else: + group.pop("_mini_trainer_lr_dtype", None) group["lr"] = group["lr"].item() return state -def _compile_after_initial_call(optimizer, options): +def _compile_after_initial_call(optimizer, options, *, cudagraphs=False): step = optimizer.step compiled = torch.compile(step, **options) initialized = False + normalize_rates = partial(_tensor_learning_rates, cudagraphs=cudagraphs) @wraps(step) def update(*args, **kwargs): @@ -47,9 +72,13 @@ def update(*args, **kwargs): # Initialize lazy momentum/state during a real call. Tracing that # mutation across multiple groups can fail in Dynamo; never insert # a fake update or bypass GradScaler to initialize it. + # A restored explicitly typed rate must already have its original + # representation during this real eager update, not only afterward. + if any("_mini_trainer_lr_dtype" in group for group in optimizer.param_groups): + normalize_rates(optimizer) result = step(*args, **kwargs) - _tensor_learning_rates(optimizer) - optimizer.register_load_state_dict_post_hook(_tensor_learning_rates) + normalize_rates(optimizer) + optimizer.register_load_state_dict_post_hook(normalize_rates) initialized = True return result return compiled(*args, **kwargs) @@ -57,16 +86,45 @@ def update(*args, **kwargs): return update -def compile_optimizer(optimizer, *, backend=None): +def compile_optimizer(optimizer, *, backend=None, cudagraphs=False): """Compile updates after scheduler construction and checkpoint restoration. Keep hooks, scaler overflow decisions and scheduler calls in their usual order. Composite Muon counters stay outside compiled child updates. Numeric learning rates become scalar tensor inputs, but checkpoint groups retain scalar rates so ordinary eager resume does not require this option. + Optional CUDA graphs move learning-rate tensors to the parameter device; + capture eligibility and benefits depend on the optimizer and workload. """ targets = [getattr(optimizer, name) for name in optimizer.optimizers] if isinstance(optimizer, MuonAuxAdamW) else [optimizer] + if cudagraphs: + devices = {p.device for target in targets for group in target.param_groups for p in group["params"]} + if backend is not None or len(devices) != 1 or next(iter(devices)).type != "cuda": + raise ValueError("Optimizer CUDA graphs require the default Inductor backend and parameters on one CUDA device.") for target in targets: + for group in target.param_groups: + _saved_rate_dtype(group) + if getattr(target, "_mini_trainer_compiled", False) and getattr(target, "_mini_trainer_optimizer_cudagraphs", False) != cudagraphs: + raise ValueError("Optimizer is already compiled with a different CUDA graph setting.") + if ( + cudagraphs + and isinstance(target, (torch.optim.SGD, torch.optim.Adam, torch.optim.AdamW)) + and any( + group.get("fused") + and not ( + isinstance(group["lr"], torch.Tensor) + and group["lr"].dtype == torch.float32 + or not isinstance(group["lr"], torch.Tensor) + and group.get("_mini_trainer_lr_dtype") == "float32" + ) + for group in target.param_groups + ) + ): + raise ValueError( + "Native fused optimizer CUDA graphs require an explicit float32 tensor learning rate. " + "Supply that rate when constructing the optimizer, or leave optimizer CUDA graphs disabled; " + "numeric/float64 rates are not silently rounded." + ) if isinstance(target, (torch.optim.Adam, torch.optim.AdamW)) and any( group.get("foreach") and not group.get("capturable") for group in target.param_groups ): @@ -83,6 +141,9 @@ def compile_optimizer(optimizer, *, backend=None): # Newton-Schulz intentionally rounds intermediate values to BF16. # Preserve those casts instead of silently changing its iteration. options["options"] = {"emulate_precision_casts": True} - target.step = _compile_after_initial_call(target, options) + if cudagraphs: + options.setdefault("options", {})["triton.cudagraphs"] = True + target.step = _compile_after_initial_call(target, options, cudagraphs=cudagraphs) target._mini_trainer_compiled = True + target._mini_trainer_optimizer_cudagraphs = cudagraphs return optimizer diff --git a/tests/test_benchmark_synthetic.py b/tests/test_benchmark_synthetic.py index b5b95f6..f5e5f5f 100644 --- a/tests/test_benchmark_synthetic.py +++ b/tests/test_benchmark_synthetic.py @@ -70,7 +70,18 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): monkeypatch.setattr( sys, "argv", - ["benchmark", "--output", str(tmp_path / "failed"), "--device", "cuda:0", "--compile", "--compile-mode", "reduce-overhead"], + [ + "benchmark", + "--output", + str(tmp_path / "failed"), + "--device", + "cuda:0", + "--compile", + "--compile-mode", + "reduce-overhead", + "--compile-optimizer", + "--optimizer-cudagraphs", + ], ) monkeypatch.setattr(torch.cuda, "is_available", lambda: False) monkeypatch.setattr(torch, "set_num_threads", lambda threads: None) @@ -81,6 +92,7 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): assert report["status"] == "failed" assert report["device"] == "cuda:0" assert report["compile_mode"] == "reduce-overhead" + assert report["optimizer_cudagraphs"] is True assert report["error"]["type"] == "RuntimeError" assert "test_accuracy" not in report @@ -110,3 +122,31 @@ def test_compile_mode_requires_compilation_before_creating_outputs(tmp_path): with pytest.raises(ValueError, match="Unknown model compile mode"): model_compile_options(True, "invalid") assert not list(tmp_path.iterdir()) + + +def test_optimizer_graphs_require_compilation_before_output(tmp_path): + import pytest + + from dev.benchmarks.run import run + from mini_trainer.train import main + + with pytest.raises(ValueError, match="requires compile_optimizer=True"): + run(tmp_path / "benchmark", optimizer_cudagraphs=True) + with pytest.raises(ValueError, match="requires compile_optimizer=True"): + main(input=str(tmp_path / "missing"), output=str(tmp_path / "train"), optimizer_cudagraphs=True) + assert not list(tmp_path.iterdir()) + + +def test_optimizer_graphs_reject_cpu_before_output(tmp_path): + import pytest + + from dev.benchmarks.run import run + from mini_trainer.train import main + + with pytest.raises(ValueError, match="require CUDA"): + run(tmp_path / "benchmark", device="cpu", compile_optimizer=True, optimizer_cudagraphs=True) + with pytest.raises(ValueError, match="require CUDA"): + main( + input=str(tmp_path / "missing"), output=str(tmp_path / "train"), device="cpu", compile_optimizer=True, optimizer_cudagraphs=True + ) + assert not list(tmp_path.iterdir()) diff --git a/tests/test_optimizer_steps.py b/tests/test_optimizer_steps.py index e6d12f8..ca606c3 100644 --- a/tests/test_optimizer_steps.py +++ b/tests/test_optimizer_steps.py @@ -240,15 +240,24 @@ def test_compiled_foreach_adamw_requires_capturable_before_mutation(): def _assert_compiled_optimizer_state(actual, expected, key=None): - if isinstance(actual, torch.Tensor): + if key == "lr": + # Compiled checkpoints deliberately serialize tensor rates as numbers. + # Compare those values exactly, including explicitly selected float32. + actual = actual.item() if isinstance(actual, torch.Tensor) else actual + expected = expected.item() if isinstance(expected, torch.Tensor) else expected + assert actual == expected + elif isinstance(actual, torch.Tensor): # Dynamo moves Adam's scalar step counter to CUDA. Check its exact # numeric state; all non-counter tensor devices must remain unchanged. if key == "step" and actual.numel() == expected.numel() == 1: expected = expected.to(actual.device) torch.testing.assert_close(actual, expected, rtol=1e-5, atol=2e-6) elif isinstance(actual, dict): - assert actual.keys() == expected.keys() - for name in actual: + # Non-default rate precision is checkpoint reconstruction metadata; + # eager state has no such marker. Roundtrip tests verify it separately. + actual_keys = actual.keys() - {"_mini_trainer_lr_dtype"} + assert actual_keys == expected.keys() - {"_mini_trainer_lr_dtype"} + for name in actual_keys: _assert_compiled_optimizer_state(actual[name], expected[name], name) elif isinstance(actual, (list, tuple)): assert type(actual) is type(expected) and len(actual) == len(expected) @@ -259,7 +268,8 @@ def _assert_compiled_optimizer_state(actual, expected, key=None): @pytest.mark.parametrize("kind", KINDS) -def test_cuda_compiled_optimizer_matches_eager_updates(kind): +@pytest.mark.parametrize("cudagraphs", [False, True]) +def test_cuda_compiled_optimizer_matches_eager_updates(kind, cudagraphs): from mini_trainer.training.compilation import compile_optimizer if os.environ.get("RUN_CUDA_TESTS") != "1": @@ -268,11 +278,18 @@ def test_cuda_compiled_optimizer_matches_eager_updates(kind): torch.manual_seed(71) initial = torch.randn(8, 8, device="cuda") parameters = [torch.nn.Parameter(initial.clone()) for _ in range(2)] - optimizers = [make_optimizer(kind, [parameter]) for parameter in parameters] + # Native fused tensor-lr kernels require float32; use the same explicitly + # chosen rate and scheduler arithmetic for both reference and candidate. + rate = torch.tensor(0.01, device="cuda", dtype=torch.float32) if cudagraphs and kind.startswith("fused") else 0.01 + optimizers = [ + make_optimizer(kind, [parameter], lr=rate.clone() if isinstance(rate, torch.Tensor) else rate) for parameter in parameters + ] schedulers = [torch.optim.lr_scheduler.StepLR(opt, step_size=1, gamma=0.93) for opt in optimizers] - compile_optimizer(optimizers[1]) + compile_optimizer(optimizers[1], cudagraphs=cudagraphs) scalers = [torch.amp.GradScaler("cuda", init_scale=8, growth_interval=100) for _ in optimizers] - for step in range(6): + for step in range(8): + if cudagraphs and step == 6: + optimizers[1].load_state_dict(copy.deepcopy(optimizers[1].state_dict())) gradient = torch.randn_like(initial) decisions = [] for parameter, optimizer, scheduler, scaler in zip(parameters, optimizers, schedulers, scalers, strict=True): @@ -287,3 +304,91 @@ def test_cuda_compiled_optimizer_matches_eager_updates(kind): assert decisions[0] == decisions[1] == (step != 2) torch.testing.assert_close(parameters[0], parameters[1], rtol=1e-5, atol=2e-6) _assert_compiled_optimizer_state(optimizers[0].state_dict(), optimizers[1].state_dict()) + + +def test_optimizer_graph_validation_precedes_mutation(): + from mini_trainer.training.compilation import compile_optimizer + + parameter = torch.nn.Parameter(torch.ones(4, 4)) + optimizer = torch.optim.SGD([parameter], lr=0.1) + original_step = optimizer.step + with pytest.raises(ValueError, match="one CUDA device"): + compile_optimizer(optimizer, cudagraphs=True) + assert optimizer.step == original_step and not optimizer.state + assert not optimizer._optimizer_state_dict_post_hooks + assert optimizer.param_groups[0]["lr"] == 0.1 + + +@pytest.mark.parametrize("kind", ["fused_sgd", "fused_adamw"]) +def test_cuda_fused_graph_rates_are_not_silently_rounded(kind): + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for native fused CUDA graph rate validation") + parameter = torch.nn.Parameter(torch.ones(4, 4, device="cuda")) + optimizer = make_optimizer(kind, [parameter]) + original_step = optimizer.step + with pytest.raises(ValueError, match="explicit float32 tensor learning rate"): + compile_optimizer(optimizer, cudagraphs=True) + assert optimizer.step == original_step and not optimizer.state + assert optimizer.param_groups[0]["lr"] == 0.01 + assert not optimizer._optimizer_state_dict_post_hooks + + +@pytest.mark.parametrize("kind", ["fused_sgd", "fused_adamw"]) +def test_cuda_graph_checkpoint_retains_explicit_rate_precision(kind): + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to check native fused graph checkpoint precision") + parameter = torch.nn.Parameter(torch.ones(4, 4, device="cuda")) + optimizer = make_optimizer(kind, [parameter], lr=torch.tensor(0.01, dtype=torch.float32, device="cuda")) + compile_optimizer(optimizer, cudagraphs=True) + parameter.grad = torch.ones_like(parameter) + optimizer.step() + state = copy.deepcopy(optimizer.state_dict()) + assert isinstance(state["param_groups"][0]["lr"], float) + assert state["param_groups"][0]["_mini_trainer_lr_dtype"] == "float32" + # Match the entrypoint order: restore first, then compile the fresh optimizer. + restored = make_optimizer(kind, [parameter]) + restored.load_state_dict(state) + compile_optimizer(restored, cudagraphs=True) + for _ in range(4): + restored.step() + assert restored.param_groups[0]["lr"].dtype == torch.float32 + assert restored.param_groups[0]["lr"].device == parameter.device + assert restored.state_dict()["param_groups"][0]["lr"] == state["param_groups"][0]["lr"] + + +def test_compiled_checkpoint_preserves_nondefault_rate_dtype_on_cpu(): + from mini_trainer.training.compilation import compile_optimizer + + parameter = torch.nn.Parameter(torch.ones(4, 4)) + optimizer = torch.optim.SGD([parameter], lr=torch.tensor(0.01, dtype=torch.float32)) + compile_optimizer(optimizer, backend="eager") + parameter.grad = torch.ones_like(parameter) + optimizer.step() + state = copy.deepcopy(optimizer.state_dict()) + assert state["param_groups"][0]["_mini_trainer_lr_dtype"] == "float32" + restored = torch.optim.SGD([parameter], lr=0.1) + restored.load_state_dict(state) + observed = [] + restored.register_step_pre_hook(lambda opt, args, kwargs: observed.append(opt.param_groups[0]["lr"].dtype)) + compile_optimizer(restored, backend="eager") + for _ in range(3): + restored.step() + assert restored.param_groups[0]["lr"].dtype == torch.float32 + assert observed == [torch.float32] * 3 + assert restored.state_dict()["param_groups"][0]["lr"] == state["param_groups"][0]["lr"] + + +def test_invalid_saved_rate_dtype_fails_before_compilation(): + from mini_trainer.training.compilation import compile_optimizer + + optimizer = torch.optim.SGD([torch.nn.Parameter(torch.ones(4))], lr=0.1) + optimizer.param_groups[0]["_mini_trainer_lr_dtype"] = "invalid" + original_step = optimizer.step + with pytest.raises(ValueError, match="Invalid saved learning-rate dtype"): + compile_optimizer(optimizer) + assert optimizer.step == original_step and not optimizer.state + assert not optimizer._optimizer_state_dict_post_hooks diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index 03540d9..b356e55 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -135,7 +135,7 @@ def test_mixed_muon_adamw_updates_and_counter(): @pytest.mark.parametrize("normalized", [False, True]) -@pytest.mark.parametrize("compiled_optimizer", [False, True]) +@pytest.mark.parametrize("compiled_optimizer", [False, True, "cudagraphs"]) @pytest.mark.parametrize("compile_mode", [None, "default", "reduce-overhead"]) def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, compiled_optimizer, compile_mode): from mini_trainer.modeling import Classifier @@ -155,7 +155,8 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, comp "device": device, "dtype": "float16", "quantized_training": True, - "compile_optimizer": compiled_optimizer, + "compile_optimizer": bool(compiled_optimizer), + "optimizer_cudagraphs": compiled_optimizer == "cudagraphs", "compile": compile_mode is not None, "compile_mode": compile_mode, "seed": 42, @@ -180,6 +181,7 @@ def test_training_entrypoint_checkpoint_and_inference(tmp_path, normalized, comp # plumbing, not identical stochastic continuation with a changed schedule. args["epochs"] = 2 args["compile_optimizer"] = False + args["optimizer_cudagraphs"] = False args["compile"] = False args["compile_mode"] = None args["name"] = "resumed" @@ -517,7 +519,8 @@ def test_fma_uses_represented_weights_without_mutation_or_rounding(position): @pytest.mark.parametrize("kind", ["sgd", "adamw"]) -def test_cuda_compiled_quantized_update_matches_float_before_rounding(kind): +@pytest.mark.parametrize("cudagraphs", [False, True]) +def test_cuda_compiled_quantized_update_matches_float_before_rounding(kind, cudagraphs): from mini_trainer.modeling._quantized_training import TrainingWeight from mini_trainer.training.compilation import compile_optimizer @@ -533,7 +536,7 @@ def test_cuda_compiled_quantized_update_matches_float_before_rounding(kind): options.update(momentum=0.9, nesterov=True) optimizer, eager = cls([weight], **options), cls([reference], **options) schedulers = [torch.optim.lr_scheduler.StepLR(opt, step_size=1, gamma=0.93) for opt in (optimizer, eager)] - compile_optimizer(optimizer) + compile_optimizer(optimizer, cudagraphs=cudagraphs) for _ in range(6): with torch.no_grad(): reference.copy_(weight.dequantize()) @@ -669,3 +672,55 @@ def test_normalized_compilation_preserves_embedding_loss_gradients(mode): ) assert model.linear.parametrizations.weight.original1.grad.norm() > 0 torch.testing.assert_close(results[0], results[1], rtol=1e-3, atol=1e-4) + + +@pytest.mark.parametrize("kind", ["sgd", "adamw", "muon"]) +def test_optimizer_graphs_replay_and_restore_device_rates(kind): + import copy + + from mini_trainer.trainer import _optimizer_step + from mini_trainer.training import MuonAuxAdamW + from mini_trainer.training.compilation import compile_optimizer + + device = cuda() + # Earlier tests deliberately compile many optimizer variants sharing the + # same step code objects. Isolate that cache history, never the steps or + # checkpoint transition within this test, and require actual replay below. + torch._dynamo.reset() + torch.manual_seed(61) + model = nn.Sequential(nn.Linear(64, 64), nn.ReLU(), nn.Linear(64, 4)).to(device) + prepare_quantized_training(model) + cls = {"sgd": torch.optim.SGD, "adamw": torch.optim.AdamW, "muon": MuonAuxAdamW}[kind] + options = {"momentum": 0.9} if kind == "sgd" else {} + optimizer = cls([{"name": "graph", "params": list(model.parameters())}], lr=0.01, weight_decay=0.01, **options) + scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.93) + scaler = torch.amp.GradScaler("cuda", enabled=False) + compile_optimizer(optimizer, cudagraphs=True) + assert compile_optimizer(optimizer, cudagraphs=True) is optimizer + with pytest.raises(ValueError, match="different CUDA graph setting"): + compile_optimizer(optimizer) + inputs = torch.randn(8, 64, device=device) + + def backward(): + optimizer.zero_grad(set_to_none=True) + model(inputs).square().mean().backward() + + for _ in range(7): + backward() + assert _optimizer_step(optimizer, scaler) + scheduler.step() + backward() + # Profile only the optimizer: model autotuning or model graph launches + # cannot satisfy this assertion. Warmup above initializes and captures it. + with torch.profiler.profile(activities=[torch.profiler.ProfilerActivity.CPU, torch.profiler.ProfilerActivity.CUDA]) as profile: + assert _optimizer_step(optimizer, scaler) + assert any(event.key == "cudaGraphLaunch" and event.count > 0 for event in profile.key_averages()) + optimizer.load_state_dict(copy.deepcopy(optimizer.state_dict())) + for group in optimizer.param_groups: + assert group["lr"].device == device and group["lr"].dtype == torch.float64 + for _ in range(3): + backward() + assert _optimizer_step(optimizer, scaler) + scheduler.step() + if kind == "muon": + assert optimizer._step_count == 11 From 0fd9e3243e43f3147a34d309776b7740bba4e31f Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 08:25:00 +0200 Subject: [PATCH 034/155] fix: share CUDA graph iteration boundaries across model and optimizer Mark each training iteration explicitly so separately compiled optimizer graphs do not retire backward gradient buffers. Propagate graph mode through composite Muon optimizers and document the custom-loop contract. Regression reproduced with floating and INT8 weights before the fix. Validation: static checks; full CPU suite (323 passed, 129 skipped, one known EMA xfail); CUDA optimizer/checkpoint matrix (141 passed); expanded SGD, AdamW and Muon iteration regression (6 passed). --- docs/quantized-training.md | 7 ++++- mini_trainer/trainer.py | 5 ++++ mini_trainer/training/compilation.py | 1 + tests/test_optimizer_steps.py | 41 ++++++++++++++++++++++++++++ 4 files changed, 53 insertions(+), 1 deletion(-) diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 286bf55..f8c0a33 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -102,7 +102,12 @@ During AOT fake-tensor tracing, updates expose floating arithmetic followed by functional requantization and storage copies. Eager native updates are retained. Learning rates stay on CUDA during replay; checkpoints retain numeric values and explicit non-default rate precision. This leaves the AMP gate, scheduler and -MuonAuxAdamW outer counter in their existing roles. See the +MuonAuxAdamW outer counter in their existing roles. The training loop explicitly +marks each graph iteration before the model runs, keeping backward gradient +buffers alive until the optimizer consumes them. Custom training loops must call +[`torch.compiler.cudagraph_mark_step_begin()`](https://docs.pytorch.org/docs/2.12/generated/torch.compiler.cudagraph_mark_step_begin.html) +before each training iteration when +combining compiled models with optimizer graph replay. See the [optimizer graph requirements](../dev/README.md#optimizer-cuda-graphs), including native fused float32-rate restrictions. The native fused optimizer tests use floating parameters; they do not establish native fused updates of INT8 weights. diff --git a/mini_trainer/trainer.py b/mini_trainer/trainer.py index 1aad88f..ad38ad5 100644 --- a/mini_trainer/trainer.py +++ b/mini_trainer/trainer.py @@ -122,6 +122,11 @@ def train_one_epoch( start_time = time.time() for i, (batch, target) in enumerate(pbar): + if getattr(optimizer, "_mini_trainer_optimizer_cudagraphs", False): + # Model, backward, and optimizer graphs belong to one iteration. + # Automatic inference can otherwise retire backward's gradient buffers + # before the separately compiled optimizer consumes them. + torch.compiler.cudagraph_mark_step_begin() step = n_batches * epoch + i if len(batch.shape) != 4: raise RuntimeError(f"Incorrect {batch.shape=}, expected 4 dimensions, not {len(batch.shape)}.") diff --git a/mini_trainer/training/compilation.py b/mini_trainer/training/compilation.py index 7e02987..f4e0e0e 100644 --- a/mini_trainer/training/compilation.py +++ b/mini_trainer/training/compilation.py @@ -146,4 +146,5 @@ def compile_optimizer(optimizer, *, backend=None, cudagraphs=False): target.step = _compile_after_initial_call(target, options, cudagraphs=cudagraphs) target._mini_trainer_compiled = True target._mini_trainer_optimizer_cudagraphs = cudagraphs + optimizer._mini_trainer_optimizer_cudagraphs = cudagraphs return optimizer diff --git a/tests/test_optimizer_steps.py b/tests/test_optimizer_steps.py index ca606c3..807af93 100644 --- a/tests/test_optimizer_steps.py +++ b/tests/test_optimizer_steps.py @@ -392,3 +392,44 @@ def test_invalid_saved_rate_dtype_fails_before_compilation(): compile_optimizer(optimizer) assert optimizer.step == original_step and not optimizer.state assert not optimizer._optimizer_state_dict_post_hooks + + +@pytest.mark.parametrize("quantized", [False, True]) +@pytest.mark.parametrize("kind", ["sgd", "adamw", "muon"]) +def test_cuda_model_and_optimizer_graphs_share_training_iterations(quantized, kind): + from mini_trainer.modeling.ema import EMATeacher + from mini_trainer.modeling.quantized_training import prepare_quantized_training + from mini_trainer.training.compilation import compile_optimizer + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify combined model and optimizer graph lifetimes") + torch._dynamo.reset() + torch.manual_seed(31) + device = torch.device("cuda:0") + raw = torch.nn.Sequential(torch.nn.Flatten(), torch.nn.Linear(12, 16), torch.nn.ReLU(), torch.nn.Linear(16, 3)).to(device) + if quantized: + prepare_quantized_training(raw) + model = torch.compile(raw, mode="reduce-overhead") + optimizer = make_optimizer(kind, raw.parameters()) + scheduler = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.93) + compile_optimizer(optimizer, cudagraphs=True) + loader = DataLoader(TensorDataset(torch.randn(24, 3, 2, 2), torch.arange(24) % 3), batch_size=4) + logger = Mock() + logger.status.return_value = "combined graph regression" + train_one_epoch( + model, + EMATeacher(enable=False, total_steps=6), + torch.nn.CrossEntropyLoss(), + optimizer, + torch.amp.GradScaler("cuda", init_scale=8), + scheduler, + loader, + 0, + logger, + device=device, + dtype=torch.float16, + ) + assert scheduler.last_epoch == 6 + assert logger.consume.call_count == 6 + if kind == "muon": + assert optimizer._step_count == 6 From 9fb6d3e3c7d16ad44b5c1bbfce7115a7fa17e6f2 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 08:31:39 +0200 Subject: [PATCH 035/155] docs: record optimizer graph performance and first-use failure Preserve the failed INT8 attempt alongside the warmed retry and matched floating baselines. Report memory, accuracy, later-phase timing and whole-call timing separately without claiming cold-start reliability or a general QT speedup. --- docs/benchmarks.md | 60 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 60 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 133f5ae..bf20bca 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -717,3 +717,63 @@ predictions and the class specification are at `/tmp/mini-trainer-embedding-blai After this change, both synthetic CUDA oracle runs again reached 100% with model and MuonAuxAdamW optimizer compilation, `reduce-overhead`, FP16 AMP, training and checkpoint reload. Reports are at `/tmp/mini-trainer-embedding-oracle`. + +### Optimizer graph replay: iteration fix and first-use limitation + +Revision `0fd9e32` explicitly marks each training iteration when optimizer CUDA +graphs are enabled. Before this fix, the separately compiled optimizer could +access a gradient whose model graph storage had already been retired. Six-batch +regressions now pass for SGD, AdamW and MuonAuxAdamW, with both floating and INT8 +weights; the composite optimizer's outer update counter remains intact. + +The following local MNIST runs used the same source and lock hashes, dataset +manifest, seed 42, dense model, SGD (learning rate 0.3, momentum 0.9), batch 128, +15 epochs, FP16 AMP, CPU cache, zero loader/cache workers, model compilation with +`reduce-overhead`, and optimizer compilation. Hardware was the RTX 3080 Ti Laptop +GPU with PyTorch 2.12.0/CUDA 13.0; Torch and Inductor each used one thread. +All completed runs reloaded the checkpoint for held-out inference and saved +finite scores. This is a single-seed functional/performance probe, not a quality +gate or convergence comparison. + +| Weights | Optimizer graphs | Test accuracy | Peak MiB | Median train epoch 3–15 s | Training wall s | +| --- | --- | ---: | ---: | ---: | ---: | +| Float | Off | 93.12% | 217.41 | 0.193 | 14.56 | +| Float | On | 93.12% | 217.42 | 0.209 | 15.43 | +| INT8 | Off | 92.84% | 149.60 | 0.292 | 37.26 | +| INT8 | On, warmed retry | 92.84% | 167.60 | 0.230 | 18.04 | + +**The first INT8 graph-enabled attempt failed**, during the first compiled +backward before any optimizer update, with `These storage data ptrs are not +allocated in pool (0, 1) but should be`. The baseline and subsequent graph-enabled +retry completed. First-use kernel tuning or compilation is a suspected cause, +not an established diagnosis; the warmed retry does not resolve this failure. +Do not interpret these results as reliable cold-start support. + +Run order was float off, float on, failed INT8 on, INT8 off, INT8 on retry. +Caches were not cleared, and no tests ran concurrently. The failed attempt +warmed caches, so whole-call times are not a controlled comparison of compilation +cost. Optimizer replay reduced the INT8 later-phase median by about 21%, but +raised its peak allocation by 12%. INT8 with replay still ran about 19% slower +per later epoch than float without replay, while using about 23% less peak +memory. Float replay was slower. Keep optimizer graphs opt-in; these results +do not establish a general QT speed advantage. + +Reports, predictions, logs, and the original failures are retained locally in +`tmp-optimizer-cudagraphs/pairs` and `tmp-optimizer-cudagraphs/pairs-fixed` +(ignored generated artifacts, not repository-hosted results). Reproduce each +completed configuration with a fresh output directory: + +```bash +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ +MPLCONFIGDIR=/tmp/mini-trainer-mpl MPLBACKEND=Agg \ +.venv/bin/python -m dev.benchmarks.run --output /tmp/mnist-optimizer-graphs \ + --dataset mnist --data-root examples/mnist --seed 42 --model-profile dense \ + --optimizer sgd --learning-rate 0.3 --epochs 15 --batch-size 128 \ + --compile --compile-mode reduce-overhead --compile-optimizer \ + --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 \ + --allow-nondeterministic +``` + +Add `--quantized-training` for INT8 and `--optimizer-cudagraphs` for optimizer +replay. Fresh-cache failure reproduction and a fix must precede recommending +this combination in a continuous benchmark profile. From 48f805818d83170fad8a11668bead13b3a137779 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 08:47:43 +0200 Subject: [PATCH 036/155] fix: release failed INT8 tuning candidates before graph pool checks Expected Triton candidate failures retained temporary tensors in exception tracebacks during CUDA graph warmup. Clear those tracebacks while preserving infinite candidate scores and propagation of unexpected failures. Validate immediate release, forced fresh tuning in dense backward graphs, the full CPU suite (326 passed, 134 skipped, known EMA xfail), CUDA kernel tests (17 passed), CUDA optimizer/checkpoint tests (145 passed), and 15-epoch MNIST with forced retuning and checkpoint inference (92.84% accuracy). --- docs/benchmarks.md | 41 ++++++++++++ docs/quantized-training.md | 3 + mini_trainer/modeling/_quantized_matmul.py | 11 +++- tests/test_quantized_training.py | 77 ++++++++++++++++++++++ 4 files changed, 131 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index bf20bca..6f6104c 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -777,3 +777,44 @@ MPLCONFIGDIR=/tmp/mini-trainer-mpl MPLBACKEND=Agg \ Add `--quantized-training` for INT8 and `--optimizer-cudagraphs` for optimizer replay. Fresh-cache failure reproduction and a fix must precede recommending this combination in a continuous benchmark profile. + +#### First-use tuning failure follow-up + +Forced retuning reproduced the first-backward pool failure three times with +already compiled kernels. Allocation diagnostics showed that the mismatched +storage reference expired during the graph check's garbage collection. Rejected +Triton candidates had requested 122,880 bytes of shared memory on a GPU limited +to 101,376 bytes; their exception tracebacks retained temporary tensors. + +The local benchmark callback now clears tracebacks for the same expected +candidate failures that Triton already scores with infinite timing. This releases +those tensors promptly. Unexpected errors still propagate, kernel arithmetic is +unchanged, and CUDA graph assertions remain enabled. No global garbage collection +or third-party monkey patch is added. A diagnostic two-epoch MNIST run completed +with forced retuning after this change. + +Regression coverage includes immediate tensor release without `gc.collect()`, +unexpected-error propagation, and three dense-model forward/backward/update +iterations with fresh local tuning. The latter disables only that tuner's memory +and disk result caches; it preserves installed dependencies and compiled kernel +caches and requires no dataset download: + +```bash +CUDA_VISIBLE_DEVICES=0 RUN_CUDA_TESTS=1 OMP_NUM_THREADS=1 \ +TORCHINDUCTOR_COMPILE_THREADS=1 \ +bash dev/check.sh test tests/test_quantized_training.py -k fresh_kernel_tuning +``` + +This addresses the reproduced failure mechanism. The earlier timing table remains +historical evidence, including its failed attempt; it does not become a controlled +cold-start timing comparison or establish a general QT speed advantage. + +The production fix also completed the full 15-epoch MNIST configuration above +with forced local retuning: 92.84% held-out accuracy, finite saved scores, +167.60 MiB peak allocation, 0.203 s median train epoch 3–15, and 24.33 s training +wall time. Checkpoint reload and inference completed. Results are retained in +`tmp-optimizer-cudagraphs/tuning-fixed-mnist`; source hash +`23df40b2027a8a18de797f60c93c84aad72653c797a35e0f1883e9efcccbf2e4`. +This verifies the formerly failing path. It forces tuner results to be recomputed, +not a completely empty compiler cache, and has no concurrent matched float run, +so its timing is not evidence of a new speedup. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index f8c0a33..163d43d 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -100,6 +100,9 @@ separate option. See the [measured results](benchmarks.md#optimizer-fma-dispatch `--compile-optimizer --optimizer-cudagraphs` opts into optimizer graph replay. During AOT fake-tensor tracing, updates expose floating arithmetic followed by functional requantization and storage copies. Eager native updates are retained. +Expected failed kernel-tuning candidates release their exception tracebacks +promptly so temporary tensors do not outlive graph pool tracking during first +use; unexpected kernel errors still propagate. Learning rates stay on CUDA during replay; checkpoints retain numeric values and explicit non-default rate precision. This leaves the AMP gate, scheduler and MuonAuxAdamW outer counter in their existing roles. The training loop explicitly diff --git a/mini_trainer/modeling/_quantized_matmul.py b/mini_trainer/modeling/_quantized_matmul.py index ada2271..0f772c7 100644 --- a/mini_trainer/modeling/_quantized_matmul.py +++ b/mini_trainer/modeling/_quantized_matmul.py @@ -3,11 +3,20 @@ import torch import triton from torchao.prototype.quantized_training.int8_mm import _scaled_int8_mm_kernel as _upstream_kernel +from triton.compiler.errors import CompileTimeAssertionFailure +from triton.runtime.errors import OutOfResources, PTXASError from triton.testing import do_bench_cudagraph def _benchmark(kernel, quantiles): - return do_bench_cudagraph(kernel, rep=5, quantiles=quantiles) + try: + return do_bench_cudagraph(kernel, rep=5, quantiles=quantiles) + except (OutOfResources, CompileTimeAssertionFailure, PTXASError) as error: + # Triton rejects these candidates with infinite timing. Release their + # traceback frames here: they can retain temporary tensors until GC, + # beyond the lifetime tracked by the surrounding model CUDA graph. + error.__traceback__ = None + return [float("inf")] * len(quantiles) # Reuse TorchAO's kernel and candidate configurations, but create a separate diff --git a/tests/test_quantized_training.py b/tests/test_quantized_training.py index acc0a28..eaaf244 100644 --- a/tests/test_quantized_training.py +++ b/tests/test_quantized_training.py @@ -187,3 +187,80 @@ def test_compiled_integer_parameter_gradients(): assert torch.count_nonzero(actual.grad) > 0 error = (actual.grad.float() - expected.grad.float()).norm() / expected.grad.float().norm() assert error < 0.04 + + +@pytest.mark.parametrize("kind", ["resources", "ptxas"]) +def test_failed_tuning_candidate_releases_temporary_tensors(monkeypatch, kind): + import weakref + + pytest.importorskip("torchao") + from triton.runtime.errors import OutOfResources, PTXASError + + from mini_trainer.modeling import _quantized_matmul as matmul + + error = OutOfResources(2, 1, "shared memory") if kind == "resources" else PTXASError("invalid candidate") + references = [] + + def candidate(): + temporary = torch.ones(4) + references.append(weakref.ref(temporary)) + raise error + + monkeypatch.setattr(matmul, "do_bench_cudagraph", lambda kernel, **kwargs: kernel()) + assert matmul._benchmark(candidate, (0.5, 0.2, 0.8)) == [float("inf")] * 3 + # No gc.collect(): graph pool tracking needs prompt release on return. + assert references[0]() is None + assert error.__traceback__ is None + + +def test_tuning_preserves_unexpected_failures(monkeypatch): + pytest.importorskip("torchao") + from mini_trainer.modeling import _quantized_matmul as matmul + + def candidate(): + raise RuntimeError("unexpected kernel failure") + + monkeypatch.setattr(matmul, "do_bench_cudagraph", lambda kernel, **kwargs: kernel()) + with pytest.raises(RuntimeError, match="unexpected kernel failure"): + matmul._benchmark(candidate, (0.5, 0.2, 0.8)) + + +def test_dense_backward_graph_with_fresh_kernel_tuning(monkeypatch): + pytest.importorskip("torchao") + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for fresh INT8 kernel tuning in CUDA graphs") + from dev.benchmarks.models import DenseImageMLP + from mini_trainer.modeling import Classifier, EmbeddingContext + from mini_trainer.modeling import _quantized_matmul as matmul + from mini_trainer.modeling.quantized_training import prepare_quantized_training + + torch._dynamo.reset() + torch.manual_seed(42) + monkeypatch.setattr(matmul._kernel, "cache", {}) + monkeypatch.setattr(matmul._kernel, "cache_results", False) + calls = [] + benchmark = matmul._kernel._do_bench + + def record_tuning(kernel, quantiles): + calls.append(True) + return benchmark(kernel, quantiles) + + monkeypatch.setattr(matmul._kernel, "_do_bench", record_tuning) + raw = DenseImageMLP() + raw.fc = Classifier(2048, 10, hidden=False, normalized=False) + raw.cuda() + prepare_quantized_training(raw) + model = torch.compile(raw, mode="reduce-overhead") + optimizer = torch.optim.SGD(raw.parameters(), lr=0.01) + images = torch.randn(128, 3, 28, 28, device="cuda") + labels = torch.arange(128, device="cuda") % 10 + for _ in range(3): + torch.compiler.cudagraph_mark_step_begin() + optimizer.zero_grad() + with torch.autocast("cuda", dtype=torch.float16), EmbeddingContext(): + loss = torch.nn.functional.cross_entropy(model(images), labels) + loss.backward() + assert torch.isfinite(loss) + assert all(p.grad is not None and torch.isfinite(p.grad).all() for p in raw.parameters()) + optimizer.step() + assert calls, "The regression must retune kernels even when disk caches are warm" From e5387d8036cec7bc96e326a843b29d9263a80634 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 08:58:45 +0200 Subject: [PATCH 037/155] docs: record QT update bottlenecks and rejected requantization fusion Retain warmed profiler findings and two-seed MNIST comparisons. The tensor-arithmetic candidate passed CUDA regressions but did not reduce memory or improve later-phase training time, so production code remains unchanged. --- docs/benchmarks.md | 48 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 48 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 6f6104c..5a237e4 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -818,3 +818,51 @@ wall time. Checkpoint reload and inference completed. Results are retained in This verifies the formerly failing path. It forces tuner results to be recomputed, not a completely empty compiler cache, and has no concurrent matched float run, so its timing is not evidence of a new speedup. + +### Rejected experiment: compiler-visible row requantization + +After `48f8058`, a warmed fourth training epoch was profiled on the same dense +MNIST configuration (31 batches of 128, FP16 AMP, CPU cache, zero workers). +INT8 used model and optimizer graph replay; the float reference used model graph +replay and the ordinary compiled optimizer. These are the respective configurations +under investigation, not a controlled test of the optimizer graph flag alone. +Profiler runs are diagnostic and must not be used as wall-time benchmarks. + +In the INT8 trace, the scaled matrix kernels accumulated 9.96 ms of device time, +versus 19.89 ms for the large floating update kernels and 7.32 ms for row +requantization. Gradient unscaling and clipping's scaling pass each took about +7.4 ms. Nested compiled-region timings overlap their child kernel timings and +must not be added to them. The traces identify update-buffer traffic and gradient +handling as worthwhile targets; they do not establish a kernel-only speedup. +Local traces and tables are retained under +`tmp-optimizer-cudagraphs/profile-int8` and `profile-float`. + +A candidate replaced the opaque row requantization call during fake-tensor +tracing with TorchAO's tensor arithmetic, intending to let Inductor fuse the +floating update and row reduction. Eager execution was unchanged. All 67 CUDA +model/optimizer regression cases passed, but the following real training results +did not justify retaining it. The candidate was reverted. + +| Seed | Requantization | Test accuracy | Peak MiB | Median train epoch 3–15 s | Training wall s | +| --- | --- | ---: | ---: | ---: | ---: | +| 42 | Existing kernel | 92.84% | 167.60 | 0.144 | 10.95 | +| 42 | Tensor arithmetic candidate | 93.10% | 167.60 | 0.175 | 17.04 | +| 43 | Existing kernel | 93.02% | 167.60 | 0.144 | 11.20 | +| 43 | Tensor arithmetic candidate | 92.78% | 167.60 | 0.146 | 11.88 | + +Settings match the 15-epoch SGD/INT8 optimizer-graph MNIST probe above. Both +baseline seeds ran before the candidate was installed; both candidate seeds ran +after its CUDA checks finished. No tests or other benchmark runs overlapped the +timed runs. Caches were not cleared, so the first candidate's whole-call time +includes additional compilation work and is not a controlled cold-start measure. +Peak memory did not fall, later-phase time did not improve, and stochastic +rounding changed the trajectories. Two seeds do not establish quality equivalence. +Reports are retained locally in `tmp-optimizer-cudagraphs/fusion-before-{42,43}` +and `fusion-after-{42,43}`. + +The next update-path experiment should avoid constructing the full floating +updated-weight buffer explicitly, while keeping new INT8 codes/scales as +functional outputs and making the final parameter copies visible to the compiler. +It must preserve optimizer state, scalar learning-rate precision, AMP gating, +checkpoint behavior, and stochastic-rounding quality. Simply exposing more tensor +arithmetic did not provide that improvement here. From 1152b78ae3dd8bd682078d07a5c6d8c5010578cc Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 09:12:45 +0200 Subject: [PATCH 038/155] docs: record functional INT8 update trials and timing variability Retain numerical validation and alternating AdamW comparisons. The candidate was reverted because warmed repeats showed no memory benefit and slower training. --- docs/benchmarks.md | 39 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 39 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 5a237e4..29d5647 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -866,3 +866,42 @@ functional outputs and making the final parameter copies visible to the compiler It must preserve optimizer state, scalar learning-rate precision, AMP gating, checkpoint behavior, and stochastic-rounding quality. Simply exposing more tensor arithmetic did not provide that improvement here. + +### Rejected experiment: functional fused weight update + +A second candidate reused the native update kernel to produce fresh INT8 codes +and scales directly, followed by compiler-visible parameter copies. It avoided +an explicit floating updated-weight result in that kernel's caller and consumed +learning-rate tensors on CUDA. Six kernel cases matched the existing native +update exactly across FP32, FP16 and BF16, with division-based updates and +noncontiguous inputs. All 67 CUDA model/optimizer cases and 23 kernel cases passed. + +Inspection confirmed that compiled AdamW reached the new operator. Compiled SGD +continued through its existing decomposition, including with `foreach=False`, so +this experiment did not establish an SGD optimization. The following AdamW runs +used the dense MNIST profile above, learning rate 0.001, 15 epochs, batch 128, +FP16 AMP, CPU cache, and both model and optimizer graph replay. Baselines ran from +an isolated snapshot of `e5387d8`; the candidate used the working checkout and the +same virtual environment and data. No timed run overlapped another run or tests. + +| Seed/run | Existing median train epoch 3–15 s | Candidate median s | Existing / candidate accuracy | +| --- | ---: | ---: | ---: | +| 42, initial pair | 0.152 | 0.162 | 94.14% / 94.34% | +| 43, reversed order | 0.267 | 0.133 | 94.02% / 94.00% | +| 42, warmed before→after | 0.139 | 0.153 | 94.14% / 94.34% | +| 42, warmed after→before | 0.197 | 0.220 | 94.14% / 94.34% | + +All runs peaked at 218.10 MiB. Timing varied substantially between runs, so the +second seed alone would give a misleading speedup claim. Both additional warmed +pairs favored the existing implementation by about 10–12%. Whole-call times in +those pairs were 13.05 vs 13.69 s and 16.68 vs 17.11 s. The candidate was reverted; +passing numerical tests alone did not justify extra kernel complexity without a +measured memory or speed benefit. Its patch and reports are retained locally in +`tmp-optimizer-cudagraphs/rejected-functional-update.patch`, +`functional-adam-{before,after}*`, and `adam-repeat-*`. + +Update fusion remains a possible future direction, but merely placing the final +weight arithmetic inside the row kernel did not improve this workload. Further +performance validation should also include larger compute workloads, where the +integer GEMMs have more opportunity to offset update and scheduling overhead, +and should retain alternating run order and report variability. From d9ecab63f07c520ee9665182e7419ec5d4a92c50 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 09:29:54 +0200 Subject: [PATCH 039/155] feat: retain paired model and optimizer graph benchmarks in CI Add the three-seed optimizer graph profile alongside existing controls, with visible summaries and retained failure artifacts. Record the real MNIST profile showing lower INT8 peak memory and faster training phases across all three matched seeds, with timing and quality limitations. Validation: static checks, shell syntax, workflow YAML, full CPU suite (327 passed, 134 skipped, known EMA xfail), and twelve completed real GPU runs across the two graph settings. --- .github/workflows/benchmarks.yml | 6 ++++- dev/benchmarks/README.md | 14 ++++++++++ dev/check-benchmarks.sh | 10 +++++--- docs/benchmarks.md | 44 ++++++++++++++++++++++++++++++++ tests/test_benchmark_datasets.py | 7 ++--- 5 files changed, 74 insertions(+), 7 deletions(-) diff --git a/.github/workflows/benchmarks.yml b/.github/workflows/benchmarks.yml index 41f2519..9a06d78 100644 --- a/.github/workflows/benchmarks.yml +++ b/.github/workflows/benchmarks.yml @@ -104,6 +104,9 @@ jobs: - name: Multi-seed CUDA graph quantized training if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() run: bash dev/check-benchmarks.sh qt-cudagraphs benchmark-qt-cudagraphs + - name: Multi-seed model and optimizer graph quantized training + if: (inputs.qt || vars.BENCHMARK_QT == 'true') && (inputs.real_data || vars.BENCHMARK_REAL_DATA == 'true') && !cancelled() + run: bash dev/check-benchmarks.sh qt-optimizer-cudagraphs benchmark-qt-optimizer-cudagraphs - name: Synthetic GPU precision profiles run: bash dev/check-benchmarks.sh gpu benchmark-gpu - name: Real-data progression @@ -113,7 +116,7 @@ jobs: if: always() run: | python3 -m dev.benchmarks.summarize benchmark-gpu >> "$GITHUB_STEP_SUMMARY" - for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense benchmark-qt-large-batch benchmark-qt-cudagraphs; do + for qt_results in benchmark-qt benchmark-qt-real benchmark-qt-dense benchmark-qt-large-batch benchmark-qt-cudagraphs benchmark-qt-optimizer-cudagraphs; do if [[ -d "$qt_results" ]]; then python3 -m dev.benchmarks.summarize "$qt_results" >> "$GITHUB_STEP_SUMMARY" fi @@ -136,6 +139,7 @@ jobs: benchmark-qt-dense/ benchmark-qt-large-batch/ benchmark-qt-cudagraphs/ + benchmark-qt-optimizer-cudagraphs/ !benchmark-qt/**/data/**/*.png !benchmark-gpu/**/data/**/*.png - name: Remove the disposable GPU environment diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index cc6450b..3e1157e 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -480,3 +480,17 @@ Explicit modes are recorded in JSON reports; process failures retain the exact arguments. Neither a mode flag nor a completed real-data run guarantees CUDA graph replay, convergence equivalence or a speedup. Compare all timings and memory against float under the same mode, rather than an older float baseline. + +### Model and optimizer graph comparison + +```bash +CUDA_VISIBLE_DEVICES=0 BENCHMARK_DATA_ROOT=examples \ + bash dev/check-benchmarks.sh qt-optimizer-cudagraphs /tmp/qt-optimizer-cudagraphs +``` + +This adds `--optimizer-cudagraphs` to the same three-seed, batch-512, 60-epoch +comparison for both float and INT8. The optional QT plus real-data GPU workflow +runs it alongside the existing profiles, publishes its summary, and retains +reports and failures for 90 days. Keep the older profiles as controls: optimizer +graph replay is opt-in and does not improve every workload. See the +[measured larger-batch results](../../docs/benchmarks.md#larger-batch-model-and-optimizer-graph-results). diff --git a/dev/check-benchmarks.sh b/dev/check-benchmarks.sh index 52c781b..35b2c11 100644 --- a/dev/check-benchmarks.sh +++ b/dev/check-benchmarks.sh @@ -9,7 +9,7 @@ if [[ -e "$results" ]]; then echo 'Results directory must be new.' >&2 exit 2 fi -case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch|qt-cudagraphs) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense, qt-large-batch or qt-cudagraphs.' >&2; exit 2 ;; esac +case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch|qt-cudagraphs|qt-optimizer-cudagraphs) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense, qt-large-batch, qt-cudagraphs or qt-optimizer-cudagraphs.' >&2; exit 2 ;; esac mkdir -p -- "$results" status=0 run_profile() { @@ -56,13 +56,17 @@ elif [[ "$mode" == qt-dense ]]; then --model-profile dense --optimizer sgd --learning-rate 0.3 --epochs 15 --batch-size 128 --compile \ --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --allow-nondeterministic "${quantization[@]}" done -elif [[ "$mode" == qt-large-batch || "$mode" == qt-cudagraphs ]]; then +elif [[ "$mode" == qt-large-batch || "$mode" == qt-cudagraphs || "$mode" == qt-optimizer-cudagraphs ]]; then compile_mode=() profile=mnist-large-batch - if [[ "$mode" == qt-cudagraphs ]]; then + if [[ "$mode" == qt-cudagraphs || "$mode" == qt-optimizer-cudagraphs ]]; then compile_mode=(--compile-mode reduce-overhead) profile=mnist-cudagraphs fi + if [[ "$mode" == qt-optimizer-cudagraphs ]]; then + compile_mode+=(--optimizer-cudagraphs) + profile=mnist-optimizer-cudagraphs + fi : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/}" for seed in 42 43 44; do precisions=(float int8) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 29d5647..81b860b 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -905,3 +905,47 @@ weight arithmetic inside the row kernel did not improve this workload. Further performance validation should also include larger compute workloads, where the integer GEMMs have more opportunity to offset update and scheduling overhead, and should retain alternating run order and report variability. + +### Larger-batch model and optimizer graph results + +The current implementation at `1152b78` was measured with the shared +`qt-cudagraphs` profile and a second pass enabling optimizer graph replay for +both precisions. Each pass used seeds 42, 43 and 44, alternating float/INT8 order +by seed, with the dense MNIST model, SGD (learning rate 0.3), batch 512, 60 epochs, +FP16 AMP, CPU cache and zero workers. Both precisions used model `reduce-overhead` +and optimizer compilation. Hardware was the RTX 3080 Ti Laptop GPU, PyTorch +2.12.0/CUDA 13.0, with one Torch thread and one Inductor compiler thread. + +| Seed | Optimizer graphs | Float / INT8 accuracy | Float / INT8 peak MiB | Float / INT8 median train epoch 3–60 s | Float / INT8 training wall s | +| --- | --- | --- | --- | --- | --- | +| 42 | Off | 93.06% / 92.82% | 221.73 / 167.91 | 0.0640 / 0.0568 | 38.26 / 48.87 | +| 43 | Off | 92.76% / 93.38% | 221.73 / 167.91 | 0.0581 / 0.0626 | 31.37 / 28.82 | +| 44 | Off | 92.74% / 92.40% | 221.73 / 167.91 | 0.0604 / 0.0606 | 32.51 / 26.41 | +| 42 | On | 93.06% / 92.82% | 221.74 / 171.92 | 0.0598 / 0.0577 | 29.11 / 28.14 | +| 43 | On | 92.76% / 93.38% | 221.74 / 171.92 | 0.0623 / 0.0565 | 34.21 / 27.80 | +| 44 | On | 92.74% / 92.40% | 221.74 / 171.92 | 0.0595 / 0.0566 | 33.12 / 26.55 | + +With optimizer replay enabled, INT8 used 22.5% less peak memory and its later +training phases were 3.6%, 9.3% and 4.9% faster than float with the same setting. +Whole-call times were also lower in all three pairs. Comparing INT8 replay with +the faster float median observed across either setting for each seed leaves +smaller advantages of 3.6%, 2.7% and 4.9%. These are measured benefits in this +compute profile, not guarantees for other models or hardware. Without optimizer +replay, the later-phase speed comparison was mixed. + +INT8-minus-float accuracy differences were −0.24, +0.62 and −0.34 percentage +points. All 12 runs completed checkpoint reload and held-out inference with +finite scores. Paired dataset manifests match; all runs share source hash +`23df40b2027a8a18de797f60c93c84aad72653c797a35e0f1883e9efcccbf2e4` +and the same lock hash. No tests or other benchmark runs overlapped these runs. +Caches were not cleared: first-use costs and sequential cache warming affect +whole-call comparisons, and these results do not establish cold-start speed or +statistical quality equivalence. Accuracy still has no automated real-data gate. + +Reports and predictions are retained locally in +`tmp-optimizer-cudagraphs/large-current` and `large-optimizer-graphs`. The second +configuration is now reproducible through the shared +`qt-optimizer-cudagraphs` mode and included in the optional QT plus real-data GPU +workflow, with visible summaries and retained artifacts. Use +[the shared command](../dev/benchmarks/README.md#model-and-optimizer-graph-comparison) +for future comparisons rather than treating these timings as fixed thresholds. diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index e5c70ec..2bbc233 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -70,7 +70,7 @@ def test_summary_preserves_failures_and_unmeasured_fields(tmp_path): assert "No reports produced" in summarize(Path(tmp_path / "missing")) -@pytest.mark.parametrize("mode", ["qt", "qt-large-batch", "qt-cudagraphs"]) +@pytest.mark.parametrize("mode", ["qt", "qt-large-batch", "qt-cudagraphs", "qt-optimizer-cudagraphs"]) def test_shared_harness_records_process_failures(tmp_path, mode): import os import shlex @@ -92,7 +92,7 @@ def test_shared_harness_records_process_failures(tmp_path, mode): timeout=30, ) assert result.returncode == 1 - prefix = "mnist-cudagraphs" if mode == "qt-cudagraphs" else "mnist-large-batch" + prefix = {"qt-cudagraphs": "mnist-cudagraphs", "qt-optimizer-cudagraphs": "mnist-optimizer-cudagraphs"}.get(mode, "mnist-large-batch") profiles = ( ("synthetic-float", "synthetic-int8") if mode == "qt" @@ -105,11 +105,12 @@ def test_shared_harness_records_process_failures(tmp_path, mode): assert report["quantized_training"] == ("int8" in profile) assert "test_accuracy" not in report arguments = report["arguments"] - if mode == "qt-cudagraphs": + if mode in ("qt-cudagraphs", "qt-optimizer-cudagraphs"): assert arguments[arguments.index("--compile-mode") + 1] == "reduce-overhead" assert "--compile" in arguments and "--compile-optimizer" in arguments else: assert "--compile-mode" not in arguments + assert ("--optimizer-cudagraphs" in arguments) == (mode == "qt-optimizer-cudagraphs") assert "requested" in (output / "summary.md").read_text() From a0a27c09803858afa5d541a92b80384d99490927 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 09:38:09 +0200 Subject: [PATCH 040/155] fix: respect container and Slurm CPU budgets for automatic workers Include visible cgroup v1/v2 ancestor quotas and per-task Slurm CPU allocations in the existing affinity-aware worker budget. Preserve explicit counts, reserve and caps; ignore malformed or unavailable limits. Validation: full suite 347 passed, 134 skipped, known EMA xfail; static checks. Repeated cached and real streaming loader probes produced identical batches and documented throughput gains. Two sandboxed focused runs were interrupted after worker IPC stalls; the full suite completed with multiprocessing access. --- dev/README.md | 15 +++++++ docs/benchmarks.md | 38 ++++++++++++++++ mini_trainer/data/_workers.py | 71 +++++++++++++++++++++++++++++- tests/utils/test_loader.py | 14 ++++-- tests/utils/test_workers.py | 83 +++++++++++++++++++++++++++++++++++ 5 files changed, 216 insertions(+), 5 deletions(-) create mode 100644 tests/utils/test_workers.py diff --git a/dev/README.md b/dev/README.md index c8391f1..426ab78 100644 --- a/dev/README.md +++ b/dev/README.md @@ -285,3 +285,18 @@ AMP skips, rate precision, same-optimizer restoration and eager checkpoint resum Capture eligibility, extra gradient copies, graph workspace memory and first-use compilation costs still depend on the optimizer and workload. Measure both float and INT8 with the same options; enabling graphs alone is not evidence of a speedup. + +## Automatic CPU budgets + +Automatic loader and cache worker counts now use the smallest detected process +CPU count, affinity mask, visible Linux cgroup CPU quota, and positive +`SLURM_CPUS_PER_TASK` allocation. Cgroup v1 and v2 ancestor limits are included; +fractional CPU quotas are rounded down before applying the existing four-CPU +reserve and worker caps. Explicit worker counts, including zero, remain unchanged. +Unreadable, unlimited, or malformed quota data falls back to the other signals. + +These are resource ceilings, not a measurement of contention from other jobs. +For a deliberately shared allocation, set worker counts explicitly when needed. +The relevant interfaces are documented by the +[Linux kernel](https://docs.kernel.org/admin-guide/cgroup-v2.html#cpu-interface-files) +and [Slurm](https://slurm.schedmd.com/sbatch.html#OPT_SLURM_CPUS_PER_TASK). diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 81b860b..1ff7856 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -949,3 +949,41 @@ configuration is now reproducible through the shared workflow, with visible summaries and retained artifacts. Use [the shared command](../dev/benchmarks/README.md#model-and-optimizer-graph-comparison) for future comparisons rather than treating these timings as fixed thresholds. + +### Loading audit on the delivered implementation + +The shared loader and reader probes were rerun after `d9ecab6`, using one Torch +thread, seven measured repetitions after warmup, and alternating reference/current +order. All comparisons verified exactly identical batches. + +| Path | Workers | Reference samples/s | Current samples/s | Ratio | +| --- | ---: | ---: | ---: | ---: | +| CPU cache, synthetic 64×64 | 0 | 241,783 | 625,168 | 2.59× | +| CPU cache, synthetic 64×64 | 1 | 40,090 | 43,843 | 1.09× | +| MNIST streaming, resized to 224×224 | 0 | 4,037 | 4,881 | 1.21× | +| MNIST streaming, resized to 224×224 | 1 | 2,447 | 3,444 | 1.41× | +| Blair streaming, resized to 224×224 | 0 | 2,309 | 2,643 | 1.14× | +| Blair streaming, resized to 224×224 | 1 | 1,956 | 2,448 | 1.25× | + +The cached probe used 2,048 generated uint8 images and batch 64, comparing scalar +fetch/stack against batched gathering. The reader probe used 128 real images per +dataset and batch 16 through `get_inference_dataloader`, comparing the retained +Torchvision resize path with the current reader. Worker runs used spawn. Timings +exclude worker startup, cache construction and model compute; the cached probe +also excludes decoding, and both exclude H2D. Reader filesystem caches were warm. +The streaming reader and batching implementation are shared by float and INT8 +inference; cached gathering also serves training. These ratios are loading-path +measurements, not end-to-end model throughput claims. + +Reports, input file hashes, all repetition times, and equality flags are retained +in `tmp-optimizer-cudagraphs/audit/{cached-*,reader-*}.json`. Reproduce with the +shared `dev.benchmarks.loader` and `dev.benchmarks.reader` commands documented in +[the benchmark guide](../dev/benchmarks/README.md). + +The audit also found that automatic worker selection still omitted CPU bandwidth +quotas and Slurm task allocations. The budget helper now includes cgroup v1/v2 +limits (including visible ancestors) and `SLURM_CPUS_PER_TASK`, while preserving +explicit overrides and the existing reserve/caps. Regression tests cover quota +and namespace parsing, fractional/unlimited/malformed limits, scheduler budgets, +and affinity fallbacks. This closes an allocation-aware defaulting gap; it does +not infer the instantaneous load or private CPU shares of competing processes. diff --git a/mini_trainer/data/_workers.py b/mini_trainer/data/_workers.py index e0e2bdc..ae5cab9 100644 --- a/mini_trainer/data/_workers.py +++ b/mini_trainer/data/_workers.py @@ -1,4 +1,62 @@ import os +import re +from pathlib import Path + + +def _read_text(path): + try: + return path.read_text() + except (OSError, UnicodeError): + return "" + + +def _cgroup_cpu_count(): + """Smallest visible v1/v2 bandwidth quota, including ancestor limits.""" + groups = {} + for line in _read_text(Path("/proc/self/cgroup")).splitlines(): + fields = line.split(":", 2) + if len(fields) == 3: + for controller in fields[1].split(","): + groups[controller] = fields[2] + limits = [] + for line in _read_text(Path("/proc/self/mountinfo")).splitlines(): + left, separator, right = line.partition(" - ") + fields, filesystem = left.split(), right.split() + if not separator or len(fields) < 5 or len(filesystem) < 3: + continue + if filesystem[0] == "cgroup2": + group = groups.get("") + filenames = ("cpu.max",) + elif filesystem[0] == "cgroup" and "cpu" in filesystem[2].split(","): + group = groups.get("cpu") + filenames = ("cpu.cfs_quota_us", "cpu.cfs_period_us") + else: + continue + if group is None: + continue + root, mount = (Path(re.sub(r"\\([0-7]{3})", lambda m: chr(int(m[1], 8)), value)) for value in fields[3:5]) + group = Path(group) + if not all(path.is_absolute() and ".." not in path.parts for path in (root, mount, group)): + continue + try: + relative = group.relative_to(root) + except ValueError: + # A cgroup namespace can expose paths relative to its own root. + relative = group.relative_to("/") + current = mount / relative + while True: + values = " ".join(_read_text(current / name) for name in filenames).split() + if len(values) == 2: + try: + quota, period = map(int, values) + if quota > 0 and period > 0: + limits.append(quota // period) + except ValueError: + pass # v2 "max", or malformed/unavailable quota data. + if current == mount: + break + current = current.parent + return min(limits) if limits else None def _available_cpu_count() -> int: @@ -14,7 +72,18 @@ def _available_cpu_count() -> int: counts.append(len(os.sched_getaffinity(0))) except (AttributeError, OSError, NotImplementedError): pass - return min(counts) if counts else (os.cpu_count() or 0) + if not counts: + counts.append(os.cpu_count() or 0) + quota = _cgroup_cpu_count() + if quota is not None: + counts.append(quota) + try: + allocated = int(os.environ.get("SLURM_CPUS_PER_TASK", "")) + if allocated > 0: + counts.append(allocated) + except ValueError: + pass + return min(counts) def _default_worker_count(cap: int, reserve: int = 4, minimum: int = 0) -> int: diff --git a/tests/utils/test_loader.py b/tests/utils/test_loader.py index ec454b3..c7cb195 100644 --- a/tests/utils/test_loader.py +++ b/tests/utils/test_loader.py @@ -124,7 +124,7 @@ def test_label_processing_and_hook(label, multilabel): @pytest.mark.parametrize("available,expected", [(1, 0), (4, 0), (7, 2), (8, 4), (64, 16)]) -def test_automatic_training_workers_respect_affinity(metadata, monkeypatch, available, expected): +def test_automatic_training_workers_respect_affinity(metadata, monkeypatch, available, expected, isolated_cpu_limits): monkeypatch.setattr(_workers.os, "cpu_count", lambda: 256) monkeypatch.setattr(_workers.os, "process_cpu_count", lambda: 256, raising=False) monkeypatch.setattr(_workers.os, "sched_getaffinity", lambda _: set(range(available)), raising=False) @@ -134,7 +134,7 @@ def test_automatic_training_workers_respect_affinity(metadata, monkeypatch, avai @pytest.mark.parametrize("process_count,affinity_count,expected", [(6, 128, 2), (128, 8, 4), (80, 80, 32)]) -def test_inference_uses_smaller_process_limit(metadata, monkeypatch, process_count, affinity_count, expected): +def test_inference_uses_smaller_process_limit(metadata, monkeypatch, process_count, affinity_count, expected, isolated_cpu_limits): monkeypatch.setattr(_workers.os, "process_cpu_count", lambda: process_count, raising=False) monkeypatch.setattr(_workers.os, "sched_getaffinity", lambda _: set(range(affinity_count)), raising=False) _, loader = get_inference_dataloader(metadata["path"], resize_size=4) @@ -142,14 +142,14 @@ def test_inference_uses_smaller_process_limit(metadata, monkeypatch, process_cou @pytest.mark.parametrize("host_count", [None, 1, 8]) -def test_worker_detection_fallbacks(monkeypatch, host_count): +def test_worker_detection_fallbacks(monkeypatch, host_count, isolated_cpu_limits): monkeypatch.delattr(_workers.os, "process_cpu_count", raising=False) monkeypatch.delattr(_workers.os, "sched_getaffinity", raising=False) monkeypatch.setattr(_workers.os, "cpu_count", lambda: host_count) assert _workers._available_cpu_count() == (host_count or 0) -def test_worker_detection_failed_affinity_and_unknown_process_count(monkeypatch): +def test_worker_detection_failed_affinity_and_unknown_process_count(monkeypatch, isolated_cpu_limits): def unavailable(_): raise OSError("Affinity unavailable") @@ -535,3 +535,9 @@ def test_default_collator_can_reuse_repository_batch_sampler(metadata, cache, wo # torch.stack's C-level sequence handling (which bypasses list methods). direct = dataset.__getitems__([4, 1, 1]) torch.testing.assert_close(torch.stack(direct), dataset[[4, 1, 1]]) + + +@pytest.fixture +def isolated_cpu_limits(monkeypatch): + monkeypatch.setattr(_workers, "_cgroup_cpu_count", lambda: None) + monkeypatch.delenv("SLURM_CPUS_PER_TASK", raising=False) diff --git a/tests/utils/test_workers.py b/tests/utils/test_workers.py new file mode 100644 index 0000000..63a5064 --- /dev/null +++ b/tests/utils/test_workers.py @@ -0,0 +1,83 @@ +"""Automatic worker budgets under container and scheduler CPU limits.""" + +import pytest + +from mini_trainer.data import _workers + + +def mock_cgroup(monkeypatch, files, membership="0::/team/task", mount="/", location="/sys/fs/cgroup", version=2): + filesystem = "cgroup2 cgroup rw" if version == 2 else "cgroup cgroup rw,cpu,cpuacct" + escaped = location.replace(" ", r"\040") + contents = { + "/proc/self/cgroup": membership, + "/proc/self/mountinfo": f"29 23 0:26 {mount} {escaped} rw - {filesystem}", + **files, + } + monkeypatch.setattr(_workers, "_read_text", lambda path: contents.get(str(path), "")) + + +def test_cgroup_v2_respects_parent_quota(monkeypatch): + mock_cgroup( + monkeypatch, + { + "/sys/fs/cgroup/team/task/cpu.max": "max 100000", + "/sys/fs/cgroup/team/cpu.max": "250000 100000", + "/sys/fs/cgroup/cpu.max": "1600000 100000", + }, + ) + assert _workers._cgroup_cpu_count() == 2 + + +@pytest.mark.parametrize( + "value,expected", + [("50000 100000", 0), ("600000 100000", 6), ("max 100000", None), ("-1 100000", None), ("1 0", None), ("garbled", None)], +) +def test_cgroup_v2_fractional_unlimited_and_invalid_limits(monkeypatch, value, expected): + mock_cgroup(monkeypatch, {"/sys/fs/cgroup/team/task/cpu.max": value}) + assert _workers._cgroup_cpu_count() == expected + + +@pytest.mark.parametrize("membership", ["0::/host/team/task", "0::/task"]) +def test_cgroup_mount_root_and_namespace_paths(monkeypatch, membership): + mock_cgroup( + monkeypatch, {"/limits cpu/task/cpu.max": "300000 100000"}, membership=membership, mount="/host/team", location="/limits cpu" + ) + assert _workers._cgroup_cpu_count() == 3 + + +def test_cgroup_v1_cpu_controller_and_parent_limit(monkeypatch): + mock_cgroup( + monkeypatch, + { + "/sys/fs/cgroup/team/task/cpu.cfs_quota_us": "-1", + "/sys/fs/cgroup/team/task/cpu.cfs_period_us": "100000", + "/sys/fs/cgroup/team/cpu.cfs_quota_us": "800000", + "/sys/fs/cgroup/team/cpu.cfs_period_us": "100000", + }, + membership="4:cpu,cpuacct:/team/task", + version=1, + ) + assert _workers._cgroup_cpu_count() == 8 + + +def test_missing_or_malformed_cgroup_metadata_is_ignored(monkeypatch): + monkeypatch.setattr(_workers, "_read_text", lambda _: "invalid metadata") + assert _workers._cgroup_cpu_count() is None + + +def test_unreadable_quota_is_ignored(tmp_path): + assert _workers._read_text(tmp_path / "missing") == "" + assert _workers._read_text(tmp_path) == "" + + +@pytest.mark.parametrize( + "quota,slurm,expected", + [(2, "64", 2), (32, "6", 6), (None, "7", 7), (None, "garbled", 64), (None, "0", 64), (None, "-1", 64), (128, "128", 64), (0, "8", 0)], +) +def test_available_cpus_use_smallest_valid_budget(monkeypatch, quota, slurm, expected): + monkeypatch.setattr(_workers.os, "process_cpu_count", lambda: 128, raising=False) + monkeypatch.setattr(_workers.os, "sched_getaffinity", lambda _: set(range(64)), raising=False) + monkeypatch.setattr(_workers, "_cgroup_cpu_count", lambda: quota) + monkeypatch.setenv("SLURM_CPUS_PER_TASK", slurm) + assert _workers._available_cpu_count() == expected + assert _workers._default_worker_count(32) <= max(0, expected - 4) From ad2ab9971e9fbfa6aa3d64433d539ef045774190 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 09:49:39 +0200 Subject: [PATCH 041/155] docs: consolidate native QT and loading completion evidence Record requirement-level validation for real INT8 arithmetic/storage, paired speed and memory gains, loading throughput, resource-aware worker defaults, synthetic and hierarchical inference, checkpoints and minimal packaging. Keep hardware, cold-start, operator and distributed limitations explicit. --- README.md | 8 ++- docs/quantized-training-validation.md | 79 +++++++++++++++++++++++++++ docs/quantized-training.md | 2 + docs/roadmap.md | 14 ++++- 4 files changed, 97 insertions(+), 6 deletions(-) create mode 100644 docs/quantized-training-validation.md diff --git a/README.md b/README.md index 244beb9..b0e812c 100644 --- a/README.md +++ b/README.md @@ -140,6 +140,8 @@ An opt-in [PTQ and QAT Python API](docs/quantization.md) targets native x86 INT8 inference. This is an initial backend increment; CPU float32 QAT, integer inference and ordinary AMP are distinct capabilities. -Opt-in [CUDA INT8 training](docs/quantized-training.md) now has an initial Linear -model/checkpoint integration. Its documented coverage and performance limits are -separate from [x86 PTQ/QAT inference](docs/quantization.md). +Opt-in [CUDA INT8 training](docs/quantized-training.md) supports Linear weights, +integer forward/backward products, checkpoint restoration and CUDA inference. +See the [validation audit](docs/quantized-training-validation.md) for measured +memory, speed and loading benefits, supported configurations and limitations. +[x86 PTQ/QAT inference](docs/quantization.md) is a separate backend. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md new file mode 100644 index 0000000..629c453 --- /dev/null +++ b/docs/quantized-training-validation.md @@ -0,0 +1,79 @@ +# Quantized training and loading validation + +This audit covers the delivered objective: genuine quantized training that lowers +memory and increases training speed on supported workloads, plus faster shared +loading for float and quantized training/inference. The capability is opt-in; +speedups are workload-dependent. Validation was completed on 2026-09-09 using +Python 3.13.7, PyTorch 2.12.0/CUDA 13.0, TorchAO 0.17.0 and an RTX 3080 Ti Laptop GPU. + +## Requirements and evidence + +| Requirement | Delivered behavior and evidence | +| --- | --- | +| Actual QT rather than AMP or fake quantization | Eligible Linear weights and saved linear inputs use INT8; forward, input-gradient and weight-gradient products use integer kernels. There is no retained floating master weight. The recipe reports actual coverage and physical parameter bytes. Numerical/storage tests are in `tests/test_quantized_training.py`; integration details are in [CUDA INT8 training](quantized-training.md). | +| Lower training memory and faster training | The [three-seed batch-512 MNIST comparison](benchmarks.md#larger-batch-model-and-optimizer-graph-results), with model and optimizer graph replay, used 22.5% less peak memory and had 3.6–9.3% faster later training phases than float under the same settings. Whole-call times were lower in all three pairs. Against the faster observed float setting per seed, later-phase advantages were 2.7–4.9%. | +| Quality and realistic model coverage | CPU/float-CUDA/INT8-CUDA synthetic runs each reached the 100% oracle. MNIST uses held-out images and three paired seeds. Fresh two-level Blair runs exercised normalized hierarchical heads, hidden Linear layers, MuonAuxAdamW, AMP, compilation, optimizer graph replay, checkpoint reload and inference. Details follow below. | +| Training-state compatibility | Tests cover SGD, AdamW and MuonAuxAdamW, AMP overflow gating, scheduler advancement, the composite optimizer counter, changing learning rates, checkpoint restoration and compiled graph replay. CUDA kernel and optimizer/model matrices passed before the final loader-only change; fresh real runs also passed afterward. | +| Faster training and inference loading | [Repeated loading probes](benchmarks.md#loading-audit-on-the-delivered-implementation) verified identical batches: cached gathering was 2.59× faster without workers and 1.09× with one worker; uncached real-image inference loading was 1.14–1.41× faster. The reader/batching paths are shared independently of model weight precision. | +| Controlled loading resources | Cache construction/read-ahead are bounded; worker selection honors affinity, process limits, visible cgroup quotas and Slurm task CPU allocations. Existing caps/reserve and explicit overrides remain. Tests cover zero workers, spawn, cache modes, sample order, batch ownership, distributed sampling and quota fallbacks. | +| Usable checkpoints and inference | The native QT recipe restores parameter types before loading weights. Synthetic, MNIST and Blair runs reloaded checkpoints and produced finite held-out scores with matching class/split contracts. Separate tests cover masked rows, normalized heads and single-sample CUDA inference. | +| Repeatable continuous validation and visible results | Shared commands run CPU, synthetic QT, real-data and multi-seed graph profiles. The optional GPU workflow includes `qt-optimizer-cudagraphs`, publishes summaries and retains reports/failures for 90 days. Configuration, source/lock hashes, dataset manifests, score semantics and coverage are recorded. Current measured results are committed in [benchmarks](benchmarks.md). | +| Installed-package/default-path safety | `bash dev/check.sh all`: 347 passed, 134 skipped, one existing EMA expected failure. Static checks passed. `bash dev/check-wheel.sh` passed minimal imports, CLI/resources, CPU training, reload and prediction in a disposable installed-wheel environment; the working CUDA environment was not synchronized. | + +## Fresh synthetic and hierarchical progression + +The final progression used `a0a27c0`. All synthetic profiles passed their explicit +100% oracle gate. Blair used the existing reviewed 25/15-class specification, +3,704 training images, 912 validation images and 1,161 held-out images. Both +precisions used seed 42, TinyConv with a 64-feature hidden head, five epochs, +batch 32, FP16 AMP, CPU cache, zero workers, model `reduce-overhead`, optimizer +compilation and optimizer graph replay. + +| Profile | Held-out fine-level accuracy | Parent accuracy | +| --- | ---: | ---: | +| Synthetic CPU | 100% | — | +| Synthetic float CUDA | 100% | — | +| Synthetic INT8 CUDA | 100% | — | +| Blair float | 64.86% | 78.12% | +| Blair INT8 | 66.24% | 81.40% | + +Blair INT8 coverage is `fc.hidden` and `fc.linear`, including the normalized +head's direction parameter. Convolutions, normalization magnitudes, biases and +optimizer state remain floating. Paired manifests match and all saved scores are +finite. These short Blair runs establish functional coverage, not statistical +quality superiority or a hierarchical training speedup. + +Local reports are retained in `tmp-optimizer-cudagraphs/audit/`, including the +loading repetitions, synthetic/Blair outputs, full-suite log and installed-wheel +log. Two focused loader attempts stalled in restricted-sandbox worker IPC and +were interrupted; the complete suite subsequently passed with multiprocessing +access. Reproduction commands live in [the benchmark guide](../dev/benchmarks/README.md) +and [development guide](../dev/README.md). + +## Limits of the delivered capability + +- Native QT execution is CUDA Linear-based. Other operations are reported as + floating or rejected for explicit unsupported selections. CPU preparation is + not CPU integer execution. CPU integer inference uses the separate [PTQ/QAT + backend](quantization.md). +- Small or convolution-dominated workloads can be slower under QT. Compiler + caches were not cleared for the paired timing study; it does not establish a + cold-start speed advantage. The reproduced first-use tuning failure was fixed + and validated with forced retuning, as documented in the benchmark history. +- Native QT ONNX export, checkpoint averaging, DDP/FSDP, quantized activation + normalization and EMA are unsupported. Native fused optimizers are not claimed + for INT8 weights. Gradients and optimizer state remain floating point. +- Controlled checkpoint continuation is covered; arbitrary stochastic resume + does not promise identical trajectories because RNG/sampler state is not + generally checkpointed. Real-data completion has no automatic quality gate. +- Loading ratios isolate the stated loading paths; they are not end-to-end + inference speedups. CPU resource ceilings do not reveal contention from other + jobs or partition a shared allocation automatically. +- Optional GPU CI requires an appropriate runner and datasets. The shared runs + were executed locally; this audit does not claim that a remote GPU workflow + has been dispatched or completed. Birds and iNaturalist remain stretch work. + +These limits qualify the supported capability; broader hardware/operator +coverage, optimizer-state quantization and further model-quality studies remain +future increments. The requested native QT and shared-loader objective has +verified implementations and measured benefits within the supported regime. diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 163d43d..8ee7106 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -10,6 +10,8 @@ Install the optional `quantization` extra while explicitly retaining the intende PyTorch CUDA backend, as described in the README. The current implementation uses TorchAO's experimental Triton kernels. It has been exercised on an RTX 3080 Ti; CPU preparation and checkpoint inspection do not establish CPU execution support. +See the [validation audit](quantized-training-validation.md) for current tests, +measured training/loading benefits and the limits of those results. ```bash mt_train -i /path/to/data --device cuda --quantized-training --dtype float16 --cache cpu --cache-workers 0 diff --git a/docs/roadmap.md b/docs/roadmap.md index e4be429..6445a74 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -88,15 +88,23 @@ explicit upload commands or a serving deployment. ## 4. Training efficiency and augmentation +The native QT and shared-loader increment is delivered for the documented CUDA +Linear regime. The [validation audit](quantized-training-validation.md) records +three-seed memory/speed benefits, real loading comparisons, synthetic and +hierarchical checkpoint/inference coverage, allocation-aware worker defaults, +and installed-package checks. Broader hardware/operator coverage, cold-start +performance and additional quality studies remain future work. The chronological +results below retain earlier failures and mixed comparisons. + The primary implementation target is **actual quantized training and faster data loading**. QT must reduce retained training storage and demonstrate lower peak memory and faster training on supported workloads. QAT with floating-point master weights is a separate capability and does not complete this target. The initial CUDA integer forward/backward kernel probe and cached-loader benchmark are documented in [the benchmark guide](../dev/benchmarks/README.md). An initial [CUDA INT8 Linear integration](quantized-training.md) connects model -preparation and checkpoint loading to the training entry point. Broader operator +preparation and checkpoint loading to the training entry point. At that stage, broader operator and optimizer coverage, stronger convergence evidence and real-workload speedups -remain required. Paired synthetic, MNIST and hierarchical Blair smoke runs now +remained required. Paired synthetic, MNIST and hierarchical Blair smoke runs now record quality, storage and timing; [these small workloads are slower under QT](benchmarks.md#integrated-int8-training). Loader hardening, float16/bfloat16 AMP and benchmark infrastructure do not complete that target. The implementation and comparison plan is in @@ -120,7 +128,7 @@ confirms 26–28% lower peak memory but mixed speed results and 0.10–0.54 perc points lower accuracy. [Functional fused requantization](benchmarks.md#functional-fused-requantization) then reduced peak memory to 30–31% below float, with slightly higher accuracy in all three pairs. Whole-run times improved, but later-phase speed remained mixed. -Reliable speed gains, cold-start cost, and broader workload validation remain open. +At that stage, reliable speed gains, cold-start cost, and broader workload validation remained open; see the current audit above. Deliver quantization-aware training and post-training inference quantization as separate opt-in capabilities, recording actual weight/activation bit widths, From 1e9e27d5aa442bf91d21c22ff1804e325b276024 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 10:11:17 +0200 Subject: [PATCH 042/155] docs: target EfficientNetV2 quantization at production hardware --- docs/quantized-training-validation.md | 66 +++++++++++++++++++++++++-- docs/roadmap.md | 11 +++-- 2 files changed, 69 insertions(+), 8 deletions(-) diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 629c453..627f56c 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -1,11 +1,69 @@ # Quantized training and loading validation -This audit covers the delivered objective: genuine quantized training that lowers -memory and increases training speed on supported workloads, plus faster shared -loading for float and quantized training/inference. The capability is opt-in; -speedups are workload-dependent. Validation was completed on 2026-09-09 using +This audit records an initial Linear-only milestone, not completion of the +representative-model or deployment objective. It covers genuine quantized training +and shared loading for float and quantized training/inference. The capability is +opt-in; speedups are workload-dependent. Local validation was recorded on 2026-09-09 using Python 3.13.7, PyTorch 2.12.0/CUDA 13.0, TorchAO 0.17.0 and an RTX 3080 Ti Laptop GPU. +## Primary model and deployment targets + +The primary model is EfficientNetV2 with a symmetric hidden layer (`hidden=True`) +and a normalized `Classifier` or `HierarchicalClassifier`. Start with the +`efficientnet_v2_s` configuration used in `examples/blair.ipynb`, with both heads +on the same reviewed Blair splits. The dense MNIST model is a kernel diagnostic; +TinyConv on Blair is an integration check. Neither establishes performance or +quality for the primary model. + +| Deployment target | Execution path to validate | Required measurements | +| --- | --- | --- | +| HPC: A40, A100, B300-class GPU systems with AMD EPYC hosts | PyTorch GPU training | End-to-end training time, steady-state throughput, allocated/reserved GPU peaks, host memory, loading/transfer costs, convergence and checkpoint/resume; record actual GPU, allocation, precision and kernels separately for each system. | +| Local batch processing: NVIDIA Spark or the intended RTX desktop with Ryzen 7 9800X3D | PyTorch training/fine-tuning; ONNX GPU inference | Full and frozen-backbone training separately; export parity, actual execution-provider placement, batch throughput, latency and memory including preprocessing/transfers. Exact installed device and runtime support must be verified. | +| Edge integration: Raspberry Pi or similar | ONNX CPU inference | Export/score parity, accuracy, model size, process memory, batch-one latency and sustained throughput on the actual ARM device, with explicit thread settings. | + +These are acceptance targets, not claims of hardware or backend support. Laptop +measurements remain useful for debugging but cannot establish gains on these +systems. Do not assume a single quantized artifact or kernel recipe works across +CUDA PyTorch, ONNX GPU and ONNX ARM CPU. The current native QT checkpoint cannot +be exported directly through the ONNX path. ONNX Runtime training/fine-tuning +would be a separate integration; an inference export does not provide it. + +Next extend the shared dataset harness to select backbone and head independently, +preserving existing defaults. Record initialization/pretrained provenance, +symmetric width, normalization, image size and all training settings. Compare +float and quantized paths with identical splits and paired seeds, reporting +quantized versus floating operators and physical storage before making speed +claims. Profile the real backbone before selecting convolution or other storage +reductions. Keep kernel probes separate from full training and deployment results. + +Accuracy superiority has not been demonstrated. The three paired dense MNIST +accuracy differences were -0.24, +0.62 and -0.34 percentage points (INT8 minus +float); Blair below is one short seed. Agree quality tolerances before evaluating +candidate recipes, retain negative results, and report uncertainty separately +from speed and memory measurements. Target-machine runs and representative-model +comparisons remain outstanding. + +### EfficientNetV2-S structural coverage probe + +On 2026-09-09, CPU-only preparation through `Classifier.build` and +`HierarchicalClassifier.build`, using `model_type="efficientnet_v2_s"`, +`num_classes=25`, `hidden=True`, `normalized=True` and +`model_args={"pretrained": False}`, produced the same storage counts: + +| Quantity | Bytes / count | +| --- | ---: | +| All floating parameters before preparation (FP32) | 87,407,112 bytes | +| Selected head weights before preparation | 6,681,600 bytes | +| Selected head INT8 weights including row scales | 1,675,620 bytes | +| Reduction relative to all parameter bytes | 5.73% | +| Convolutions left floating | 170 | + +`prepare_quantized_training(model)` selected only `classifier.hidden` and +`classifier.linear`. This constructed random models without downloading weights; +the hierarchical construction did not supply taxonomy masks or execute a forward +pass. It establishes operator/parameter coverage only, not hierarchical runtime +correctness, activation storage, peak training memory, convergence or throughput. + ## Requirements and evidence | Requirement | Delivered behavior and evidence | diff --git a/docs/roadmap.md b/docs/roadmap.md index 6445a74..cb4b80b 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -88,12 +88,15 @@ explicit upload commands or a serving deployment. ## 4. Training efficiency and augmentation -The native QT and shared-loader increment is delivered for the documented CUDA -Linear regime. The [validation audit](quantized-training-validation.md) records +An initial native QT and shared-loader milestone is implemented for the documented +CUDA Linear regime. The [validation audit](quantized-training-validation.md) records three-seed memory/speed benefits, real loading comparisons, synthetic and hierarchical checkpoint/inference coverage, allocation-aware worker defaults, -and installed-package checks. Broader hardware/operator coverage, cold-start -performance and additional quality studies remain future work. The chronological +and installed-package checks. The primary EfficientNetV2 configuration with symmetric +hidden layers and normalized flat/hierarchical heads remains unvalidated. HPC +PyTorch training, local GPU training/ONNX inference and ARM ONNX edge inference +are required deployment targets, not extensions of a completed objective. +Broader operator coverage, cold-start performance and quality studies remain open. The chronological results below retain earlier failures and mixed comparisons. The primary implementation target is **actual quantized training and faster data loading**. From c06470c1f39224d9a07a62836a63e8fc3439496c Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 10:19:33 +0200 Subject: [PATCH 043/155] feat: benchmark representative backbones with matched classifier heads --- dev/benchmarks/README.md | 76 +++++++++++++++++++++++++++ dev/benchmarks/run.py | 75 ++++++++++++++++++++++---- docs/quantized-training-validation.md | 10 ++-- tests/test_benchmark_datasets.py | 55 +++++++++++++++++++ 4 files changed, 204 insertions(+), 12 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 3e1157e..1734735 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -283,6 +283,82 @@ pinned CPU transfer batches and exact CPU/CUDA cache contents. ## Integrated QT dataset profiles +### Representative EfficientNetV2 configuration + +The runner accepts a registered `--backbone` independently of the dataset and +`--head flat|hierarchical`. `--hidden symmetric` uses the backbone embedding width; +`--normalized` enables the normalized head for flat classification too. Existing +benchmark defaults are preserved. Explicit backbones start without pretrained +weights unless `--pretrained` is supplied; that flag permits a download. Reload +uses the saved checkpoint without requesting pretrained initialization again. + +For a matched Blair comparison on an available CUDA machine: + +```bash +for head in flat hierarchical; do + for precision in float int8; do + qt_args=() + if [[ "$precision" == int8 ]]; then qt_args=(--quantized-training); fi + CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ + .venv/bin/python -m dev.benchmarks.run \ + --dataset blair --data-root examples/blair \ + --class-spec examples/blair/blair_model/class_spec.json \ + --backbone efficientnet_v2_s --head "$head" --hidden symmetric --normalized \ + --image-size 128 --epochs 5 --batch-size 32 --seed 42 \ + --device cuda:0 --dtype float16 --cache CPU --cache-workers 0 --num-workers 0 \ + "${qt_args[@]}" --output "tmp-efficientnet/$head-$precision" + done +done +``` + +Each output directory must be new. This is an eager baseline; add identical +compilation settings to both precisions when measuring compilation. Repeat paired +seeds with alternating precision order before interpreting quality or speed. +Both heads use identical leaf indices and the same dataset manifest, including +the reviewed parent taxonomy. Flat training projects the taxonomy to the leaf +mapping; it does not regenerate splits or reorder classes. Reports retain the +backbone, effective hidden width, normalization, image size and initialization +choice. Held-out inference currently checks scores and quality; it is not an +ONNX latency benchmark. + +For offline pipeline verification use `--device cpu --dtype float32` without +`--quantized-training`. Smaller image sizes or short runs may check execution, +but must be reported as diagnostic settings rather than the production workload. +The CPU integration test exercises both real EfficientNetV2 heads with synthetic +image files and a deliberately reordered taxonomy; it checks training, reload, +predictions, class order and identical split manifests without a download. + +### Validation when target hardware is unavailable + +Use the local GPU to vary batch size, resolution, cache mode and worker count +one at a time. Compare each quantized run with the same floating configuration; +record out-of-memory failures, conversion costs and floating operator coverage. +Smaller memory budgets and constrained CPU affinity can exercise resource limits, +but cannot reproduce a different GPU architecture, bandwidth or ARM instruction +set. Keep compiler warmup separate from steady-state measurements. The CPU suite +is a correctness check, not an estimate of Raspberry Pi latency. + +When machines become available, use the same revision, lock file, dataset +manifest, checkpoints and commands, with a separately selected compatible backend: + +- HPC: run paired PyTorch training on the allocated GPU/CPU resources, recording + exact hardware, CPU affinity/quota, software versions and whether storage is + local or shared. Measure each GPU generation separately. Multi-GPU QT remains + unsupported and requires its own implementation and validation. +- Local batch processing: repeat training and frozen-backbone fine-tuning as + distinct workloads. Then export and measure ONNX GPU inference with provider + placement evidence; a listed provider alone does not prove GPU execution. +- Edge: transfer the export, preprocessing recipe, class mapping and fixed input + samples to the ARM device. Verify scores before measuring batch-one latency, + sustained throughput and process memory under explicit thread counts. + +The current native QT checkpoint is not ONNX-exportable. Establish floating +EfficientNetV2 export parity first, then implement and validate an appropriate +deployment quantization path for each provider. Export/runtime work and target +machine verification remain open; these commands do not establish those results. +Retain reports, logs, failures and predictions with the existing benchmark +summary/artifact workflow so external runs can be reviewed without machine access. + Install the optional `quantization` extra with the intended CUDA backend explicitly selected (see the repository README), then use new output directories: diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 7ef1147..0054d88 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -6,7 +6,7 @@ import platform import subprocess import time -from argparse import ArgumentParser +from argparse import ArgumentParser, BooleanOptionalAction from datetime import UTC, datetime from importlib.metadata import version from pathlib import Path @@ -18,7 +18,7 @@ from mini_trainer.data import get_inference_dataloader from mini_trainer.hierarchical.integration import HierarchicalBuilder from mini_trainer.hierarchical.model import HierarchicalClassifier -from mini_trainer.modeling import Classifier +from mini_trainer.modeling import Classifier, classification_module from mini_trainer.train import main as train from mini_trainer.training import MuonAuxAdamW from mini_trainer.training.compilation import MODEL_COMPILE_MODES, model_compile_options, validate_optimizer_compilation @@ -56,7 +56,12 @@ def run( quantized_training: bool = False, compile: bool = False, compile_optimizer: bool = False, - hidden: int = 0, + hidden: bool | int = 0, + backbone: str | None = None, + head: str = "auto", + normalized: bool | None = None, + image_size: int | None = None, + pretrained: bool = False, batch_size: int = 32, cache_workers: int | None = None, model_profile: str = "default", @@ -69,6 +74,20 @@ def run( validate_optimizer_compilation(compile_optimizer, optimizer_cudagraphs, device) if model_profile not in ("default", "dense") or optimizer not in ("muon", "adamw", "sgd"): raise ValueError("Unknown model or optimizer profile.") + if backbone is not None and model_profile != "default": + raise ValueError("Choose an explicit backbone or a model profile, not both.") + if head not in ("auto", "flat", "hierarchical"): + raise ValueError("Unknown head type.") + hierarchical = head == "hierarchical" or (head == "auto" and dataset == "blair") + if hierarchical and dataset != "blair": + raise ValueError("Hierarchical benchmarks require the reviewed Blair taxonomy.") + normalized = hierarchical if normalized is None else normalized + if hierarchical and not normalized: + raise ValueError("HierarchicalClassifier requires normalization.") + if image_size is not None and image_size < 1: + raise ValueError("Image size must be positive.") + if pretrained and backbone is None: + raise ValueError("Pretrained initialization requires an explicit backbone.") if learning_rate is None: learning_rate = 0.1 if dataset == "synthetic" else 0.01 if not np.isfinite(learning_rate) or learning_rate <= 0: @@ -119,12 +138,18 @@ def run( manifest, spec = prepare_real(root, output, name=dataset, seed=seed, class_spec=Path(class_spec) if class_spec else None) manifest_path = output / "dataset_manifest.json" spec_path = output / "class_spec.json" - spec_path.write_text(json.dumps(spec, indent=2) + "\n") + # Preserve the taxonomy in the manifest, but train the flat head using + # its exact leaf indices so paired runs share both splits and class order. + training_spec = spec + if dataset == "blair" and not hierarchical: + training_spec = {"num_classes": spec["num_classes"][0], "cls2idx": spec["cls2idx"]["0"]} + spec_path.write_text(json.dumps(training_spec, indent=2) + "\n") size = 28 if dataset == "mnist" else 64 model_type = "dev.benchmarks.models:TinyConv" if model_profile == "dense": model_type = "dev.benchmarks.models:DenseImageMLP" - hierarchical = dataset == "blair" + model_type = backbone or model_type + size = image_size or size records = manifest["records"] train_records = [record for record in records if record["split"] != "test"] data_index = output / "train_index.json" @@ -160,7 +185,8 @@ def run( model_builder_kwargs={ "model_type": model_type, "hidden": hidden if hidden else False, - "normalized": hierarchical, + "normalized": normalized, + **({"model_args": {"pretrained": pretrained}} if backbone else {}), "cls": HierarchicalClassifier if hierarchical else Classifier, }, dataloader_builder_kwargs={ @@ -192,7 +218,12 @@ def run( else None ) weights = output / "training/weights/last.pt" - model, preprocess = Classifier.build(weights=str(weights), device=target_device, dtype=torch.float32) + model, preprocess = Classifier.build( + weights=str(weights), + device=target_device, + dtype=torch.float32, + **({"model_args": {"pretrained": False}} if backbone else {}), + ) model.eval() quantization_recipe = getattr(model, "_quantized_training_recipe", None) if quantized_training and not quantization_recipe: @@ -247,6 +278,12 @@ def run( }, "dataset": dataset, "model_profile": model_profile, + "backbone": model_type, + "head": "hierarchical" if hierarchical else "flat", + "normalized": normalized, + "image_size": size, + "pretrained": pretrained, + "hidden_width": classification_module(model).preclassification_size, "optimizer": optimizer, "learning_rate": learning_rate, "momentum": 0.9 if optimizer == "sgd" else None, @@ -306,7 +343,7 @@ def run( "dataset_manifest_sha256": hashlib.sha256(manifest_path.read_bytes()).hexdigest(), "checkpoint_sha256": hashlib.sha256(weights.read_bytes()).hexdigest(), "score_semantics": "model_eval_forward", - "class_mapping": model.fc.metadata["cls2idx"], + "class_mapping": classification_module(model).metadata["cls2idx"], "test_used_for_training_or_selection": False, } (output / "report.json").write_text(json.dumps(result, indent=2) + "\n") @@ -336,7 +373,17 @@ def main(): parser.add_argument("--model-profile", choices=["default", "dense"], default="default") parser.add_argument("--optimizer", choices=["muon", "adamw", "sgd"], default="muon") parser.add_argument("--learning-rate", type=float) - parser.add_argument("--hidden", type=int, default=0) + parser.add_argument( + "--hidden", + type=lambda value: True if value == "symmetric" else int(value), + default=0, + help="0 disables the hidden layer; symmetric uses the backbone width; otherwise a positive width.", + ) + parser.add_argument("--backbone", help="Registered model name, for example efficientnet_v2_s.") + parser.add_argument("--head", choices=["auto", "flat", "hierarchical"], default="auto") + parser.add_argument("--normalized", action=BooleanOptionalAction, default=None) + parser.add_argument("--image-size", type=int) + parser.add_argument("--pretrained", action="store_true", help="Allow downloading pretrained backbone weights.") parser.add_argument("--batch-size", type=int, default=32) parser.add_argument( "--allow-nondeterministic", @@ -370,6 +417,11 @@ def main(): compile_optimizer=args.compile_optimizer, optimizer_cudagraphs=args.optimizer_cudagraphs, hidden=args.hidden, + backbone=args.backbone, + head=args.head, + normalized=args.normalized, + image_size=args.image_size, + pretrained=args.pretrained, batch_size=args.batch_size, cache_workers=args.cache_workers, model_profile=args.model_profile, @@ -394,6 +446,11 @@ def main(): "compile_optimizer": args.compile_optimizer, "optimizer_cudagraphs": args.optimizer_cudagraphs, "hidden": args.hidden, + "backbone": args.backbone, + "head": args.head, + "normalized": args.normalized, + "image_size": args.image_size, + "pretrained": args.pretrained, "batch_size": args.batch_size, "cache_workers": args.cache_workers, "model_profile": args.model_profile, diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 627f56c..c528a5a 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -28,9 +28,13 @@ CUDA PyTorch, ONNX GPU and ONNX ARM CPU. The current native QT checkpoint cannot be exported directly through the ONNX path. ONNX Runtime training/fine-tuning would be a separate integration; an inference export does not provide it. -Next extend the shared dataset harness to select backbone and head independently, -preserving existing defaults. Record initialization/pretrained provenance, -symmetric width, normalization, image size and all training settings. Compare +The shared dataset harness now selects backbone and head independently while +preserving existing defaults. A CPU integration test trains and reloads both +EfficientNetV2-S heads on synthetic image files with a reviewed test taxonomy, +checking identical split manifests and leaf-class ordering. Reports record +initialization choice, symmetric width, normalization and image size; commands +and the target-machine handoff are in the [benchmark guide](../dev/benchmarks/README.md). +Next compare float and quantized paths with identical splits and paired seeds, reporting quantized versus floating operators and physical storage before making speed claims. Profile the real backbone before selecting convolution or other storage diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index 2bbc233..eaedfec 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -191,3 +191,58 @@ def test_summary_later_epoch_median_requires_timing_scope(tmp_path): report["phase_measurements"] = report["phase_measurements"][:2] path.write_text(json.dumps(report)) assert "| — | 123.00s |" in summarize(tmp_path) + + +def test_efficientnet_flat_and_hierarchical_share_blair_splits(tmp_path): + import numpy as np + import torch + + from dev.benchmarks.run import run + + root = tmp_path / "images" + make_dataset(root) + spec = { + "labels": {"a": ["species_a", "parent_a"], "b": ["species_b", "parent_b"]}, + "num_classes": [2, 2], + "cls2idx": {"0": {"species_b": 0, "species_a": 1}, "1": {"parent_a": 0, "parent_b": 1}}, + } + spec_path = tmp_path / "taxonomy.json" + spec_path.write_text(json.dumps(spec)) + reports = [] + previous_threads = torch.get_num_threads() + torch.set_num_threads(1) + try: + for head in ("flat", "hierarchical"): + reports.append( + run( + tmp_path / head, + epochs=1, + dataset="blair", + data_root=root, + class_spec=spec_path, + backbone="efficientnet_v2_s", + head=head, + hidden=True, + normalized=True, + image_size=32, + batch_size=4, + cache_workers=0, + ) + ) + finally: + torch.set_num_threads(previous_threads) + flat, hierarchical = reports + assert flat["dataset_manifest_sha256"] == hierarchical["dataset_manifest_sha256"] + assert flat["class_mapping"] == spec["cls2idx"]["0"] + assert hierarchical["class_mapping"]["0"] == flat["class_mapping"] + assert flat["hidden_width"] == hierarchical["hidden_width"] == 1280 + assert all(report["normalized"] and not report["pretrained"] for report in reports) + with np.load(tmp_path / "flat/predictions.npz") as a, np.load(tmp_path / "hierarchical/predictions.npz") as b: + np.testing.assert_array_equal(a["paths"], b["paths"]) + np.testing.assert_array_equal(a["labels"], b["labels"]) + assert a["scores"].shape == b["scores"].shape == (2, 2) + assert b["scores_1"].shape == (2, 2) + indices = [json.loads((tmp_path / head / "train_index.json").read_text()) for head in ("flat", "hierarchical")] + assert indices[0]["path"] == indices[1]["path"] + assert indices[0]["split"] == indices[1]["split"] + assert indices[0]["class"] == [labels[0] for labels in indices[1]["class"]] From f14f62df8091078f95f8209f3413eb7531a66480 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 10:29:14 +0200 Subject: [PATCH 044/155] docs: record EfficientNetV2 head-only QT regressions on Blair --- docs/benchmarks.md | 45 +++++++++++++++++++++++++++ docs/quantized-training-validation.md | 8 +++++ 2 files changed, 53 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 1ff7856..3939cab 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -987,3 +987,48 @@ explicit overrides and the existing reserve/caps. Regression tests cover quota and namespace parsing, fractional/unlimited/malformed limits, scheduler budgets, and affinity fallbacks. This closes an allocation-aware defaulting gap; it does not infer the instantaneous load or private CPU shares of competing processes. +## EfficientNetV2-S on Blair: initial representative-model comparison + +On 2026-09-09, revision `c06470c` completed four local CUDA runs using the actual +EfficientNetV2-S backbone with a symmetric 1280-feature hidden layer and normalized +flat or hierarchical classifiers. All four used the same reviewed Blair manifest +(3,704 train / 912 validation / 1,161 test), seed 42, five epochs, size 128, batch +32, MuonAuxAdamW at head LR 0.01, FP16 AMP, CPU cache, zero workers, no augmentation +and no model/optimizer compilation. Backbone initialization was random, not +pretrained; these are execution/convergence diagnostics rather than a reproduction +of the notebook's pretrained training recipe. + +| Head | Precision | Fine accuracy | Parent accuracy | Peak allocated MiB | Median train phase, epochs 3–5 (s) | Training call (s) | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| Flat | Float | 56.33% | — | 1230.15 | 13.027 | 75.706 | +| Flat | INT8 | 34.80% | — | 1191.86 | 15.433 | 102.584 | +| Hierarchical | Float | 48.23% | 62.62% | 1230.16 | 15.061 | 93.804 | +| Hierarchical | INT8 | 46.43% | 60.12% | 1188.37 | 16.457 | 99.235 | + +The current head-only recipe reduced whole-model parameter storage from 87,407,112 +to 82,401,132 bytes (5.73%). Peak allocated memory fell by about 3.1% for the flat +head and 3.4% for the hierarchical head. Later training phases were approximately +18.5% and 9.3% slower respectively. All 170 convolutions remained floating. + +The flat accuracy regression is material: -21.53 percentage points, versus -1.81 +points for hierarchical fine accuracy and -2.50 for parent accuracy. One short +seed does not establish the cause or statistical generality, and matching seed +values does not guarantee identical stochastic training trajectories under QT. +Nevertheless, these results do not support recommending this recipe for the +representative model. Next compare pretrained initialization and inspect the +optimization/numerical behavior before interpreting broader quality effects. + +All four runs trained, reloaded their checkpoints and produced finite held-out +scores. Source, lock and dataset manifest hashes match across the four reports. +Source SHA256 is `8da1c2dbc1218d4d655925e0497fb0598ecd48bcafa6197c4ec8d5328fbb2bf6`; +manifest SHA256 is `1cef6c7d9133d889b6c8eff0ffef29b1db5d6653ab4adf0c005739af8c2921c6`. +Reports, logs, checkpoints and predictions are retained locally under ignored +`tmp-efficientnet-baseline/`. Commands are in the +[benchmark guide](../dev/benchmarks/README.md#representative-efficientnetv2-configuration). + +Hardware was the RTX 3080 Ti Laptop GPU; runtime versions are recorded per report. +Runs were sequential in flat-float, flat-INT8, hierarchical-float, +hierarchical-INT8 order without clearing compiler caches. Timings include local +loading/logging effects and do not predict A40/A100/B300 or Spark/desktop speed. +The missing optional dendrogram visualization dependencies emitted warnings; +training and prediction completed. No ONNX or target-hardware inference was tested. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index c528a5a..7be8f6b 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -47,6 +47,14 @@ candidate recipes, retain negative results, and report uncertainty separately from speed and memory measurements. Target-machine runs and representative-model comparisons remain outstanding. +The first [representative EfficientNetV2-S comparison](benchmarks.md#efficientnetv2-s-on-blair-initial-representative-model-comparison) +now completes training/reload/inference for both heads and precisions on real +Blair data with random initialization. It shows only 3.1–3.4% lower peak allocation, +slower training and lower accuracy under current head-only QT, including a large +flat-head regression. This is evidence against recommending the present recipe, +not completion of the speed/quality target. Pretrained runs and numerical +investigation are next; target-machine measurements and ONNX deployment remain open. + ### EfficientNetV2-S structural coverage probe On 2026-09-09, CPU-only preparation through `Classifier.build` and From 8732cbccb5cf9cc1961faaf3e24cf4c233fef684 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 10:38:52 +0200 Subject: [PATCH 045/155] docs: compare pretrained EfficientNetV2 QT and initial gradients --- docs/benchmarks.md | 46 +++++++++++++++++++++++++++ docs/quantized-training-validation.md | 11 +++++-- 2 files changed, 55 insertions(+), 2 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 3939cab..ef0a771 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1032,3 +1032,49 @@ hierarchical-INT8 order without clearing compiler caches. Timings include local loading/logging effects and do not predict A40/A100/B300 or Spark/desktop speed. The missing optional dendrogram visualization dependencies emitted warnings; training and prediction completed. No ONNX or target-hardware inference was tested. + +### Pretrained initialization and fixed-batch numerical check + +The matched pretrained comparison at revision `f14f62d` used the same settings, +source hash and dataset manifest as above, adding `--pretrained`. This loaded +Torchvision's cached `efficientnet_v2_s-dd5fe13b.pth` backbone, with a newly +initialized symmetric normalized head. Execution order was INT8 then float for +each head; no other runs overlapped the measurements. +The pretrained file SHA256 is +`dd5fe13b1d60ec15317ccc8ca158186e134d3366c3dde9cb9a4e301f2dc66c74`. + +| Head | Precision | Fine accuracy | Parent accuracy | Peak allocated MiB | Median train phase, epochs 3–5 (s) | Training call (s) | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| Flat | Float | 83.46% | — | 1234.15 | 14.673 | 89.091 | +| Flat | INT8 | 83.63% | — | 1191.36 | 15.291 | 88.937 | +| Hierarchical | Float | 78.21% | 90.61% | 1234.16 | 14.558 | 92.052 | +| Hierarchical | INT8 | 80.62% | 91.30% | 1191.37 | 14.994 | 95.987 | + +The large flat accuracy regression did not recur with pretrained initialization +in this seed. That neither establishes superiority nor identifies the cause of +the random-initialization regression. INT8 later training phases remained 4.2% +slower for the flat head and 3.0% slower for the hierarchical head; peak allocation +was about 3.5% lower. Whole-call timing is mixed. All checkpoints reloaded and +produced finite held-out scores; coverage is still limited to the two head layers. +Reports and predictions are retained under ignored `tmp-efficientnet-pretrained/`. + +After those runs, a separate diagnostic compared deep-copied flat models before +any optimizer update on 32 identical Blair training images, sampled with seed 42. +Dropout and stochastic depth were disabled, BatchNorm remained in training mode, +and both used FP16 AMP with loss scaled by 1024 for backward. This isolates an +initial forward/backward discrepancy, not stochastic update or convergence behavior. + +| Initialization | Float / INT8 loss | Score relative L2 error | Backbone gradient relative L2 error | Hidden gradient relative L2 error | Output gradient relative L2 error | +| --- | --- | ---: | ---: | ---: | ---: | +| Random | 3.74278 / 3.74148 | 1.43% | 4.68% | 4.61% | 1.31% | +| Pretrained | 3.77631 / 3.77755 | 1.92% | 9.10% | 9.61% | 1.53% | + +Relative errors are L2 difference divided by the floating reference norm, +aggregated over each parameter group. The initial pretrained gradient discrepancy +was larger despite its better eventual accuracy: these initial errors alone do +not explain the earlier result. Preparation uses deterministic rounding; subsequent +updates use stochastic rounding and share the CUDA RNG with stochastic layers. +Further investigation should separate repeated-seed variation, optimizer updates +and stochastic-layer effects before changing training numerics. The diagnostic +script, exact image paths and results are retained beside the reports as +`gradient_probe.py`, `gradient-probe.json` and `gradient-probe.log`. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 7be8f6b..9bd7772 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -52,8 +52,15 @@ now completes training/reload/inference for both heads and precisions on real Blair data with random initialization. It shows only 3.1–3.4% lower peak allocation, slower training and lower accuracy under current head-only QT, including a large flat-head regression. This is evidence against recommending the present recipe, -not completion of the speed/quality target. Pretrained runs and numerical -investigation are next; target-machine measurements and ONNX deployment remain open. +not completion of the speed/quality target. It motivated the pretrained comparison +below; target-machine measurements and ONNX deployment remain open. + +The subsequent [pretrained comparison and fixed-batch diagnostic](benchmarks.md#pretrained-initialization-and-fixed-batch-numerical-check) +did not reproduce the large flat accuracy drop in the same seed. INT8 still had +3–4% slower later training phases and only about 3.5% lower peak allocation. +Initial gradient discrepancies do not by themselves explain the convergence +difference. Repeated-seed quality checks, optimizer/stochastic-layer investigation +and backbone cost profiling remain necessary; no production speedup is established. ### EfficientNetV2-S structural coverage probe From 84fb1dce5a085e2e2d7a346130af19bbdb8969fc Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 10:54:25 +0200 Subject: [PATCH 046/155] fix: bound Gram matrix size when initializing large normalized heads --- mini_trainer/modeling/classifier.py | 4 +++- tests/test_classifier_shapes.py | 37 +++++++++++++++++++++++++++++ 2 files changed, 40 insertions(+), 1 deletion(-) diff --git a/mini_trainer/modeling/classifier.py b/mini_trainer/modeling/classifier.py index 349f8c5..ebec53b 100644 --- a/mini_trainer/modeling/classifier.py +++ b/mini_trainer/modeling/classifier.py @@ -41,7 +41,9 @@ def init_spherical_repulsion(cls, layer: nn.Module, iterations: int = 100, lr: f for _ in range(iterations): w.div_(w.norm(dim=1, keepdim=True).clamp(min=1e-9)) - grad = w @ w.t() @ w + # Associate through the smaller Gram matrix. A class-by-class matrix + # is prohibitive for heads with tens of thousands of output classes. + grad = w @ (w.t() @ w) if num_classes > w.size(1) else w @ w.t() @ w proj = (grad * w).sum(dim=1, keepdim=True) * w w.sub_((lr / num_classes) * (grad - proj)) diff --git a/tests/test_classifier_shapes.py b/tests/test_classifier_shapes.py index 8bab2f0..4b0e08c 100644 --- a/tests/test_classifier_shapes.py +++ b/tests/test_classifier_shapes.py @@ -35,3 +35,40 @@ def test_custom_hidden_width_checkpoint_roundtrip(normalized): inputs = torch.randn(2, 8) with torch.no_grad(): torch.testing.assert_close(restored(inputs), head(inputs)) + + +@pytest.mark.parametrize("shape", [(8, 32), (32, 8)]) +def test_spherical_initialization_preserves_reference_update(shape): + from mini_trainer.modeling import Classifier + + layer = torch.nn.Linear(shape[1], shape[0], bias=False) + torch.manual_seed(73) + reference = torch.empty_like(layer.weight).normal_() + for _ in range(100): + reference.div_(reference.norm(dim=1, keepdim=True).clamp(min=1e-9)) + gradient = reference @ reference.t() @ reference + projection = (gradient * reference).sum(dim=1, keepdim=True) * reference + reference.sub_((0.5 / shape[0]) * (gradient - projection)) + reference.div_(reference.norm(dim=1, keepdim=True).clamp(min=1e-9)) + torch.manual_seed(73) + result = Classifier.init_spherical_repulsion(layer) + assert result is layer + torch.testing.assert_close(layer.weight, reference, rtol=2e-5, atol=2e-6) + + +def test_large_normalized_head_does_not_allocate_class_gram_matrix(): + from torch.utils._python_dispatch import TorchDispatchMode + + from mini_trainer.modeling import Classifier + + class RejectClassGram(TorchDispatchMode): + def __torch_dispatch__(self, func, types, args=(), kwargs=None): + if func == torch.ops.aten.mm.default: + left, right = args + assert (left.shape[0], right.shape[1]) != (10000, 10000) + return func(*args, **(kwargs or {})) + + with RejectClassGram(): + head = Classifier(in_features=8, out_features=10000, hidden=True, normalized=True) + assert torch.isfinite(head.linear.weight).all() + torch.testing.assert_close(head.linear.weight.norm(dim=1), torch.ones(10000)) From d8c46e92992d544a04ae2bfd946f8f8d68a2d479 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 10:54:25 +0200 Subject: [PATCH 047/155] docs: measure EfficientNetV2 training with large class counts --- docs/benchmarks.md | 64 +++++++++++++++++++++++++++ docs/quantized-training-validation.md | 14 ++++++ 2 files changed, 78 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index ef0a771..65b90a6 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1078,3 +1078,67 @@ Further investigation should separate repeated-seed variation, optimizer updates and stochastic-layer effects before changing training numerics. The diagnostic script, exact image paths and results are retained beside the reports as `gradient_probe.py`, `gradient-probe.json` and `gradient-probe.log`. + +### Large-class head capacity and initialization + +The production envelope includes 10,000–1,000,000 classes. A 25-class head cannot +represent that regime. Initial 100k-class capacity probes failed in both float +and INT8 before training: spherical initialization evaluated `(W @ W.T) @ W`, +requesting a 100k-by-100k FP32 intermediate (37.25 GiB). At one million classes +that intermediate would require 3.64 TiB. + +The initializer now uses `W @ (W.T @ W)` when class count exceeds embedding +width. This keeps the smaller Gram matrix (6.25 MiB at width 1280) while preserving +the mathematical spherical-repulsion update. Floating-point association changes, +so initialization is not bitwise identical for these tall heads. Tests compare +100 iterations against the former formula and reject class-by-class allocations +during construction of a 10k-class normalized head. Heads no taller than their +embedding width retain the previous association. + +After that fix, eight fresh CUDA processes exercised full EfficientNetV2-S with +random initialization, symmetric normalized heads, 32 synthetic uint8 images at +128x128, FP16 AMP and eager MuonAuxAdamW. Hierarchical models used a synthetic +two-level taxonomy with 100 leaf classes per parent. Three warmup steps preceded +five measured steps; each measured loss was finite and each optimizer update +executed. The probe follows the trainer's forward/zero-grad/backward/clip/update +order, but excludes loading, logging, scheduling, checkpointing and deployment. + +| Classes | Head | Float / INT8 parameter MB | Float / INT8 peak allocated MiB | Float / INT8 median step ms | +| --- | --- | ---: | ---: | ---: | +| 10,000 | Flat | 138.56 / 95.29 | 1450.4 / 1373.2 | 89.78 / 78.68 | +| 10,000 | Hierarchical | 138.56 / 95.29 | 1450.4 / 1373.2 | 91.96 / 89.65 | +| 100,000 | Flat | 600.08 / 211.57 | 3910.4 / 4130.3 | 126.87 / 138.36 | +| 100,000 | Hierarchical | 600.08 / 211.57 | 3910.4 / 4130.4 | 144.66 / 144.83 | + +MB here is decimal; MiB is binary. At 100k classes, the physical parameter +reduction is substantial, but peak training allocation increases. QT's transient +floating normalization/gradient/update tensors therefore need investigation; +quantized parameter storage does not guarantee lower peak memory. These five-step, +single-process samples are capacity diagnostics, not robust speed estimates or +quality comparisons. Only head layers are quantized. No million-class training, +ONNX runtime or target-hardware performance claim follows from them. + +The original failure logs and probe are retained under ignored +`tmp-efficientnet-classes/`; rerun logs, script and JSON measurements are under +`tmp-efficientnet-classes-fixed/`. Both used the local RTX 3080 Ti Laptop GPU. +One-million-class validation should record initialization peak separately from +training peak, then check actual updates, checkpoint/reload and deployment on a +machine with enough memory. Eliminating the quadratic Gram matrix leaves linear +parameter, gradient and temporary storage costs; it does not establish that a +million-class run fits the local 16 GiB GPU. + +### Warmed training trace + +A separate pretrained 25-class float run captured three warmed training batches +through the actual Blair training loop, including MuonAuxAdamW updates. The trace +contains 137.85 ms of summed CUDA kernel/copy/memset event duration; a single +NCHW-to-NHWC conversion kernel contributes 7.52 ms (5.45%). Depthwise convolution +weight-gradient, batch normalization, SiLU and optimizer kernels are also visible. +This suggests examining layout and backbone costs alongside the large-head cases. + +The denominator counts device events once and excludes nested CPU operators and +GPU annotations. It is not wall time or a speedup prediction: profiling perturbs +execution, and summed durations do not account for overlap. Raw trace, grouped +operator data, script and run outputs are retained under ignored +`tmp-efficientnet-profile/`. Do not sum the grouped operator table directly; +it includes nested attribution as well as device events and would double-count. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 9bd7772..f1f1465 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -15,6 +15,20 @@ on the same reviewed Blair splits. The dense MNIST model is a kernel diagnostic; TinyConv on Blair is an integration check. Neither establishes performance or quality for the primary model. +Class count is an independent scaling dimension: production cases may have +10,000–1,000,000 classes. At embedding width 1280, output weights alone occupy +51.2 MB, 512 MB or 5.12 GB in FP32 at 10k, 100k or 1M classes. The 25-class Blair +head is not representative of those parameter, gradient, optimizer-state or score +storage costs. Include synthetic capacity probes alongside dataset quality runs; +never infer large-vocabulary accuracy from randomly assigned synthetic labels. + +The [large-class probes](benchmarks.md#large-class-head-capacity-and-initialization) +now exercise full EfficientNetV2-S training steps at 10k and 100k classes for both +normalized heads. They exposed and motivated a quadratic-memory initialization +fix. At 100k classes the current INT8 path saves parameter bytes but increases +peak training allocation, making transient normalization/gradient storage a +priority for investigation. A million-class training run remains unverified. + | Deployment target | Execution path to validate | Required measurements | | --- | --- | --- | | HPC: A40, A100, B300-class GPU systems with AMD EPYC hosts | PyTorch GPU training | End-to-end training time, steady-state throughput, allocated/reserved GPU peaks, host memory, loading/transfer costs, convergence and checkpoint/resume; record actual GPU, allocation, precision and kernels separately for each system. | From 6f9b094713b5fcb7718dc0a336ec3adad0c115d2 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 11:10:47 +0200 Subject: [PATCH 048/155] perf: fuse INT8 normalization backward to bound large-head memory --- docs/benchmarks.md | 49 ++++++++++++ docs/quantized-training-validation.md | 6 +- .../modeling/_quantized_normalization.py | 75 +++++++++++++++++++ mini_trainer/modeling/_quantized_training.py | 6 +- tests/test_quantized_training_model.py | 45 +++++++++++ 5 files changed, 179 insertions(+), 2 deletions(-) create mode 100644 mini_trainer/modeling/_quantized_normalization.py diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 65b90a6..53dc741 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1142,3 +1142,52 @@ execution, and summed durations do not account for overlap. Raw trace, grouped operator data, script and run outputs are retained under ignored `tmp-efficientnet-profile/`. Do not sum the grouped operator table directly; it includes nested attribution as well as device events and would double-count. + +### Bounded INT8 normalization backward storage + +The large-head follow-up isolates a concrete temporary-allocation cost in QT's +weight-normalization Jacobian. The previous expression materialized floating unit +directions and projection/update intermediates. A CUDA row kernel now computes +the same Jacobian while allocating only the returned direction and magnitude +gradients. CPU execution, widths above 16,384 and higher-order differentiation +retain the Torch expression. The kernel does not change quantization bit widths, +optimizer state, row scales, the forward normalization or stochastic rounding. + +At 100k rows by 1280 columns with FP32 gradient metadata, three warmups and five +measured calls reduced extra allocated bytes from 1,536,800,768 to 512,400,384. +Median isolated backward duration fell from about 18.5 ms to 2.7 ms. The probe +excludes caller-owned codes, scales, magnitudes, norms and upstream gradients; +it is not an end-to-end training measurement. Reports are retained under ignored +`tmp-quantized-normalization/`. + +Fresh full-model capacity probes used the same synthetic-input settings as the +large-class comparison above. Both heads had finite losses and applied every +measured optimizer update: + +| Head, 100k classes | Float / INT8 peak allocated MiB | Float / INT8 median step ms | +| --- | ---: | ---: | +| Flat | 3910.4 / 3231.4 | 129.47 / 135.97 | +| Hierarchical | 3910.4 / 3231.4 | 125.64 / 120.12 | + +Peak allocation during the five measured steps is now 17.4% lower under QT for +both heads. Model construction and the three warmup steps are excluded from that +peak. Step timings remain mixed and these short sequential samples do not +establish a portable speed gain. Reports are under `tmp-efficientnet-normalization/`. + +The CUDA model suite passed 73 cases covering the normalization Jacobian, +negative scales, zero/trainable magnitudes, FP32/FP16/BF16, strided upstream +gradients, compilation, optimizer graph replay, checkpoint and inference behavior. +A separate allocation-bound regression also passed. Source hashing includes the +new kernel so compiled backward graphs cannot reuse the previous implementation. +The full CPU-default suite passed with 351 tests passed, 141 skipped and the +existing EMA expected failure; static lint, formatting and import checks passed. + +Both pretrained Blair INT8 heads were rerun for five epochs under the previous +settings and produced finite held-out predictions after checkpoint reload. Flat +accuracy was 84.07% (previous INT8 83.63%); hierarchical fine/parent accuracy was +79.16%/91.73% (previous INT8 80.62%/91.30%). These mixed single-seed changes do not +establish quality equivalence or superiority; reduction-order changes can alter +quantized optimization trajectories. Reports are under `tmp-normalization-blair/`, +with source SHA256 `6aba096dca741e09863487cf4ed96dce21dea44b68b5612625caa3f3beb347d2`. +Repeated-seed convergence, million-class capacity, ONNX deployment and performance +on the intended target machines remain open. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index f1f1465..7c4bdc7 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -27,7 +27,11 @@ now exercise full EfficientNetV2-S training steps at 10k and 100k classes for bo normalized heads. They exposed and motivated a quadratic-memory initialization fix. At 100k classes the current INT8 path saves parameter bytes but increases peak training allocation, making transient normalization/gradient storage a -priority for investigation. A million-class training run remains unverified. +priority for investigation. The subsequent +[normalization backward kernel](benchmarks.md#bounded-int8-normalization-backward-storage) +reduces measured 100k-class training-step peak allocation by 17.4% versus float +for both heads. Timings and single-seed quality changes remain mixed. A +million-class training run remains unverified. | Deployment target | Execution path to validate | Required measurements | | --- | --- | --- | diff --git a/mini_trainer/modeling/_quantized_normalization.py b/mini_trainer/modeling/_quantized_normalization.py new file mode 100644 index 0000000..2d70b16 --- /dev/null +++ b/mini_trainer/modeling/_quantized_normalization.py @@ -0,0 +1,75 @@ +"""CUDA normalization gradients without full floating direction intermediates.""" + +import torch +import triton +import triton.language as tl + + +@triton.jit +def _backward_rows( + codes, + scales, + magnitude, + norm, + gradient, + direction_gradient, + magnitude_gradient, + columns, + code_row_stride, + code_column_stride, + scale_stride, + magnitude_stride, + norm_stride, + gradient_row_stride, + gradient_column_stride, + BLOCK: tl.constexpr, +): + row = tl.program_id(0) + column = tl.arange(0, BLOCK) + valid = column < columns + code = tl.load(codes + row * code_row_stride + column * code_column_stride, valid, 0).to(tl.float32) + scale = tl.load(scales + row * scale_stride).to(tl.float32) + length = tl.load(norm + row * norm_stride) + gain = tl.load(magnitude + row * magnitude_stride).to(tl.float32) + grad = tl.load(gradient + row * gradient_row_stride + column * gradient_column_stride, valid, 0).to(tl.float32) + sign = tl.where(scale > 0, 1.0, tl.where(scale < 0, -1.0, 0.0)) + unit = code * (sign / length) + projection = tl.sum(grad * unit, 0) + result = (grad - projection * unit) * (gain / (tl.abs(scale) * length)) + tl.store(direction_gradient + row * columns + column, result, valid) + tl.store(magnitude_gradient + row, projection) + + +@torch.library.custom_op("mini_trainer::int8_weight_norm_backward", mutates_args=()) +def int8_weight_norm_backward( + codes: torch.Tensor, scales: torch.Tensor, magnitude: torch.Tensor, norm: torch.Tensor, gradient: torch.Tensor +) -> tuple[torch.Tensor, torch.Tensor]: + """Apply the row normalization Jacobian, allocating only the returned gradients.""" + direction_gradient = torch.empty(codes.shape, device=codes.device, dtype=scales.dtype) + magnitude_gradient = torch.empty(magnitude.shape, device=magnitude.device, dtype=magnitude.dtype) + with torch.cuda.device(codes.device): + _backward_rows[(codes.shape[0],)]( + codes, + scales, + magnitude, + norm, + gradient, + direction_gradient, + magnitude_gradient, + codes.shape[1], + *codes.stride(), + scales.stride(0), + magnitude.stride(0), + norm.stride(0), + *gradient.stride(), + BLOCK=triton.next_power_of_2(codes.shape[1]), + enable_fp_fusion=False, + ) + return direction_gradient, magnitude_gradient + + +@int8_weight_norm_backward.register_fake +def _fake_backward(codes, scales, magnitude, norm, gradient): + return torch.empty(codes.shape, device=codes.device, dtype=scales.dtype), torch.empty( + magnitude.shape, device=magnitude.device, dtype=magnitude.dtype + ) diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index 70c32c9..e158419 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -12,6 +12,7 @@ from torchao.prototype.quantized_training.int8 import Int8QuantizedTrainingLinearWeight, quantize_int8_rowwise from ._quantized_matmul import scaled_int8_mm as _native_scaled_int8_mm +from ._quantized_normalization import int8_weight_norm_backward from ._quantized_update import quantize_int8_rows, update_int8_rows_ # Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary @@ -20,7 +21,8 @@ # Compute once when importing. _IMPLEMENTATION_HASH = hashlib.sha256( b"".join( - Path(__file__).with_name(name).read_bytes() for name in ("_quantized_training.py", "_quantized_matmul.py", "_quantized_update.py") + Path(__file__).with_name(name).read_bytes() + for name in ("_quantized_training.py", "_quantized_matmul.py", "_quantized_update.py", "_quantized_normalization.py") ) ).hexdigest() @@ -250,6 +252,8 @@ def forward(ctx, direction, magnitude): @staticmethod def backward(ctx, gradient): codes, scales, magnitude, norm = ctx.saved_tensors + if codes.device.type == "cuda" and codes.shape[1] <= 16384 and not torch.is_grad_enabled(): + return int8_weight_norm_backward(codes, scales, magnitude, norm, gradient) unit = codes.float() * (scales.sign() / norm).unsqueeze(1) projection = (gradient.float() * unit).sum(dim=1, keepdim=True) direction_gradient = (gradient.float() - projection * unit) * (magnitude.float() / (scales.float().abs() * norm).unsqueeze(1)) diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index b356e55..b4de067 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -724,3 +724,48 @@ def backward(): scheduler.step() if kind == "muon": assert optimizer._step_count == 11 + + +@pytest.mark.parametrize("width", [17, 1280]) +@pytest.mark.parametrize("dtype", [torch.float32, torch.float16, torch.bfloat16]) +def test_cuda_normalization_backward_matches_float_jacobian(width, dtype): + from mini_trainer.modeling._quantized_training import TrainingWeight + + device = cuda() + torch.manual_seed(19) + direction = nn.Parameter(TrainingWeight.from_float(torch.randn(7, width, device=device, dtype=dtype))) + with torch.no_grad(): + direction.scale[::2].neg_() + magnitude = nn.Parameter(torch.randn(7, 1, device=device, dtype=dtype)) + with torch.no_grad(): + magnitude[0].zero_() + reference_direction = (direction.int_data.float() * direction.scale.float().unsqueeze(1)).requires_grad_() + reference_magnitude = magnitude.detach().float().requires_grad_() + expected = torch._weight_norm(reference_direction, reference_magnitude, 0) + gradient = torch.randn(width, 7, device=device, dtype=dtype).T # Exercise strided upstream gradients. + actual = torch._weight_norm(direction, magnitude, 0) + actual.backward(gradient) + expected.backward(gradient.float()) + tolerance = {torch.float32: 2e-5, torch.float16: 2e-3, torch.bfloat16: 2e-2}[dtype] + torch.testing.assert_close(direction.grad.float(), reference_direction.grad, rtol=tolerance, atol=tolerance) + torch.testing.assert_close(magnitude.grad.float(), reference_magnitude.grad, rtol=tolerance, atol=tolerance) + + +def test_cuda_normalization_backward_bounds_temporary_storage(): + from mini_trainer.modeling._quantized_normalization import int8_weight_norm_backward + + device = cuda() + codes = torch.randint(-127, 128, (2048, 1280), device=device, dtype=torch.int8) + scales = torch.ones(2048, device=device) + magnitude = torch.ones(2048, 1, device=device) + norm = codes.float().norm(dim=1) + gradient = torch.randn(2048, 1280, device=device) + warmup = int8_weight_norm_backward(codes, scales, magnitude, norm, gradient) + del warmup + torch.cuda.synchronize() + torch.cuda.reset_peak_memory_stats() + start = torch.cuda.memory_allocated() + result = int8_weight_norm_backward(codes, scales, magnitude, norm, gradient) + torch.cuda.synchronize() + output_bytes = sum(value.numel() * value.element_size() for value in result) + assert torch.cuda.max_memory_allocated() - start <= output_bytes + 1024 * 1024 From 09df44e8debe029e86132a532fe5ac1051c99de5 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 11:23:50 +0200 Subject: [PATCH 049/155] test: validate EfficientNetV2 ONNX exports and document INT8 limitations --- docs/benchmarks.md | 52 ++++++++++++++++++++++++++++++++++++++++++++++ docs/onnx.md | 10 ++++++++- tests/test_onnx.py | 22 ++++++++++++++++++++ 3 files changed, 83 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 53dc741..a72302a 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1191,3 +1191,55 @@ quantized optimization trajectories. Reports are under `tmp-normalization-blair/ with source SHA256 `6aba096dca741e09863487cf4ed96dce21dea44b68b5612625caa3f3beb347d2`. Repeated-seed convergence, million-class capacity, ONNX deployment and performance on the intended target machines remain open. + +### EfficientNetV2 ONNX CPU export and inference quantization + +The pretrained floating Blair checkpoints were exported through the existing +generic API for both symmetric normalized EfficientNetV2-S heads. ONNX Runtime +CPU parity passed at batch sizes 1, 2, 4 and 8, including 16 real held-out images. +The maximum observed absolute score difference was 1.67e-5 for the flat head and +1.29e-5 for the hierarchical head, within the existing combined rtol=1e-4, +atol=1e-5 checks. The test suite now also exports both actual architectures with +random initialization and verifies dynamic batch sizes 1–4 without downloads. + +Runtime dependencies were installed at the locked versions (ONNX 1.22.0, +ONNX Runtime 1.29.0, ONNX Script 0.7.1), with constraints preserving the existing +CUDA PyTorch, Torchvision, TorchAO and NumPy versions. This was an environment +installation, not a dependency declaration or lock-file change. +The experiment's complete package-version record is retained in +`tmp-efficientnet-onnx/environment.json`; its protobuf dependency resolved to +7.36.1 rather than the lock's 7.35.0, so this is not a fully locked-environment run. + +A separate exploratory post-training quantization trial used ONNX Runtime static +QDQ with per-channel INT8 weights and signed INT8 activations, targeting Conv, +Gemm and MatMul. MinMax calibration used 128 training images sampled with seed 42; +neither validation nor test images entered calibration. The exported graph +already folded BatchNorm into convolution; preprocessing ran shape inference +without further graph optimization before quantization. This follows the starting +point described in [ONNX Runtime's quantization guide](https://onnxruntime.ai/docs/performance/model-optimizations/quantization.html), +but the result is **not an acceptable deployment recipe**: + +| Head | Float / INT8 fine accuracy | Float / INT8 parent accuracy | Float / INT8 graph plus weight bytes | +| --- | ---: | ---: | ---: | +| Flat | 83.55% / 55.73% | — | 88,624,980 / 25,398,200 | +| Hierarchical | 78.12% / 70.28% | 90.61% / 85.62% | 88,645,350 / 25,418,559 | + +All 1,161 held-out images produced finite outputs. Float and INT8 were evaluated +through the same CPU provider and preprocessing; the floating accuracy differs +slightly from the earlier CUDA AMP results. Artifacts are roughly 71% smaller, +but the execution profile confirms only 63 QLinearConv operations and two QGemm +operations, alongside 107 floating Conv operations. QDQ nodes and integer stored +weights therefore do not establish integer execution throughout the backbone. + +These are exploratory held-out checks, not a confirmatory quality comparison or +a timing study. Subsequent recipe tuning should use validation images, localize +weight/activation error, and investigate incomplete quantized-operator fusion. +Do not select a production recipe from artifact size alone. No ONNX GPU or ARM +execution was tested, and no native QT checkpoint was converted in this trial. + +Exports, calibration records with source-image hashes, full predictions, runtime +profiles, report JSON and scripts are retained under ignored +`tmp-efficientnet-onnx/`. Preprocessing remains external to the graph; the local +experiment used the checkpoint's repository preprocessing and does not yet provide +a standalone raw-image deployment recipe. Missing telemetry/cache write access +emitted environment warnings; export and inference completed. diff --git a/docs/onnx.md b/docs/onnx.md index 08616c0..7166b0d 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -98,7 +98,15 @@ used by BioCLIP, plus every classifier head family. This is representative backe coverage, not certification of every model in the catalog. Arbitrary custom operators and data-dependent Python control flow remain subject to the [PyTorch ONNX exporter's support](https://docs.pytorch.org/docs/stable/onnx_export.html). -GPU providers, quantized graphs and arbitrary spatial dimensions are not validated. +The actual EfficientNetV2-S backbone with symmetric normalized flat/hierarchical +heads is also covered by offline dynamic-batch export tests. Trained Blair +checkpoints passed ONNX Runtime CPU parity on real images; see the +[deployment experiment](benchmarks.md#efficientnetv2-onnx-cpu-export-and-inference-quantization). +GPU providers, ARM execution and arbitrary spatial dimensions remain unvalidated. +An experimental ONNX static INT8 recipe executed on the local CPU but lost +substantial accuracy and retained floating convolution execution. It is not a +supported production recipe. Native CUDA QT checkpoints remain a separate backend +and are not made ONNX-exportable by these floating-checkpoint experiments. These local bundles are a foundation for Hugging Face hosting. Model cards, evaluation attachments and Hub upload commands remain separate roadmap work. diff --git a/tests/test_onnx.py b/tests/test_onnx.py index ba12827..80e2bde 100644 --- a/tests/test_onnx.py +++ b/tests/test_onnx.py @@ -203,3 +203,25 @@ def run(self, *args, **kwargs): with pytest.raises(AssertionError, match="ONNX parity failed"): export_onnx(model, torch.randn(2, 3, 5, 5), tmp_path / "bundle") assert list(tmp_path.iterdir()) == [] + + +@pytest.mark.parametrize("head", [Classifier, HierarchicalClassifier]) +def test_efficientnet_v2_s_symmetric_normalized_heads(tmp_path, head): + torch.manual_seed(42) + kwargs = {} + if head is HierarchicalClassifier: + kwargs["sparse_masks"] = [torch.arange(25) % 15] + model, _ = head.build( + model_type="efficientnet_v2_s", + model_args={"pretrained": False}, + num_classes=25, + hidden=True, + normalized=True, + **kwargs, + ) + example = torch.randn(2, 3, 128, 128) + destination = export_onnx(model, example, tmp_path / "efficientnet-v2", verification_inputs=[torch.randn(3, 3, 128, 128)]) + manifest = json.loads((destination / "manifest.json").read_text()) + assert {case["batch_size"] for case in manifest["verification"]["cases"]} == {1, 2, 3, 4} + assert len(manifest["outputs"]) == (2 if head is HierarchicalClassifier else 1) + assert manifest["classifiers"][0]["metadata"]["in_features"] == 1280 From 1046a86abc409a393da3103ee5163b97334a7171 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 11:48:28 +0200 Subject: [PATCH 050/155] docs: measure ONNX calibration quality with mini_metrics --- docs/benchmarks.md | 111 +++++++++++++++++++++++++++++++++++++++++++++ docs/onnx.md | 10 ++-- 2 files changed, 118 insertions(+), 3 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index a72302a..d77b65a 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1243,3 +1243,114 @@ profiles, report JSON and scripts are retained under ignored experiment used the checkpoint's repository preprocessing and does not yet provide a standalone raw-image deployment recipe. Missing telemetry/cache write access emitted environment warnings; export and inference completed. + +### ONNX activation calibration, execution coverage and macro metrics + +A follow-up on the same floating checkpoints separates execution coverage from +quantization error. Recoding signed activation zero points as unsigned values +(`zero_point + 128`), retaining scales and signed weights, gave bitwise-identical +unoptimized outputs on a fixed real validation batch for both heads. With ORT +optimization enabled, all 170 convolutions then executed as QLinearConv, rather +than 63 QLinearConv plus 107 floating Conv. This is a representation-dependent +fusion result on this CPU/provider build, not evidence that other providers behave +the same way. It did not solve the MinMax quality loss. + +Diagnostic weight-only and activation-only graphs were evaluated on all 912 +validation images. Float / weight-only / activation-only accuracy was +84.10% / 83.22% / 55.92% for flat, 80.92% / 80.92% / 65.68% for hierarchical fine, +and 91.12% / 90.90% / 77.96% for hierarchical parent. These controls isolate error +sources; they are not efficient deployment graphs. Activation quantization is the +larger problem here, although weight and activation errors interact. + +One focused Percentile 99.9 calibration trial used the same 128 training images +(seed 42), asymmetric activation ranges, unsigned INT8 activations and per-channel +signed INT8 weights. Histogram collection used 2,048 bins and batches of eight, +accumulating histograms across batches without retaining every raw activation. +Conv, Gemm and MatMul remained the target operations. Validation accuracy recovered +to 82.68% flat and 78.62% / 89.47% hierarchical fine / parent. This follow-up used +validation data for diagnosis, not a new inspection of the test split. + +Quality was also evaluated through the local `mini_metrics` checkout at commit +`70cc69adc05362863439277048e06386c1f885e1` (clean working tree). The table uses +ordinary Macro-F1, Macro-Recall, Macro-Precision, Coverage and Theil's U directly +from that package. Values are proportions, not percentages. + +| Output / recipe | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat / float | 0.76207 | 0.75442 | 0.78957 | 1.00000 | 0.81479 | +| Flat / signed MinMax | 0.44030 | 0.45594 | 0.62608 | 1.00000 | 0.63969 | +| Flat / unsigned Percentile | 0.76082 | 0.75161 | 0.78966 | 1.00000 | 0.81173 | +| Hierarchical fine / float | 0.70922 | 0.69573 | 0.79877 | 1.00000 | 0.78258 | +| Hierarchical fine / signed MinMax | 0.59050 | 0.58995 | 0.74886 | 1.00000 | 0.72673 | +| Hierarchical fine / unsigned Percentile | 0.69824 | 0.67946 | 0.75561 | 1.00000 | 0.76686 | +| Hierarchical parent / float | 0.85936 | 0.82759 | 0.92038 | 1.00000 | 0.85139 | +| Hierarchical parent / signed MinMax | 0.77617 | 0.74597 | 0.89192 | 1.00000 | 0.79788 | +| Hierarchical parent / unsigned Percentile | 0.84110 | 0.80161 | 0.91253 | 1.00000 | 0.82635 | + +Every recipe uses exactly the same 912 validation images and manifest class +mapping, with independent top-1 predictions at each available output level. +There is no threshold optimization, resampling, known-label filtering or inferred +parent output for the flat head. Confidence is softmax maximum; threshold zero +makes Coverage 100% by construction. This does not measure useful abstention. +The flat Macro-F1 decrease is 0.13 percentage points, versus 1.10 / 1.83 points +for hierarchical fine / parent. These single-checkpoint descriptive results do +not establish a production acceptance threshold or statistical equivalence. + +For a CSV in the repository's `mini_metric.csv` schema, the exact metric call is: + +```python +from mini_metrics.metrics import MacroF1, evaluate_file + +metrics = evaluate_file( + "mini_metric.csv", + optimal=False, + threshold=0, + known_only=False, + per_class=False, + simple=True, + hierarchical=False, + pattern=r"^(f1|recall|precision|coverage|theilU)$", + opt_crit=MacroF1, + verbose=0, +) +``` + +In this package version, unprefixed `f1`, `recall` and `precision` identify macro +metrics; micro variants have a `micro_` prefix. Pin the metrics revision when +comparing reports. The installed package was not replaced: this experiment +selected the sibling checkout with an explicit process-local `PYTHONPATH`. +The package itself does not depend on that filesystem layout. The publication +bootstrap script provides a reference for a later confidence-thresholded study; +its `optimal=True` mode changes the evaluated sample set by splitting calibration +from evaluation. Such a study should preserve paired splits across recipes and +report Coverage alongside quality, separately from this full-coverage comparison. + +Warm CPU inference timing used one ORT intra/inter-op thread, fixed preprocessed +real inputs, three warmups and eleven measured repetitions per batch size. +Recipe execution order alternated forward/reverse each repetition. Medians below +are milliseconds per `Session.run`, excluding image loading, preprocessing, +session construction and model export; these are not end-to-end CLI latencies. + +| Head / batch | Float | Signed MinMax | Unsigned Percentile | +| --- | ---: | ---: | ---: | +| Flat / 1 | 23.29 | 35.89 | 11.91 | +| Flat / 8 | 162.81 | 243.15 | 80.67 | +| Hierarchical / 1 | 23.86 | 36.75 | 11.96 | +| Hierarchical / 8 | 175.44 | 252.97 | 84.56 | + +A separate execution profile of the Percentile graphs confirms 170 QLinearConv +and two QGemm operations, with no floating Conv. Sigmoid, multiplication, +normalization and other operations still execute in floating point. These are +local Intel i7-12800H x86 CPU measurements with ORT 1.29.0, not Raspberry Pi, ARM, +CUDA-provider or production throughput verification. Calibration and activation +representation both differ between the timed quantized recipes, so timing does +not isolate either change. Quality acceptance, repeated-process timing and target +hardware measurements remain open. Native CUDA QT checkpoint export is still a +separate unsupported path. + +Local scripts, scores, CSV inputs, exact metric JSON, calibration histograms, +raw timing samples and execution profiles are retained under ignored +`tmp-onnx-validation/`. The dataset manifest SHA256 is +`1cef6c7d9133d889b6c8eff0ffef29b1db5d6653ab4adf0c005739af8c2921c6`. +This documents an exploratory experiment; those local artifacts are not a shared +continuous-evaluation service or a portable deployment harness. diff --git a/docs/onnx.md b/docs/onnx.md index 7166b0d..430e77b 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -103,9 +103,13 @@ heads is also covered by offline dynamic-batch export tests. Trained Blair checkpoints passed ONNX Runtime CPU parity on real images; see the [deployment experiment](benchmarks.md#efficientnetv2-onnx-cpu-export-and-inference-quantization). GPU providers, ARM execution and arbitrary spatial dimensions remain unvalidated. -An experimental ONNX static INT8 recipe executed on the local CPU but lost -substantial accuracy and retained floating convolution execution. It is not a -supported production recipe. Native CUDA QT checkpoints remain a separate backend +An initial signed MinMax INT8 recipe lost substantial accuracy and retained +floating convolutions. A follow-up unsigned Percentile recipe executed all +convolutions as QLinearConv and roughly halved warm local CPU inference latency, +with remaining quality losses measured through `mini_metrics`; see the +[calibration and metric results](benchmarks.md#onnx-activation-calibration-execution-coverage-and-macro-metrics). +It remains exploratory, with no agreed production quality gate or target-device +verification. Native CUDA QT checkpoints remain a separate backend and are not made ONNX-exportable by these floating-checkpoint experiments. These local bundles are a foundation for Hugging Face hosting. Model cards, From f758c4c8153f0a999542b0caee580ecad4463983 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 11:57:48 +0200 Subject: [PATCH 051/155] feat: add portable ONNX inference benchmark with provider verification --- dev/benchmarks/README.md | 61 ++++++++++++ dev/benchmarks/onnx_inference.py | 156 +++++++++++++++++++++++++++++++ docs/benchmarks.md | 12 +++ tests/test_benchmark_onnx.py | 82 ++++++++++++++++ 4 files changed, 311 insertions(+) create mode 100644 dev/benchmarks/onnx_inference.py create mode 100644 tests/test_benchmark_onnx.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 1734735..4c07761 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -570,3 +570,64 @@ runs it alongside the existing profiles, publishes its summary, and retains reports and failures for 90 days. Keep the older profiles as controls: optimizer graph replay is opt-in and does not improve every workload. See the [measured larger-batch results](../../docs/benchmarks.md#larger-batch-model-and-optimizer-graph-results). + +## Portable ONNX inference measurements + +`onnx_inference` runs without importing PyTorch or `mini_trainer`. Use a prepared +NumPy, ONNX and ONNX Runtime environment appropriate to the target CPU/GPU; it +never installs packages. The repository's export extra supplies the CPU provider. +GPU provider installation is a separate environment choice, not an automatic +replacement of that installation. + +Prepare a fixed `.npz` containing preprocessed arrays keyed by exact ONNX input +names (for the default exporter, `images`). Carry the preprocessing recipe, sample +identities, split manifest and labels alongside it. Use the same input file for +every recipe and machine. A batch is fixed by its array shape; prepare separate +files for batch-one edge latency and larger batch throughput measurements. +Copy each whole ONNX bundle, including external weights, to the target machine. + +```bash +.venv/bin/python -m dev.benchmarks.onnx_inference \ + --model exported-float/model.onnx --model exported-int8/model.onnx \ + --inputs preprocessed-batch.npz --output /tmp/onnx-cpu-run-1 \ + --provider CPUExecutionProvider --threads 1 --warmup 3 --repeats 11 + +# In a separately prepared ONNX Runtime GPU environment: +python -m dev.benchmarks.onnx_inference \ + --model exported-float/model.onnx --model exported-int8/model.onnx \ + --inputs preprocessed-batch.npz --output /tmp/onnx-cuda-run-1 \ + --provider CUDAExecutionProvider --provider-options '{"device_id": "0"}' +``` + +Each output directory must be new. Reports include graph/external-weight/input +hashes, runtime versions/build, runner source hash, input shapes/dtypes, requested +and effective provider configuration, every timing sample and medians. Outputs +are saved by ONNX output name, suitable for a separate quality evaluation with +`mini_metrics` using the original class mappings and labels. This runner does not +infer score semantics, calibrate thresholds or impose a quality acceptance gate. + +Profiling uses a separate session and records actual operation/provider counts, +including CPU execution in a GPU-requested run. An unavailable provider or a graph +that executes no operations on the requested provider fails instead of reporting +a CPU run as a GPU measurement. Partial CPU execution is reported, not prohibited: +shape and other auxiliary operations may legitimately use CPU. Inspect the profile +for the expensive operations. Integer weights or QDQ nodes alone are not proof of +integer execution. Failures after output creation retain a failed `report.json`; +input/environment preflight failures leave no result directory. + +Timing covers warm `Session.run` with CPU NumPy inputs and outputs, including +host/device transfers on GPU. It excludes image IO, preprocessing, export, session +construction and profiling. It is neither GPU-only kernel time nor end-to-end +image prediction latency. Sessions for all supplied models coexist during timing; +use individual model invocations if residency exceeds device capacity, and record +that different protocol. For repeated-process evidence, run at least three fresh +invocations with separate output directories and reverse model argument order in +alternate invocations. Avoid concurrent training, tests or other benchmarks. + +Start with the exact same bundles on the local CPU, then repeat on Raspberry Pi +(or the intended ARM device) and the intended ONNX GPU device. Do not reuse +hardware-specific optimized graph caches across devices. Record the device model, +power/thermal configuration and concurrent workload alongside the report. The +unsigned activation recipe measured on x86 is a candidate to test, not a universal +GPU/ARM recipe. This inference runner does not validate HPC PyTorch quantized +training, native QT checkpoint export or million-class capacity. diff --git a/dev/benchmarks/onnx_inference.py b/dev/benchmarks/onnx_inference.py new file mode 100644 index 0000000..416ea1c --- /dev/null +++ b/dev/benchmarks/onnx_inference.py @@ -0,0 +1,156 @@ +"""Measure ONNX Session.run with explicit providers and portable preprocessed inputs.""" + +import hashlib +import json +import platform +import statistics +import time +from argparse import ArgumentParser +from collections import Counter +from pathlib import Path + +import numpy as np + + +def file_hash(path): + with Path(path).open("rb") as handle: + return hashlib.file_digest(handle, "sha256").hexdigest() + + +def model_files(path, onnx): + """Include external tensor storage, also inside graph attributes/subgraphs.""" + paths = {path.resolve()} + + def visit(message): + if isinstance(message, onnx.TensorProto): + for entry in message.external_data: + if entry.key == "location": + paths.add((path.parent / entry.value).resolve()) + for field, value in message.ListFields(): + if field.message_type is not None: + for child in value if field.is_repeated else [value]: + visit(child) + + visit(onnx.load(path, load_external_data=False)) + return [{"path": str(p), "sha256": file_hash(p), "bytes": p.stat().st_size} for p in sorted(paths)] + + +def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provider_options=None): + import onnx + import onnxruntime as ort + + if min(threads, warmup, repeats) < 1: + raise ValueError("threads, warmup and repeats must be positive") + if not models or len(set(models)) != len(models): + raise ValueError("Supply distinct model paths") + if provider not in ort.get_available_providers(): + raise ValueError(f"Requested provider {provider} unavailable; available: {ort.get_available_providers()}") + with np.load(inputs, allow_pickle=False) as archive: + feeds = {name: np.ascontiguousarray(archive[name]) for name in archive.files} + if not feeds or any(not np.isfinite(value).all() for value in feeds.values()): + raise ValueError("Inputs must contain finite named arrays") + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "environment": {"platform": platform.platform(), "machine": platform.machine(), "python": platform.python_version()}, + "versions": {"onnx": onnx.__version__, "onnxruntime": ort.__version__, "numpy": np.__version__}, + "runtime_build": ort.get_build_info(), + "runner_sha256": file_hash(__file__), + "provider": provider, + "provider_options": provider_options or {}, + "available_providers": ort.get_available_providers(), + "threads": threads, + "warmup": warmup, + "repeats": repeats, + "inputs": {"path": str(inputs), "sha256": file_hash(inputs)}, + "input_arrays": {name: {"shape": list(a.shape), "dtype": str(a.dtype)} for name, a in feeds.items()}, + "scope": ( + "Warm Session.run with CPU NumPy inputs and outputs; includes device transfers, " + "excludes image IO, preprocessing and session construction." + ), + "order": "alternating forward/reverse model order per repetition", + "models": [], + } + sessions = [] + providers = [(provider, provider_options or {})] + if provider != "CPUExecutionProvider": + providers.append("CPUExecutionProvider") + + def options(): + opts = ort.SessionOptions() + opts.intra_op_num_threads = threads + opts.inter_op_num_threads = 1 + return opts + + try: + for index, path in enumerate(map(Path, models)): + record = {"path": str(path), "files": model_files(path, onnx), "seconds": []} + report["models"].append(record) + opts = options() + opts.enable_profiling = True + opts.profile_file_prefix = str(output / f"model-{index}-profile") + session = ort.InferenceSession(str(path), sess_options=opts, providers=providers) + session.disable_fallback() + try: + if set(feeds) != {node.name for node in session.get_inputs()}: + raise ValueError(f"Input names do not match {path}") + predictions = session.run(None, feeds) + finally: + profile = session.end_profiling() + counts = Counter( + (event["args"].get("op_name"), event["args"].get("provider")) + for event in json.loads(Path(profile).read_text()) + if event.get("cat") == "Node" and event.get("args", {}).get("provider") + ) + record["execution"] = [{"op": op, "provider": ep, "count": n} for (op, ep), n in sorted(counts.items())] + record["profile"] = str(profile) + if not any(ep == provider for _, ep in counts): + raise RuntimeError(f"No profiled operation executed on requested provider {provider}") + if any(not np.isfinite(array).all() for array in predictions): + raise ValueError(f"Nonfinite predictions from {path}") + record["outputs"] = [node.name for node in session.get_outputs()] + np.savez(output / f"model-{index}-outputs.npz", **dict(zip(record["outputs"], predictions, strict=True))) + del session + session = ort.InferenceSession(str(path), sess_options=options(), providers=providers) + session.disable_fallback() + record["session_providers"] = session.get_providers() + record["session_provider_options"] = session.get_provider_options() + sessions.append(session) + for trial in range(warmup + repeats): + order = range(len(sessions)) if trial % 2 else reversed(range(len(sessions))) + for index in order: + started = time.perf_counter() + sessions[index].run(None, feeds) + elapsed = time.perf_counter() - started + if trial >= warmup: + report["models"][index]["seconds"].append(elapsed) + for record in report["models"]: + record["median_seconds"] = statistics.median(record["seconds"]) + report["status"] = "passed" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, action="append", required=True) + parser.add_argument("--inputs", type=Path, required=True, help="NPZ with exact ONNX input names and preprocessed arrays") + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--provider", required=True) + parser.add_argument("--provider-options", type=json.loads, default={}) + parser.add_argument("--threads", type=int, default=1) + parser.add_argument("--warmup", type=int, default=3) + parser.add_argument("--repeats", type=int, default=11) + args = vars(parser.parse_args()) + args["models"] = args.pop("model") + run(**args) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index d77b65a..3a052c7 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1354,3 +1354,15 @@ raw timing samples and execution profiles are retained under ignored `1cef6c7d9133d889b6c8eff0ffef29b1db5d6653ab4adf0c005739af8c2921c6`. This documents an exploratory experiment; those local artifacts are not a shared continuous-evaluation service or a portable deployment harness. + +The portable `dev.benchmarks.onnx_inference` runner now retains graph/external +weight/input hashes, named outputs, raw timings, runtime configuration and actual +operation/provider execution profiles. See the +[target-machine commands](../dev/benchmarks/README.md#portable-onnx-inference-measurements). +A local batch-eight check of the same trained flat graphs measured 181.24 ms float +and 90.57 ms Percentile INT8, with 170 floating Conv versus 170 QLinearConv and two +QGemm operations. The report is retained under ignored `tmp-onnx-portable-flat/`. +This verifies the shared runner on a real model, not repeatability across machines +or a new quality comparison. Regression tests cover external weight provenance, +retained input-contract failures, unavailable providers and an advertised GPU +provider whose graph actually executes entirely on CPU. diff --git a/tests/test_benchmark_onnx.py b/tests/test_benchmark_onnx.py new file mode 100644 index 0000000..d67dee5 --- /dev/null +++ b/tests/test_benchmark_onnx.py @@ -0,0 +1,82 @@ +import json + +import numpy as np +import pytest + +from dev.benchmarks.onnx_inference import run + +onnx = pytest.importorskip("onnx") +pytest.importorskip("onnxruntime") + + +@pytest.fixture +def model_and_inputs(tmp_path): + graph = onnx.helper.make_graph( + [onnx.helper.make_node("MatMul", ["images", "weight"], ["scores"])], + "linear", + [onnx.helper.make_tensor_value_info("images", onnx.TensorProto.FLOAT, ["batch", 2])], + [onnx.helper.make_tensor_value_info("scores", onnx.TensorProto.FLOAT, ["batch", 2])], + [onnx.numpy_helper.from_array(np.eye(2, dtype=np.float32), "weight")], + ) + model = onnx.helper.make_model(graph, opset_imports=[onnx.helper.make_opsetid("", 18)], ir_version=10) + path = tmp_path / "model.onnx" + onnx.save_model(model, path, save_as_external_data=True, all_tensors_to_one_file=True, location="weights.data", size_threshold=0) + inputs = tmp_path / "inputs.npz" + np.savez(inputs, images=np.array([[1, 2], [3, 4]], dtype=np.float32)) + return path, inputs + + +def test_cpu_measurement_records_external_weights_execution_and_outputs(model_and_inputs, tmp_path): + model, inputs = model_and_inputs + output = tmp_path / "measurement" + report = run([model], inputs, output, "CPUExecutionProvider", warmup=1, repeats=2) + assert report == json.loads((output / "report.json").read_text()) + assert report["status"] == "passed" + record = report["models"][0] + assert len(record["files"]) == 2 + assert len(record["seconds"]) == 2 + assert record["median_seconds"] > 0 + assert any(e["op"] == "MatMul" and e["provider"] == "CPUExecutionProvider" for e in record["execution"]) + with np.load(output / "model-0-outputs.npz") as actual, np.load(inputs) as expected: + np.testing.assert_array_equal(actual["scores"], expected["images"]) + with pytest.raises(FileExistsError): + run([model], inputs, output, "CPUExecutionProvider") + + +def test_unavailable_provider_does_not_fall_back(model_and_inputs, tmp_path): + model, inputs = model_and_inputs + with pytest.raises(ValueError, match="unavailable"): + run([model], inputs, tmp_path / "missing", "MissingExecutionProvider") + assert not (tmp_path / "missing").exists() + + +def test_input_contract_failure_is_retained(model_and_inputs, tmp_path): + model, inputs = model_and_inputs + np.savez(inputs, wrong=np.ones((2, 2), dtype=np.float32)) + output = tmp_path / "failure" + with pytest.raises(ValueError, match="Input names"): + run([model], inputs, output, "CPUExecutionProvider") + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" + assert "median_seconds" not in report["models"][0] + + +def test_advertised_provider_with_only_cpu_execution_fails(model_and_inputs, tmp_path, monkeypatch): + import onnxruntime as ort + + model, inputs = model_and_inputs + real_session = ort.InferenceSession + monkeypatch.setattr(ort, "get_available_providers", lambda: ["CUDAExecutionProvider", "CPUExecutionProvider"]) + # Simulate a provider being advertised but the real graph executing on CPU. + monkeypatch.setattr( + ort, + "InferenceSession", + lambda path, sess_options, providers: real_session(path, sess_options=sess_options, providers=["CPUExecutionProvider"]), + ) + output = tmp_path / "fallback" + with pytest.raises(RuntimeError, match="No profiled operation"): + run([model], inputs, output, "CUDAExecutionProvider") + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" + assert all(e["provider"] == "CPUExecutionProvider" for e in report["models"][0]["execution"]) + assert report["models"][0]["seconds"] == [] From ed60f6afc2dcb558079fdc66f453e990e18fd13f Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 12:16:36 +0200 Subject: [PATCH 052/155] feat: export native INT8 training forwards to ONNX --- dev/benchmarks/README.md | 9 +- docs/benchmarks.md | 37 ++++++++ docs/onnx.md | 51 ++++++++++- docs/quantized-training-validation.md | 6 +- docs/quantized-training.md | 7 +- mini_trainer/export.py | 16 +++- mini_trainer/modeling/_onnx_quantized.py | 16 ++++ mini_trainer/modeling/onnx.py | 68 +++++++++++--- tests/test_onnx_quantized_training.py | 109 +++++++++++++++++++++++ 9 files changed, 295 insertions(+), 24 deletions(-) create mode 100644 mini_trainer/modeling/_onnx_quantized.py create mode 100644 tests/test_onnx_quantized_training.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 4c07761..a0279d1 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -352,10 +352,11 @@ manifest, checkpoints and commands, with a separately selected compatible backen samples to the ARM device. Verify scores before measuring batch-one latency, sustained throughput and process memory under explicit thread counts. -The current native QT checkpoint is not ONNX-exportable. Establish floating -EfficientNetV2 export parity first, then implement and validate an appropriate -deployment quantization path for each provider. Export/runtime work and target -machine verification remain open; these commands do not establish those results. +Floating and native INT8 EfficientNetV2 checkpoints now have locally verified +[ONNX export paths](../../docs/onnx.md). Native export requires an explicit CUDA +reference; it retains floating convolutions and integer head products. Deployment +quantization and runtime placement must still be validated for each target +provider; these training commands do not establish those results. Retain reports, logs, failures and predictions with the existing benchmark summary/artifact workflow so external runs can be reviewed without machine access. diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 3a052c7..42e656d 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1366,3 +1366,40 @@ This verifies the shared runner on a real model, not repeatability across machin or a new quality comparison. Regression tests cover external weight provenance, retained input-contract failures, unavailable providers and an advertised GPU provider whose graph actually executes entirely on CPU. + +### Native INT8 training checkpoint export + +The native CUDA QT checkpoints from `tmp-normalization-blair/` now export their +captured integer forward through the generic ONNX API with +`reference_device="cuda:0"`. The private export copy freezes parameters so +PyTorch's tensor-subclass decomposition does not try to assign gradients to +integer storage. The training model and optimizer parameters remain untouched. +The custom scaled INT8 product lowers to MatMulInteger with INT32 accumulation, +retaining the existing row quantizer and floating scales. There is no calibration +set or conversion to a floating-weight classifier. + +An initial trained flat export failed the unchanged parity gate: maximum score +error was 0.01264 on the first real image with TF32 allowed. Disabling TF32 in the +CUDA reference removed that failure. The exporter now scopes full-FP32 reference +execution and records this precision choice; it restores caller settings. The +failure is retained in `tmp-native-int8-onnx/export.log`, and the successful +explicit-precision diagnostic in `export-no-tf32.log`. + +Both final trained exports passed rtol=1e-4/atol=1e-5 at batches 1, 2, 4 and 8, +using the same eight preprocessed validation images as the preceding portable +runtime check. The maximum observed absolute score difference was 1.72e-5 for +flat and 1.29e-5 for hierarchical fine/parent. A separate runtime profile confirms +two MatMulInteger and 170 floating Conv operations for each model on the local +ONNX Runtime CPU provider. These are native training head products, not the +170-quantized-convolution Percentile recipe from floating checkpoints. + +Seven focused tests passed with intentional CUDA access: explicit-device failure, +normalized symmetric flat/hierarchical heads on tiny and actual EfficientNetV2-S +backbones, active-class filtering, dynamic batches, caller-state/TF32 restoration, +restricted checkpoint CLI loading, and zero/tiny-input and zero/negative-scale +numerical cases. Exported bundles, source checkpoint hashes, parity manifests and +runtime profiles are retained under ignored `tmp-native-int8-onnx/verified/`. +The documented export command is in the [ONNX guide](onnx.md#native-int8-training-checkpoints). +Full-dataset quality comparison with `mini_metrics`, AMP/TF32 versus full-FP32 +quality differences, target-provider execution and speed, and million-class +export capacity remain open. No new test-set evaluation was used in this work. diff --git a/docs/onnx.md b/docs/onnx.md index 430e77b..6f65fe3 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -38,7 +38,7 @@ other score conversion is added: output semantics are those of `model.eval()`. Classifier metadata includes the effective class mappings after masking. In particular, hierarchical outputs remain separate tensors in their original order. -A private CPU copy uses the example tensor's dtype. The source model's weights, +A private copy uses the example tensor's dtype and `reference_device` (CPU by default). The source model's weights, training modes, device and classifier caches remain untouched. Root DataParallel, DDP and compiled wrappers are unwrapped. Use float32 examples for portable CPU verification, including when the source was trained in bfloat16. Float16/float64 @@ -109,8 +109,53 @@ convolutions as QLinearConv and roughly halved warm local CPU inference latency, with remaining quality losses measured through `mini_metrics`; see the [calibration and metric results](benchmarks.md#onnx-activation-calibration-execution-coverage-and-macro-metrics). It remains exploratory, with no agreed production quality gate or target-device -verification. Native CUDA QT checkpoints remain a separate backend -and are not made ONNX-exportable by these floating-checkpoint experiments. +verification. Native CUDA QT checkpoint export is a separate path described below; +these floating-checkpoint PTQ experiments do not validate it. These local bundles are a foundation for Hugging Face hosting. Model cards, evaluation attachments and Hub upload commands remain separate roadmap work. + + +## Native INT8 training checkpoints + +Native `cuda-int8-linear` checkpoints can export their captured integer forward +with an explicit CUDA reference. This requires the existing quantization and +export extras in a compatible CUDA environment; deployment itself needs only +ONNX Runtime, NumPy and external preprocessing. + +```bash +CUDA_VISIBLE_DEVICES=0 .venv/bin/mt_export --weights native-int8-weights.pt \ + --output native-int8-onnx --reference-device cuda:0 +``` + +The Python API uses `export_onnx(model, float32_images, destination, +reference_device="cuda:0")`. Float32 inputs and a CUDA reference are required for +this backend. The CLI uses the scoped native-weight loader, retaining restricted +checkpoint loading. The export copy is frozen to let the exporter unpack integer +parameter storage; the caller's weights and gradient flags are preserved. + +The graph retains the training backend's dynamic symmetric row quantization, +including clipping, ties-to-even rounding and zero-row behavior. Its scaled INT8 +products lower to `MatMulInteger` with INT32 accumulation and floating row/column +scales. Activation codes are represented as unsigned codes with zero point 128; +this preserves the signed values exactly. Weights stay signed INT8. No calibration +set, floating-weight substitution or replacement classifier is used. Convolutions +remain floating, as they do in native QT training. This is distinct from the +static Percentile recipe that also quantizes convolutions. + +CUDA reference execution temporarily disables CUDA autocast and TF32 in cuDNN +and floating matrix products, restoring the caller settings afterward. Real trained images exposed a parity +failure with TF32 enabled; full-FP32 reference execution passed the original +rtol=1e-4/atol=1e-5 checks. The manifest records the reference device and TF32 choice. +This does not promise matching scores against AMP or TF32 evaluation of the same +checkpoint, whose rounding can change the subsequent integer activation codes. + +Tests cover normalized symmetric flat/hierarchical heads, EfficientNetV2-S, +active-class filtering, dynamic batches, checkpoint CLI loading and numerical +edge cases. Trained Blair checkpoints also passed checks on eight real validation +images at batches 1, 2, 4 and 8. An exported graph still needs runtime profiling +and quality evaluation on the intended provider. CUDA/ARM ONNX execution, +full-dataset quality equivalence, million-class export capacity and production +performance of this native path remain unverified. The generic exporter does not +impose a model allowlist; configurations outside this tested coverage must pass +the same export and parity checks before a bundle is published. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 7c4bdc7..006eb65 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -42,8 +42,10 @@ million-class training run remains unverified. These are acceptance targets, not claims of hardware or backend support. Laptop measurements remain useful for debugging but cannot establish gains on these systems. Do not assume a single quantized artifact or kernel recipe works across -CUDA PyTorch, ONNX GPU and ONNX ARM CPU. The current native QT checkpoint cannot -be exported directly through the ONNX path. ONNX Runtime training/fine-tuning +CUDA PyTorch, ONNX GPU and ONNX ARM CPU. Native QT now has an opt-in +[integer-forward ONNX export](onnx.md#native-int8-training-checkpoints), verified +against a full-FP32 CUDA reference on the local CPU provider. This does not +establish ONNX GPU/ARM execution or performance. ONNX Runtime training/fine-tuning would be a separate integration; an inference export does not provide it. The shared dataset harness now selects backbone and head independently while diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 8ee7106..1e1e272 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -140,8 +140,11 @@ model.load_state_dict(state) Use the same architecture and intended dtype. This restores model state; create and restore optimizer/scheduler/scaler state in their normal order separately. The same model supports CUDA inference with `eval()` and `inference_mode()`. -ONNX export, checkpoint averaging, DDP/FSDP, quantized activation normalization -and integer convolution training are not established for this path. Distributed training and +An opt-in [ONNX export path](onnx.md#native-int8-training-checkpoints) captures +the integer forward using a full-FP32 CUDA reference and verifies ONNX Runtime CPU +parity. Target-provider performance remains unverified. Checkpoint averaging, +DDP/FSDP, quantized activation normalization and integer convolution training +are not established for this path. Distributed training and EMA are rejected by the training entry point. ## Model compilation modes diff --git a/mini_trainer/export.py b/mini_trainer/export.py index 7d7fa30..b07c9fe 100644 --- a/mini_trainer/export.py +++ b/mini_trainer/export.py @@ -10,6 +10,7 @@ from mini_trainer.modeling import Classifier, classification_module from mini_trainer.modeling.architectures.load import get_dynamic_model, resolve_backbone_getter from mini_trainer.modeling.onnx import export_onnx +from mini_trainer.modeling.quantized_training import load_training_weights def main( @@ -20,13 +21,13 @@ def main( model_args: dict | None = None, dynamic_batch: bool = True, batch_size: int = 2, + reference_device: str = "cpu", ): if batch_size < 1: raise ValueError("batch_size must be positive.") with Path(weights).open("rb") as handle: checkpoint_hash = hashlib.file_digest(handle, "sha256").hexdigest() - handle.seek(0) - state = torch.load(handle, map_location="cpu", weights_only=True) + state = load_training_weights(weights, map_location="cpu") state = state.get("model", state) metadata = Classifier.extract_metadata(state) model_type = metadata.get("backbone_class") @@ -45,7 +46,15 @@ def main( if not input_shape or any(size < 1 for size in input_shape): raise ValueError("input_shape must contain positive non-batch dimensions.") sample = torch.zeros(batch_size, *input_shape) - return export_onnx(model, sample, output, preprocessing=preprocessing, dynamic_batch=dynamic_batch, checkpoint_sha256=checkpoint_hash) + return export_onnx( + model, + sample, + output, + preprocessing=preprocessing, + dynamic_batch=dynamic_batch, + checkpoint_sha256=checkpoint_hash, + reference_device=reference_device, + ) def run(): @@ -57,6 +66,7 @@ def run(): parser.add_argument("--model-args", type=json.loads, help="JSON constructor arguments for the existing architecture loader.") parser.add_argument("--static-batch", action="store_false", dest="dynamic_batch", help="Export a fixed batch size.") parser.add_argument("--batch-size", type=int, default=2, help="Example/static batch size (default: 2).") + parser.add_argument("--reference-device", default="cpu", help="PyTorch parity device; native INT8 training requires cuda.") args = vars(parser.parse_args()) if args["preprocessing"] is not None: args["preprocessing"] = json.loads(args["preprocessing"].read_text()) diff --git a/mini_trainer/modeling/_onnx_quantized.py b/mini_trainer/modeling/_onnx_quantized.py new file mode 100644 index 0000000..961c869 --- /dev/null +++ b/mini_trainer/modeling/_onnx_quantized.py @@ -0,0 +1,16 @@ +"""Optional ONNX lowering of the native QT forward's scaled integer product.""" + +import onnx +from onnxscript import FLOAT, INT8 +from onnxscript import opset18 as op + + +def scaled_int8_mm(left: INT8, right: INT8, row_scale: FLOAT, column_scale: FLOAT) -> FLOAT: + # Preserve signed codes exactly while using the U8/S8 CPU MatMulInteger path. + # Replacing the row quantizer with DynamicQuantizeLinear would change its + # symmetric range, rounding and zero-row behavior. + offset = op.Constant(value=onnx.helper.make_tensor("offset", onnx.TensorProto.INT32, [], [128])) + zero = op.Constant(value=onnx.helper.make_tensor("zero", onnx.TensorProto.UINT8, [], [128])) + unsigned = op.Cast(op.Add(op.Cast(left, to=onnx.TensorProto.INT32), offset), to=onnx.TensorProto.UINT8) + accumulator = op.MatMulInteger(unsigned, right, zero) + return op.Mul(op.Mul(op.Cast(accumulator, to=onnx.TensorProto.FLOAT), op.Unsqueeze(row_scale, [1])), op.Unsqueeze(column_scale, [0])) diff --git a/mini_trainer/modeling/onnx.py b/mini_trainer/modeling/onnx.py index eb3e30a..cb73e7f 100644 --- a/mini_trainer/modeling/onnx.py +++ b/mini_trainer/modeling/onnx.py @@ -4,6 +4,7 @@ import hashlib import json from collections.abc import Sequence +from contextlib import contextmanager from importlib.metadata import version from pathlib import Path from tempfile import TemporaryDirectory, TemporaryFile @@ -78,16 +79,31 @@ def _dependencies(): return onnx, onnxruntime -def _copy_for_export(model, dtype): +def _copy_for_export(model, dtype, device="cpu"): while isinstance(model, (nn.DataParallel, nn.parallel.DistributedDataParallel)) or hasattr(model, "_orig_mod"): model = model._orig_mod if hasattr(model, "_orig_mod") else model.module - model = copy.deepcopy(model).cpu().to(dtype=dtype).eval() + model = copy.deepcopy(model).to(device=device, dtype=dtype).eval() for module in model.modules(): if isinstance(module, Classifier): module._dirty_cache.clear() return model +@contextmanager +def _reference_precision(device): + """Use full FP32 CUDA arithmetic for the cross-provider reference.""" + if device.type != "cuda": + yield + return + matmul_tf32 = torch.backends.cuda.matmul.allow_tf32 + try: + torch.backends.cuda.matmul.allow_tf32 = False + with torch.autocast("cuda", enabled=False), torch.backends.cudnn.flags(allow_tf32=False): + yield + finally: + torch.backends.cuda.matmul.allow_tf32 = matmul_tf32 + + def export_onnx( model: nn.Module, example_input: torch.Tensor, @@ -101,11 +117,12 @@ def export_onnx( rtol: float = 1e-4, atol: float = 1e-5, checkpoint_sha256: str | None = None, + reference_device: str | torch.device = "cpu", ) -> Path: """Export the actual eval forward without a backbone or head allowlist. example_input is a preprocessed batched floating-point tensor. A private copy - runs on CPU in its dtype; the caller's weights, modes, device and caches are + runs on reference_device in its dtype; the caller's weights, modes, device and caches are untouched. DDP/DataParallel/compiled wrappers are unwrapped. Tensor/list/tuple/ dict outputs are flattened, with structure and classifier metadata in the manifest. Dynamic batch is verified at sizes 1, 2 and 4, plus optional real @@ -114,6 +131,9 @@ def export_onnx( Preprocessing remains external; supply a JSON-compatible recipe for deployment. Unsupported operators propagate exporter errors; models are never substituted. Existing output directories are never overwritten. + Native INT8 training models require a CUDA reference_device and float32 inputs; + their captured integer forward is verified against ONNX Runtime CPU inference. + CUDA reference execution disables TF32/autocast temporarily, restoring caller settings. """ if not isinstance(example_input, torch.Tensor) or example_input.ndim < 2 or example_input.shape[0] < 1: raise ValueError("example_input must be a nonempty batched tensor.") @@ -127,15 +147,28 @@ def export_onnx( if destination.exists(): raise FileExistsError(f"Export destination already exists: {destination}") onnx, ort = _dependencies() + reference_device = torch.device(reference_device) + native_int8 = any(getattr(parameter, "_is_quantized_training", False) for parameter in model.parameters()) + translations = {} + if native_int8: + if reference_device.type != "cuda" or example_input.dtype != torch.float32: + raise ValueError("Native INT8 ONNX export requires reference_device='cuda' and float32 example inputs.") + from ._onnx_quantized import scaled_int8_mm + + translations[torch.ops.mini_trainer.scaled_int8_mm.default] = scaled_int8_mm sample = example_input.detach().cpu().clone() - reference = _copy_for_export(model, sample.dtype) + reference = _copy_for_export(model, sample.dtype, reference_device) + if native_int8: + # Tensor-subclass decomposition must not assign requires_grad to codes. + reference.requires_grad_(False) with TemporaryFile() as state_file: torch.save(reference.state_dict(), state_file) state_file.seek(0) state_hash = hashlib.file_digest(state_file, "sha256").hexdigest() # Batch one is specialized by torch.export; trace at batch two when dynamic. trace_input = sample[:1].repeat(2, *([1] * (sample.ndim - 1))) if dynamic_batch else sample - with torch.no_grad(): + trace_input = trace_input.to(reference_device) + with torch.no_grad(), _reference_precision(reference_device): observed = reference(trace_input) tensors = _flatten(observed) if not tensors: @@ -163,7 +196,12 @@ def export_onnx( bundle = Path(temporary) / "bundle" bundle.mkdir() graph = bundle / "model.onnx" - with torch.no_grad(), torch.random.fork_rng(devices=[]): + rng_devices = ( + [reference_device.index if reference_device.index is not None else torch.cuda.current_device()] + if reference_device.type == "cuda" + else [] + ) + with torch.no_grad(), torch.random.fork_rng(devices=rng_devices), _reference_precision(reference_device): torch.onnx.export( exported, (trace_input,), @@ -174,6 +212,7 @@ def export_onnx( dynamo=True, dynamic_shapes=({0: torch.export.Dim("batch", min=1)},) if dynamic_batch else None, external_data=True, + custom_translation_table=translations, ) onnx.checker.check_model(str(graph)) options = ort.SessionOptions() @@ -184,15 +223,15 @@ def export_onnx( for inputs in checks: if inputs.dtype != sample.dtype or inputs.shape[1:] != sample.shape[1:]: raise ValueError("Verification inputs must match the example's dtype and non-batch dimensions.") - with torch.inference_mode(): - expected_output = reference(inputs) + with torch.inference_mode(), _reference_precision(reference_device): + expected_output = reference(inputs.to(reference_device)) if _structure(expected_output, iter(names)) != structure: raise ValueError("Model output structure changed during verification.") expected = _flatten(expected_output) actual = session.run(names, {"images": inputs.numpy()}) batch_errors = [] for name, result, value in zip(names, actual, expected, strict=True): - target = value.detach().numpy() + target = value.detach().cpu().numpy() if ( result.shape != target.shape or result.dtype != target.dtype @@ -224,11 +263,20 @@ def export_onnx( ], "output_structure": structure, "output_semantics": "model_eval_forward", + "quantized_training_forward": native_int8, "classifiers": classifiers, "preprocessing": {"in_graph": False, "recipe": preprocessing, "requires_configuration": preprocessing is None}, "opset": opset_version, "versions": {name: version(name) for name in ("mini_trainer", "torch", "torchvision", "onnx", "onnxscript", "onnxruntime")}, - "verification": {"provider": "CPUExecutionProvider", "rtol": rtol, "atol": atol, "cases": errors}, + "verification": { + "provider": "CPUExecutionProvider", + "reference_device": str(reference_device), + "reference_tf32": False if reference_device.type == "cuda" else None, + "reference_autocast": False if reference_device.type == "cuda" else None, + "rtol": rtol, + "atol": atol, + "cases": errors, + }, } (bundle / "manifest.json").write_text(json.dumps(manifest, indent=2, allow_nan=False) + "\n") if destination.exists(): diff --git a/tests/test_onnx_quantized_training.py b/tests/test_onnx_quantized_training.py new file mode 100644 index 0000000..8574a41 --- /dev/null +++ b/tests/test_onnx_quantized_training.py @@ -0,0 +1,109 @@ +"""Native CUDA QT forward to portable integer ONNX, without float substitution.""" + +import json +import os + +import pytest +import torch + +from mini_trainer.export import main as export_checkpoint +from mini_trainer.hierarchical.model import HierarchicalClassifier +from mini_trainer.modeling import Classifier, classification_module +from mini_trainer.modeling.onnx import export_onnx +from mini_trainer.modeling.quantized_training import prepare_quantized_training +from tests.test_integration_train import TinyMockModel + +onnx = pytest.importorskip("onnx") +pytest.importorskip("onnxruntime") +pytest.importorskip("onnxscript") +pytest.importorskip("torchao") +pytest.importorskip("triton") + + +def cuda(): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for native INT8 ONNX parity") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + return torch.device("cuda:0") + + +def test_native_export_requires_explicit_cuda_reference(tmp_path): + model = torch.nn.Linear(8, 3) + prepare_quantized_training(model) + with pytest.raises(ValueError, match="reference_device='cuda'"): + export_onnx(model, torch.ones(2, 8), tmp_path / "bundle") + assert not (tmp_path / "bundle").exists() + + +@pytest.mark.parametrize("backbone", ["tiny", "efficientnet_v2_s"]) +@pytest.mark.parametrize("head", [Classifier, HierarchicalClassifier]) +def test_normalized_symmetric_native_heads_export(tmp_path, monkeypatch, head, backbone): + device = cuda() + monkeypatch.setattr(torch.backends.cudnn, "allow_tf32", True) + monkeypatch.setattr(torch.backends.cuda.matmul, "allow_tf32", True) + torch.manual_seed(42) + kwargs = {} + if head is HierarchicalClassifier: + kwargs["sparse_masks"] = [torch.tensor([0, 0, 1])] + model, _ = head.build( + model_type=TinyMockModel() if backbone == "tiny" else backbone, + num_classes=3, + hidden=True, + normalized=True, + model_args={"pretrained": False}, + device=device, + **kwargs, + ) + prepare_quantized_training(model) + classification_module(model).set_active_features([0, 2]) + model.train() + parameters = list(model.parameters()) + before = [(p.requires_grad, p.int_data.clone(), p.scale.clone()) for p in parameters if hasattr(p, "int_data")] + sample = torch.randn(2, 3, 32, 32) + path = export_onnx(model, sample, tmp_path / "bundle", reference_device=device, verification_inputs=[torch.randn(3, 3, 32, 32)]) + assert model.training + assert torch.backends.cudnn.allow_tf32 + assert torch.backends.cuda.matmul.allow_tf32 + for p, (requires_grad, codes, scales) in zip([p for p in parameters if hasattr(p, "int_data")], before, strict=True): + assert p.requires_grad == requires_grad + torch.testing.assert_close(p.int_data, codes, rtol=0, atol=0) + torch.testing.assert_close(p.scale, scales, rtol=0, atol=0) + manifest = json.loads((path / "manifest.json").read_text()) + assert manifest["quantized_training_forward"] is True + assert manifest["verification"]["reference_device"] == "cuda:0" + assert {case["batch_size"] for case in manifest["verification"]["cases"]} == {1, 2, 3, 4} + assert len(manifest["outputs"]) == (2 if head is HierarchicalClassifier else 1) + graph = onnx.load(path / "model.onnx") + assert sum(node.op_type == "MatMulInteger" for node in graph.graph.node) == 2 + + +def test_native_checkpoint_cli_export(tmp_path): + cuda() + model, _ = Classifier.build(model_type=TinyMockModel(), num_classes=3, hidden=True, normalized=True, model_args={"pretrained": False}) + prepare_quantized_training(model) + weights = tmp_path / "weights.pt" + torch.save(model.state_dict(), weights) + safe_globals = list(torch.serialization.get_safe_globals()) + path = export_checkpoint(str(weights), str(tmp_path / "bundle"), input_shape=[3, 5, 5], reference_device="cuda:0") + assert torch.serialization.get_safe_globals() == safe_globals + assert json.loads((path / "manifest.json").read_text())["quantized_training_forward"] is True + + +def test_native_linear_zero_tiny_and_negative_scale_rows(tmp_path): + device = cuda() + torch.manual_seed(42) + model = torch.nn.Linear(17, 5).to(device) + prepare_quantized_training(model) + with torch.no_grad(): + model.weight.scale[0].neg_() + model.weight.scale[1].zero_() + inputs = torch.randn(2, 3, 17) + inputs[0, 0].zero_() + inputs[0, 1].mul_(1e-15) + inputs[1, 0].mul_(1e5) + with torch.autocast("cuda", dtype=torch.float16): + path = export_onnx(model, inputs, tmp_path / "bundle", reference_device=device) + assert torch.is_autocast_enabled("cuda") + manifest = json.loads((path / "manifest.json").read_text()) + assert manifest["outputs"][0]["dtype"] == "float32" From fd1c7791ebd6dacd570370a433aba958ef06931a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 12:24:18 +0200 Subject: [PATCH 053/155] docs: evaluate native ONNX quality across Blair validation --- docs/benchmarks.md | 73 ++++++++++++++++++++++++++++++++++++++++++++-- docs/onnx.md | 10 +++++-- 2 files changed, 78 insertions(+), 5 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 42e656d..8950331 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1400,6 +1400,73 @@ restricted checkpoint CLI loading, and zero/tiny-input and zero/negative-scale numerical cases. Exported bundles, source checkpoint hashes, parity manifests and runtime profiles are retained under ignored `tmp-native-int8-onnx/verified/`. The documented export command is in the [ONNX guide](onnx.md#native-int8-training-checkpoints). -Full-dataset quality comparison with `mini_metrics`, AMP/TF32 versus full-FP32 -quality differences, target-provider execution and speed, and million-class -export capacity remain open. No new test-set evaluation was used in this work. +The full-validation follow-up below measures `mini_metrics` quality and numerical +limits, including AMP/TF32 comparisons. Target-provider execution and speed, +confidence-threshold equivalence and million-class export capacity remain open. +No new test-set evaluation was used in this work. + + +### Native ONNX full-validation quality and numerical limits + +The native QT checkpoints and ONNX bundles above were compared on all 912 Blair +validation images in batches of eight. Checkpoint and graph/external-weight hashes +were verified, class mappings were checked against the dataset manifest, and +all modes used the same preprocessed inputs. This is inference comparison of each +fixed checkpoint, not a new training comparison or a test-set evaluation. + +ONNX Runtime CPU and the full-FP32 CUDA reference made **identical top-1 +predictions for every image**, at both hierarchical levels. Their Macro-F1, +Macro-Recall, Macro-Precision, Coverage and Theil's U therefore matched exactly. +Values below are proportions from `mini_metrics` at clean checkout revision +`70cc69adc05362863439277048e06386c1f885e1`, using the preceding full-coverage metric +call with no threshold optimization, filtering or resampling. + +| Output / execution | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat / ONNX and CUDA FP32 | 0.769112 | 0.758984 | 0.796193 | 1.000000 | 0.810716 | +| Flat / CUDA AMP FP16 | 0.768593 | 0.758712 | 0.795913 | 1.000000 | 0.810290 | +| Hierarchical fine / ONNX and CUDA FP32 | 0.730698 | 0.720275 | 0.799803 | 1.000000 | 0.787638 | +| Hierarchical fine / CUDA AMP FP16 | 0.726950 | 0.716638 | 0.795216 | 1.000000 | 0.786153 | +| Hierarchical parent / ONNX and CUDA FP32 | 0.867051 | 0.844461 | 0.913180 | 1.000000 | 0.857871 | +| Hierarchical parent / CUDA AMP FP16 | 0.867051 | 0.844461 | 0.913180 | 1.000000 | 0.857871 | + +FP32/ONNX accuracy was 83.55% flat and 81.47% / 91.45% hierarchical fine / parent. +AMP changed two flat and two hierarchical fine predictions; parent predictions +were unchanged. AMP fine accuracy was 81.36%; flat accuracy remained 83.55% +despite its two changed decisions. A third mode allowed cuDNN TF32 while keeping +floating matmul TF32 disabled and autocast off; it produced the same decisions +and metrics as the full-FP32 reference in this batch-eight run. Allowing TF32 does +not force a particular cuDNN kernel and is not evidence of TF32 execution. This +result does not establish parity for other batch sizes, devices or kernel choices. + +The full split **did not pass universal score parity** at the export defaults +rtol=1e-4/atol=1e-5, despite identical top-1 decisions: + +| Output | Images with any score outside tolerance | Score elements outside tolerance | Maximum absolute score error | +| --- | ---: | ---: | ---: | +| Flat | 6 / 912 | 142 | 0.022024 | +| Hierarchical fine | 16 / 912 | 372 | 0.012286 | +| Hierarchical parent | 17 / 912 | 222 | 0.010367 | + +The earlier eight-image export check remains valid for its tested inputs, but +must not be read as a guarantee for unseen inputs. Supplying these affected +images to strict export verification would fail; the tolerance was not relaxed. +Coverage is 100% by construction here. Confidence differences can still matter +for abstention, threshold calibration or downstream consumers even when argmax +is unchanged. No production quality acceptance criterion has been established. + +An affected-batch diagnostic exposed hidden inputs and actual integer activation +codes without changing weights. At validation index 164, the maximum hidden-input +CPU/CUDA difference was 1.07e-6. One of the hidden Linear input codes differed, +then two output Linear input codes differed; the maximum score difference was +0.009955. The other seven images in that batch had identical integer input codes. +This localizes amplification to quantization boundaries following small floating +input differences. It does not justify changing the trained quantizer or claiming +that all outliers have the same cause. + +Exact scores, prediction CSVs, package/source provenance, per-level differences, +metric JSON and the diagnostic are retained under ignored +`tmp-native-int8-validation/`. These are local x86 CPU / laptop CUDA measurements, +not target GPU or ARM verification, and no latency claim is derived from this +quality run. Confidence-thresholded evaluation, repeated-seed training quality, +large-vocabulary quality and the intended target-machine runs remain open. diff --git a/docs/onnx.md b/docs/onnx.md index 6f65fe3..f40d613 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -155,7 +155,13 @@ active-class filtering, dynamic batches, checkpoint CLI loading and numerical edge cases. Trained Blair checkpoints also passed checks on eight real validation images at batches 1, 2, 4 and 8. An exported graph still needs runtime profiling and quality evaluation on the intended provider. CUDA/ARM ONNX execution, -full-dataset quality equivalence, million-class export capacity and production -performance of this native path remain unverified. The generic exporter does not +million-class export capacity and production performance of this native path +remain unverified. On the full Blair validation split, top-1 predictions and the +requested macro metrics matched the full-FP32 CUDA reference, but some image +scores exceeded the strict export tolerance; see the +[full-validation results](benchmarks.md#native-onnx-full-validation-quality-and-numerical-limits). +Supply representative `verification_inputs` and evaluate deployment thresholds +separately; passing the default sample checks does not establish universal score +parity or confidence-threshold equivalence. The generic exporter does not impose a model allowlist; configurations outside this tested coverage must pass the same export and parity checks before a bundle is published. From c103492122363512452f94a93a15021d11462933 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 12:41:43 +0200 Subject: [PATCH 054/155] fix: prevent INT8 accumulator saturation in large-class gradients --- docs/benchmarks.md | 38 +++++++++++++ docs/onnx.md | 5 +- docs/quantized-training.md | 12 +++++ mini_trainer/modeling/_onnx_quantized.py | 18 ++++++- mini_trainer/modeling/_quantized_matmul.py | 58 ++++++++++++++++++++ tests/test_onnx_quantized_training.py | 15 ++++++ tests/test_quantized_matmul.py | 63 ++++++++++++++++++++++ 7 files changed, 206 insertions(+), 3 deletions(-) create mode 100644 tests/test_quantized_matmul.py diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 8950331..1e14fdb 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1470,3 +1470,41 @@ metric JSON and the diagnostic are retained under ignored not target GPU or ARM verification, and no latency claim is derived from this quality run. Confidence-thresholded evaluation, repeated-seed training quality, large-vocabulary quality and the intended target-machine runs remain open. + +### Long-contraction INT8 accumulator correctness + +Reviewing the million-class training envelope exposed a correctness issue before +full-model capacity testing: an INT8 input-gradient product contracts over the +number of output classes. INT32 accumulation is not safe for arbitrary signed +INT8 values once the contraction exceeds 131,071 elements. Checking only finite +losses or gradients cannot detect saturation. + +A bounded local CUDA probe with contraction 150,000 and every operand equal to +127 returned 2,147,483,648 instead of the expected FP32 value 2,419,350,016. The +result was finite. The exact integer product is 2,419,350,000 before FP32 rounding. +This is an adversarial arithmetic regression, not a claim that every large-class +training batch previously saturated. + +The backend now keeps the existing tuned kernel for contractions within the +INT32-safe bound. Longer products periodically transfer bounded INT32 partial +sums into INT64 storage inside the kernel, then convert the total to floating +point and apply row/column scales once. This avoids both integer saturation and +the loss of small residuals when large opposite-signed partial sums are converted +to float before cancellation. No floating weight master or floating matrix +multiplication is introduced. The new long-contraction kernel has a fixed tile; +its throughput on the intended accelerators still needs measurement. + +ONNX lowering likewise combines bounded MatMulInteger results in INT64 before +scaling. Its conservative chunk limit also bounds raw unsigned-activation times +signed-weight products before zero-point compensation. Ordinary 1280-wide +EfficientNetV2 forward head products retain one MatMulInteger per Linear; the +large-class risk primarily concerns the training input gradient. + +Regression coverage includes signed extrema around the 131,071/131,072 boundary, +150k and 1M contractions, strided weights, per-row and scalar/per-column scales, +large-sum cancellation, compiled execution and a 150k-input ONNX Linear. A separate +one-million-output, eight-input Linear with unit weights/inputs verifies both the +input gradient (1,000,000 per element) and weight gradient (2 per element for a +two-sample sum loss). This is a bounded gradient correctness check, **not** a +million-class EfficientNetV2 training, optimizer-memory or deployment benchmark. +Full-model initialization and training capacity remain open. diff --git a/docs/onnx.md b/docs/onnx.md index f40d613..b01255c 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -136,8 +136,9 @@ parameter storage; the caller's weights and gradient flags are preserved. The graph retains the training backend's dynamic symmetric row quantization, including clipping, ties-to-even rounding and zero-row behavior. Its scaled INT8 -products lower to `MatMulInteger` with INT32 accumulation and floating row/column -scales. Activation codes are represented as unsigned codes with zero point 128; +products lower to `MatMulInteger` with bounded INT32 partial accumulation and +floating row/column scales. Long contractions combine partial sums in INT64 +before converting and applying scales, preventing accumulator saturation. Activation codes are represented as unsigned codes with zero point 128; this preserves the signed values exactly. Weights stay signed INT8. No calibration set, floating-weight substitution or replacement classifier is used. Convolutions remain floating, as they do in native QT training. This is distinct from the diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 1e1e272..1dca00e 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -210,3 +210,15 @@ the floating path despite substantially smaller parameter storage. Corrected benchmark logging preserves CUDA peaks across all resets; older dataset-run CUDA readings do not establish whole-run memory reductions. Compiled training now saves unwrapped model keys so ordinary checkpoint restoration and resume remain valid. + +## Large-class accumulator bounds + +Input-gradient products contract over the output class count. For contractions +above 131,071, the backend now combines bounded INT32 dot products in INT64 before +converting and scaling the result. This prevents finite but saturated gradients +for large vocabularies. Shorter contractions keep the existing tuned kernel. +ONNX export also bounds integer partial products and combines them in INT64. +See the [arithmetic regression and limits](benchmarks.md#long-contraction-int8-accumulator-correctness). +A million-output Linear gradient test does not establish that a full million-class +EfficientNetV2 training configuration fits the available GPU; parameter, +initialization, gradient and optimizer storage still require separate measurement. diff --git a/mini_trainer/modeling/_onnx_quantized.py b/mini_trainer/modeling/_onnx_quantized.py index 961c869..e6a8190 100644 --- a/mini_trainer/modeling/_onnx_quantized.py +++ b/mini_trainer/modeling/_onnx_quantized.py @@ -12,5 +12,21 @@ def scaled_int8_mm(left: INT8, right: INT8, row_scale: FLOAT, column_scale: FLOA offset = op.Constant(value=onnx.helper.make_tensor("offset", onnx.TensorProto.INT32, [], [128])) zero = op.Constant(value=onnx.helper.make_tensor("zero", onnx.TensorProto.UINT8, [], [128])) unsigned = op.Cast(op.Add(op.Cast(left, to=onnx.TensorProto.INT32), offset), to=onnx.TensorProto.UINT8) - accumulator = op.MatMulInteger(unsigned, right, zero) + # Static weight width is available when lowering a Linear. Accumulate long + # contractions in INT64 between safe INT32 products, just as the CUDA kernel. + contraction = right.shape[0] if right.shape is not None else None + if not isinstance(contraction, int): + raise ValueError("Native INT8 ONNX lowering requires a statically known contraction width.") + # Also bound the raw U8/S8 product before zero-point compensation, so the + # provider need not rely on cancellation of overflowing intermediates. + limit = (2**31 - 1) // (255 * 128) + if contraction <= limit: + accumulator = op.MatMulInteger(unsigned, right, zero) + else: + accumulator = None + for start in range(0, contraction, limit): + a = op.Slice(unsigned, [start], [min(start + limit, contraction)], [1]) + b = op.Slice(right, [start], [min(start + limit, contraction)], [0]) + partial = op.Cast(op.MatMulInteger(a, b, zero), to=onnx.TensorProto.INT64) + accumulator = partial if accumulator is None else op.Add(accumulator, partial) return op.Mul(op.Mul(op.Cast(accumulator, to=onnx.TensorProto.FLOAT), op.Unsqueeze(row_scale, [1])), op.Unsqueeze(column_scale, [0])) diff --git a/mini_trainer/modeling/_quantized_matmul.py b/mini_trainer/modeling/_quantized_matmul.py index 0f772c7..ec4d62c 100644 --- a/mini_trainer/modeling/_quantized_matmul.py +++ b/mini_trainer/modeling/_quantized_matmul.py @@ -2,6 +2,7 @@ import torch import triton +import triton.language as tl from torchao.prototype.quantized_training.int8_mm import _scaled_int8_mm_kernel as _upstream_kernel from triton.compiler.errors import CompileTimeAssertionFailure from triton.runtime.errors import OutOfResources, PTXASError @@ -29,12 +30,69 @@ def _benchmark(kernel, quantiles): ) +@triton.jit +def _wide_kernel( + A, + B, + C, + ROW, + COL, + M: tl.constexpr, + N: tl.constexpr, + K: tl.constexpr, + AM: tl.constexpr, + AK: tl.constexpr, + BK: tl.constexpr, + BN: tl.constexpr, + SCALAR_COL: tl.constexpr, + BM: tl.constexpr = 16, + BN_TILE: tl.constexpr = 64, + BLOCK_K: tl.constexpr = 128, +): + rows = tl.program_id(0) * BM + tl.arange(0, BM) + cols = tl.program_id(1) * BN_TILE + tl.arange(0, BN_TILE) + inner = tl.arange(0, BLOCK_K) + total = tl.zeros((BM, BN_TILE), tl.int64) + partial = tl.zeros((BM, BN_TILE), tl.int32) + # A product is at most (-128)*(-128). Flush before INT32 can saturate. + safe_blocks: tl.constexpr = ((2**31 - 1) // (128 * 128)) // BLOCK_K + for block in range(tl.cdiv(K, BLOCK_K)): + k = block * BLOCK_K + inner + a = tl.load(A + rows[:, None] * AM + k[None, :] * AK, (rows[:, None] < M) & (k[None, :] < K), 0) + b = tl.load(B + k[:, None] * BK + cols[None, :] * BN, (k[:, None] < K) & (cols[None, :] < N), 0) + partial = tl.dot(a, b, partial, out_dtype=tl.int32) + if (block + 1) % safe_blocks == 0: + total += partial.to(tl.int64) + partial = tl.zeros((BM, BN_TILE), tl.int32) + total += partial.to(tl.int64) + row_scale = tl.load(ROW + rows, rows < M, 0) + col_scale = tl.load(COL + (tl.zeros((BN_TILE,), tl.int32) if SCALAR_COL else cols), cols < N, 0) + result = total.to(tl.float32) * row_scale[:, None].to(tl.float32) * col_scale[None, :].to(tl.float32) + tl.store(C + rows[:, None] * N + cols[None, :], result, (rows[:, None] < M) & (cols[None, :] < N)) + + @torch.library.custom_op("mini_trainer::scaled_int8_mm", mutates_args=()) def scaled_int8_mm(left: torch.Tensor, right: torch.Tensor, row_scale: torch.Tensor, column_scale: torch.Tensor) -> torch.Tensor: """Compute a scaled INT8 matrix product using locally tuned CUDA kernels.""" rows, contraction = left.shape columns = right.shape[1] output = torch.empty((rows, columns), dtype=row_scale.dtype, device=left.device) + if contraction > (2**31 - 1) // (128 * 128): + with torch.cuda.device(left.device): + _wide_kernel[(triton.cdiv(rows, 16), triton.cdiv(columns, 64))]( + left, + right, + output, + row_scale, + column_scale, + rows, + columns, + contraction, + *left.stride(), + *right.stride(), + SCALAR_COL=column_scale.numel() == 1, + ) + return output def grid(meta): return (triton.cdiv(rows, meta["BLOCK_M"]) * triton.cdiv(columns, meta["BLOCK_N"]),) diff --git a/tests/test_onnx_quantized_training.py b/tests/test_onnx_quantized_training.py index 8574a41..bc4f96b 100644 --- a/tests/test_onnx_quantized_training.py +++ b/tests/test_onnx_quantized_training.py @@ -107,3 +107,18 @@ def test_native_linear_zero_tiny_and_negative_scale_rows(tmp_path): assert torch.is_autocast_enabled("cuda") manifest = json.loads((path / "manifest.json").read_text()) assert manifest["outputs"][0]["dtype"] == "float32" + + +def test_wide_linear_onnx_avoids_int32_saturation(tmp_path): + device = cuda() + model = torch.nn.Linear(150000, 2, bias=False).to(device) + with torch.no_grad(): + model.weight.fill_(1) + prepare_quantized_training(model) + sample = torch.ones(2, 150000) + path = export_onnx(model, sample, tmp_path / "bundle", reference_device=device) + graph = onnx.load(path / "model.onnx") + assert sum(node.op_type == "MatMulInteger" for node in graph.graph.node) > 1 + with torch.inference_mode(): + actual = model(sample.to(device)) + torch.testing.assert_close(actual, torch.full((2, 2), 150000.0, device=device), rtol=1e-6, atol=1e-3) diff --git a/tests/test_quantized_matmul.py b/tests/test_quantized_matmul.py new file mode 100644 index 0000000..66b2132 --- /dev/null +++ b/tests/test_quantized_matmul.py @@ -0,0 +1,63 @@ +"""Long-contraction integer arithmetic, including large-class input gradients.""" + +import os + +import pytest +import torch + +pytest.importorskip("torchao") +pytest.importorskip("triton") + + +def cuda(): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for INT8 accumulator validation") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + return "cuda:0" + + +@pytest.mark.parametrize("contraction", [131071, 131072, 150000, 1000000]) +def test_long_dot_matches_int64_reference(contraction): + device = cuda() + from mini_trainer.modeling._quantized_training import scaled_int8_mm + + left = torch.tensor([-128, 127], dtype=torch.int8, device=device)[:, None].expand(2, contraction).contiguous() + right = torch.tensor([-128, 127, 1], dtype=torch.int8, device=device)[:, None].expand(3, contraction).contiguous().T + row_scale = torch.tensor([0.01, 0.02], device=device) + column_scale = torch.tensor([1.0, 0.5, -2.0], device=device) + expected = (left.cpu().long() @ right.cpu().long()).float() * row_scale.cpu()[:, None] * column_scale.cpu()[None, :] + actual = scaled_int8_mm(left, right, row_scale, column_scale) + torch.testing.assert_close(actual.cpu(), expected, rtol=1e-6, atol=1e-4) + + +@pytest.mark.parametrize("compiled", [False, True]) +def test_large_opposite_partial_sums_preserve_small_residual(compiled): + device = cuda() + from mini_trainer.modeling._quantized_training import scaled_int8_mm + + contraction = 262145 + left = torch.full((1, contraction), 127, dtype=torch.int8, device=device) + # Make a large partial sum inexact in FP32; a float partial-sum workaround + # loses one unit here even though the final integer result is small. + left[0, 0] = 126 + right = torch.full((contraction, 1), 127, dtype=torch.int8, device=device) + right[:131072].neg_() + scale = torch.ones(1, device=device) + function = torch.compile(scaled_int8_mm, fullgraph=True) if compiled else scaled_int8_mm + actual = function(left, right, scale, scale) + torch.testing.assert_close(actual, torch.full((1, 1), 127.0**2 + 127, device=device), rtol=0, atol=0) + + +def test_million_output_linear_input_gradient(): + device = cuda() + from mini_trainer.modeling.quantized_training import prepare_quantized_training + + model = torch.nn.Linear(8, 1000000, bias=False, device=device) + with torch.no_grad(): + model.weight.fill_(1) + prepare_quantized_training(model) + inputs = torch.ones(2, 8, device=device, requires_grad=True) + model(inputs).sum().backward() + torch.testing.assert_close(inputs.grad, torch.full_like(inputs, 1000000), rtol=1e-6, atol=1e-3) + torch.testing.assert_close(model.weight.grad, torch.full_like(model.weight.grad, 2), rtol=1e-6, atol=1e-5) From 652f8b18004952f0fe2b3a781e7d9afc7728ec0f Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 12:56:44 +0200 Subject: [PATCH 055/155] perf: reuse spherical initializer buffers for million-class heads --- docs/benchmarks.md | 50 +++++++++++++++++++++++++++++ mini_trainer/modeling/classifier.py | 6 +++- tests/test_classifier_shapes.py | 45 ++++++++++++++++++++++++++ 3 files changed, 100 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 1e14fdb..5d13c5c 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1508,3 +1508,53 @@ input gradient (1,000,000 per element) and weight gradient (2 per element for a two-sample sum loss). This is a bounded gradient correctness check, **not** a million-class EfficientNetV2 training, optimizer-memory or deployment benchmark. Full-model initialization and training capacity remain open. + +### Million-class normalized-head initialization memory + +The normalized spherical initializer still retained several class-by-width +intermediates after the earlier Gram-matrix fix. A one-million-class, 1280-wide +symmetric normalized `Classifier` exposed this independently of training. +An uncapped WSL run reached 25,619,099,648 allocated bytes (23.86 GiB) on the +16 GiB laptop GPU and was interrupted, rather than treating driver-managed +allocations beyond dedicated memory as a successful capacity result. Per-process +GPU memory was unavailable from `nvidia-smi` in this environment. + +The repeat explicitly capped the PyTorch allocator at 90% of device capacity +(about 14.4 GiB). The original implementation failed while requesting another +4.77 GiB temporary, after reaching 15,383,099,904 allocated bytes. The OOM message's +non-PyTorch usage field was nonsensical on this WSL build; the report retains it, +but conclusions use the configured allocator limit and PyTorch allocation counters. + +The update now reuses the gradient buffer for subtraction and scaling, preserving +the original arithmetic order, and releases gradient/projection buffers before +the next iteration. This removes additional full-size result buffers and prevents +the previous iteration's intermediates from overlapping the next one. The same +one-million-class head then completed all 100 initialization iterations under the +same cap, with peak allocated/reserved bytes 15,383,099,904 / 15,407,775,744. +The uncapped observation was interrupted; it is not a completed baseline timing. +The cap governs the PyTorch allocator, not independent driver-residency telemetry. + +Exact CPU and CUDA tests compare the old and new updates for tall, wide and square +weights over 100 iterations. A CUDA regression also bounds temporary allocations +on a 10k-by-1280 matrix across several iterations, so retaining old gradient and +projection buffers would fail. Initializer parameters, iteration count, seeded +arithmetic and classifier behavior are preserved. + +Reports and the capped probe are retained under ignored `tmp-million-head/`. +This establishes normalized-head construction, not full EfficientNetV2 training, +optimizer-state capacity, checkpoint/reload, large-vocabulary quality or target +hardware performance. Those phases require their own evidence. + +A subsequent full EfficientNetV2-S flat-model probe used a 92% allocator cap +(about 14.72 GiB), leaving room for the backbone. Construction completed with +peak allocated/reserved bytes 15,471,561,728 / 15,485,370,368. It then failed during +`prepare_quantized_training`, when the existing rowwise quantizer requested a +full-size rounding buffer. No optimizer step ran, so the declared momentum-free +SGD configuration is not a successful training result. The retained report is +`tmp-million-head/full-flat/report.json`. Bounded quantization preparation is the +next memory bottleneck; bypassing the cap or changing initialization iterations +would not resolve it. + +Validation for the initializer change passed 381 CPU-default tests with 139 skips +and the existing EMA expected failure. Seven focused CPU/CUDA parity and +allocation cases passed with intentional GPU access. Static checks passed. diff --git a/mini_trainer/modeling/classifier.py b/mini_trainer/modeling/classifier.py index ebec53b..2e78fb7 100644 --- a/mini_trainer/modeling/classifier.py +++ b/mini_trainer/modeling/classifier.py @@ -45,7 +45,11 @@ def init_spherical_repulsion(cls, layer: nn.Module, iterations: int = 100, lr: f # is prohibitive for heads with tens of thousands of output classes. grad = w @ (w.t() @ w) if num_classes > w.size(1) else w @ w.t() @ w proj = (grad * w).sum(dim=1, keepdim=True) * w - w.sub_((lr / num_classes) * (grad - proj)) + # Reuse the gradient buffer without changing arithmetic order. + # Release both full-size temporaries before the next iteration. + grad.sub_(proj).mul_(lr / num_classes) + w.sub_(grad) + del grad, proj w.div_(w.norm(dim=1, keepdim=True).clamp(min=1e-9)) return layer diff --git a/tests/test_classifier_shapes.py b/tests/test_classifier_shapes.py index 4b0e08c..8d24757 100644 --- a/tests/test_classifier_shapes.py +++ b/tests/test_classifier_shapes.py @@ -72,3 +72,48 @@ def __torch_dispatch__(self, func, types, args=(), kwargs=None): head = Classifier(in_features=8, out_features=10000, hidden=True, normalized=True) assert torch.isfinite(head.linear.weight).all() torch.testing.assert_close(head.linear.weight.norm(dim=1), torch.ones(10000)) + + +@pytest.mark.parametrize("shape", [(8, 32), (32, 8), (32, 32)]) +@pytest.mark.parametrize("device", ["cpu", "cuda"]) +def test_spherical_initialization_buffer_reuse_preserves_values(shape, device): + import os + + from mini_trainer.modeling import Classifier + + if device == "cuda": + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for initializer CUDA parity") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + layer = torch.nn.Linear(shape[1], shape[0], bias=False, device=device) + torch.manual_seed(73) + expected = torch.empty_like(layer.weight).normal_() + for _ in range(100): + expected.div_(expected.norm(dim=1, keepdim=True).clamp(min=1e-9)) + gradient = expected @ (expected.t() @ expected) if shape[0] > shape[1] else expected @ expected.t() @ expected + projection = (gradient * expected).sum(dim=1, keepdim=True) * expected + expected.sub_((0.5 / shape[0]) * (gradient - projection)) + expected.div_(expected.norm(dim=1, keepdim=True).clamp(min=1e-9)) + torch.manual_seed(73) + Classifier.init_spherical_repulsion(layer) + torch.testing.assert_close(layer.weight, expected, rtol=0, atol=0) + + +def test_spherical_initialization_bounds_cuda_temporaries(): + import os + + from mini_trainer.modeling import Classifier + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for initializer allocation validation") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + layer = torch.nn.Linear(1280, 10000, bias=False, device="cuda") + torch.cuda.reset_peak_memory_stats() + before = torch.cuda.memory_allocated() + Classifier.init_spherical_repulsion(layer, iterations=3) + extra = torch.cuda.max_memory_allocated() - before + # Two full-size buffers plus the small Gram matrix fit under this bound; + # retaining the prior iteration's gradient/projection does not. + assert extra < 3 * layer.weight.numel() * layer.weight.element_size() From 6c7ebe06d97c3fdd6a7fef43c63de0659e32acf0 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 13:09:25 +0200 Subject: [PATCH 056/155] perf: bound INT8 preparation memory for large classifier heads --- docs/benchmarks.md | 42 ++++++++++++++ docs/quantized-training-validation.md | 5 +- docs/quantized-training.md | 4 ++ mini_trainer/modeling/_quantized_training.py | 20 +++++++ tests/test_quantized_preparation.py | 60 ++++++++++++++++++++ 5 files changed, 130 insertions(+), 1 deletion(-) create mode 100644 tests/test_quantized_preparation.py diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 5d13c5c..9eae711 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1558,3 +1558,45 @@ would not resolve it. Validation for the initializer change passed 381 CPU-default tests with 139 skips and the existing EMA expected failure. Seven focused CPU/CUDA parity and allocation cases passed with intentional GPU access. Static checks passed. + +### Bounded preparation and laptop-sized capacity checks + +Initial native INT8 conversion now applies the existing deterministic rowwise +quantizer in chunks, avoiding full-matrix rounding/clipping temporaries. The +source matrix and final INT8 storage remain resident during conversion; other +preparation validation allocations are unchanged. Exact regression checks cover +FP32, FP16 and BF16 on CPU and CUDA, including transposed weights, zero/tiny rows, +source preservation, RNG preservation and output layout. A CUDA check using a +10k-by-1280 matrix bounds additional conversion allocation below 96 MiB, which the +previous whole-matrix conversion exceeds. + +Following the local scale limit, the full-model repeat uses **100,000 classes**, +not one million. Both randomly initialized EfficientNetV2-S configurations use a +symmetric hidden layer, normalized output head, seed 42, batch two of synthetic +128-by-128 RGB images, FP16 AMP (initial scale 128), gradient clipping at 5, and +MuonAuxAdamW (learning rate .01, weight decay zero). The hierarchical case groups +100 consecutive leaves per parent. Each run performs three successful optimizer +updates and verifies finite gradients and changed INT8 codes in the two target +rows. Forward/zero-grad/backward ordering matches the trainer. + +The same RTX 3080 Ti Laptop GPU runs under a 92% PyTorch allocator cap. Peaks below +are allocated MiB for each phase, including live model/state storage, rather than +additional scratch space or independently measured physical VRAM residency. + +| Head | Construction | Preparation | Training update peak | Changed target codes | +| --- | ---: | ---: | ---: | ---: | +| flat | 1563.6 | 1436.6 | 2899.0 | 2535 | +| hierarchical | 1564.3 | 1437.4 | 2899.8 | 2538 | + +Both runs passed. These synthetic checks establish construction, preparation and +updates at this scale, not generalization, throughput improvements, checkpoint +capacity or target-hardware performance. No new million-class full-model run was +attempted; that remains work for a larger machine. The earlier million-class +preparation failure remains a recorded failure, not a subsequently verified pass. +Local probe and phase reports are retained under ignored `tmp-million-head/` as +`full_model_100k.py`, `full-flat-100k/report.json` and +`full-hierarchical-100k/report.json`. + +Validation passed static checks and 387 CPU-default tests (146 skips and the +existing EMA expected failure), plus 13 focused CPU/CUDA preparation cases and +91 CUDA-enabled quantized-training/model regressions. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 006eb65..5b270a9 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -31,7 +31,10 @@ priority for investigation. The subsequent [normalization backward kernel](benchmarks.md#bounded-int8-normalization-backward-storage) reduces measured 100k-class training-step peak allocation by 17.4% versus float for both heads. Timings and single-seed quality changes remain mixed. A -million-class training run remains unverified. +million-class training run remains unverified. Local full-model capacity checks +use up to 100k classes; million-class full-model validation is reserved for a +larger machine. Small arithmetic regressions can still test million-element +contractions without constructing a million-class EfficientNetV2 model. | Deployment target | Execution path to validate | Required measurements | | --- | --- | --- | diff --git a/docs/quantized-training.md b/docs/quantized-training.md index 1dca00e..22e87cf 100644 --- a/docs/quantized-training.md +++ b/docs/quantized-training.md @@ -37,6 +37,10 @@ The returned recipe lists quantized modules, skipped operations, remaining floating-point parameters and physical versus reference weight storage. Automatic selection covers ordinary `nn.Linear` modules, including their functional use by Classifier heads. It preserves shared weights when every owner is selected. +Initial conversion of large weights processes row chunks of at most 4,194,304 +elements (or one row when wider), limiting the quantizer's floating temporaries. +This preserves deterministic INT8 codes and row scales; it does not reduce the +storage needed for source weights, validation, gradients or optimizer states. Row-wise PyTorch weight normalization (`dim=0`) quantizes the direction parameter while retaining its scalar magnitude per output row in floating point. Effective normalized weights reuse the integer codes with new row scales; normalization diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index e158419..12cbaa2 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -15,6 +15,9 @@ from ._quantized_normalization import int8_weight_norm_backward from ._quantized_update import quantize_int8_rows, update_int8_rows_ +# Limit rowwise preparation temporaries without changing rounding or row scales. +_PREPARATION_CHUNK_ELEMENTS = 4 * 1024 * 1024 + # Tensor-subclass dispatch hides custom autograd bodies from AOT's ordinary # graph key. Include the backend and kernel implementations so changing hidden # backward math or operator decomposition cannot reuse an earlier graph. @@ -32,6 +35,23 @@ class TrainingWeight(Int8QuantizedTrainingLinearWeight): _is_quantized_training = True + @classmethod + def from_float(cls, tensor): + if tensor.numel() <= _PREPARATION_CHUNK_ELEMENTS or tensor.ndim != 2 or not tensor.shape[1]: + return super().from_float(tensor) + values = tensor.detach() + codes = torch.empty_like(values, dtype=torch.int8) + scales = torch.empty(values.shape[0], dtype=values.dtype, device=values.device) + rows = max(1, _PREPARATION_CHUNK_ELEMENTS // values.shape[1]) + for start in range(0, values.shape[0], rows): + end = min(start + rows, values.shape[0]) + row_codes, row_scales = quantize_int8_rowwise(values[start:end]) + codes[start:end].copy_(row_codes) + scales[start:end].copy_(row_scales) + result = cls(codes, scales) + result.requires_grad_(tensor.requires_grad) + return result + def dequantize(self): return _Dequantize.apply(self) diff --git a/tests/test_quantized_preparation.py b/tests/test_quantized_preparation.py new file mode 100644 index 0000000..3991fcc --- /dev/null +++ b/tests/test_quantized_preparation.py @@ -0,0 +1,60 @@ +"""Bounded native INT8 preparation preserves the existing deterministic mapping.""" + +import os + +import pytest +import torch + +pytest.importorskip("torchao") +pytest.importorskip("triton") + + +def check_device(device): + if device == "cuda": + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 for preparation CUDA checks") + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + + +@pytest.mark.parametrize("device", ["cpu", "cuda"]) +@pytest.mark.parametrize("dtype", [torch.float32, torch.float16, torch.bfloat16]) +@pytest.mark.parametrize("strided", [False, True]) +def test_chunked_preparation_is_exact(monkeypatch, device, dtype, strided): + check_device(device) + from mini_trainer.modeling import _quantized_training as backend + + torch.manual_seed(42) + values = torch.randn(17, 23, device=device, dtype=dtype) + if strided: + values = values.T + values[0].zero_() + values[1].mul_(1e-4) + values.requires_grad_(True) + before = values.detach().clone() + expected_codes, expected_scales = backend.quantize_int8_rowwise(values.detach()) + monkeypatch.setattr(backend, "_PREPARATION_CHUNK_ELEMENTS", 64) + rng = torch.cuda.get_rng_state() if device == "cuda" else torch.get_rng_state() + result = backend.TrainingWeight.from_float(values) + torch.testing.assert_close(result.int_data, expected_codes, rtol=0, atol=0) + torch.testing.assert_close(result.scale, expected_scales, rtol=0, atol=0) + torch.testing.assert_close(values, before, rtol=0, atol=0) + torch.testing.assert_close(torch.cuda.get_rng_state() if device == "cuda" else torch.get_rng_state(), rng, rtol=0, atol=0) + assert result.requires_grad + assert result.dtype == values.dtype and result.device == values.device + assert result.int_data.stride() == expected_codes.stride() + assert result.int_data.data_ptr() != values.data_ptr() + + +def test_preparation_cuda_temporary_bound(): + check_device("cuda") + from mini_trainer.modeling._quantized_training import TrainingWeight + + values = torch.randn(10000, 1280, device="cuda") + torch.cuda.reset_peak_memory_stats() + before = torch.cuda.memory_allocated() + result = TrainingWeight.from_float(values) + extra = torch.cuda.max_memory_allocated() - before + assert result.shape == values.shape + # The previous whole-matrix round/clip/cast sequence exceeds this bound. + assert extra < 96 * 1024 * 1024 From 7cd1fed9ee20e868bc2c52976144751c99d31cc1 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 13:19:42 +0200 Subject: [PATCH 057/155] test: gate ONNX operator placement and audit CUDA exports --- dev/benchmarks/README.md | 18 +++++- dev/benchmarks/onnx_inference.py | 19 ++++++- docs/benchmarks.md | 82 +++++++++++++++++++++++++++ docs/onnx.md | 13 +++-- docs/quantized-training-validation.md | 9 ++- docs/roadmap.md | 5 +- tests/test_benchmark_onnx.py | 28 ++++++++- 7 files changed, 162 insertions(+), 12 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index a0279d1..f98c2c5 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -613,7 +613,23 @@ that executes no operations on the requested provider fails instead of reporting a CPU run as a GPU measurement. Partial CPU execution is reported, not prohibited: shape and other auxiliary operations may legitimately use CPU. Inspect the profile for the expensive operations. Integer weights or QDQ nodes alone are not proof of -integer execution. Failures after output creation retain a failed `report.json`; +integer execution. Use repeatable `--require-provider-op` arguments to require +particular **profiled runtime operation types** on the requested provider. Every +required type must occur in each supplied model and every occurrence must execute +on that provider; missing types and partial fallback fail before timing. Inspect +the profile first: graph optimizations can change operation names or fuse nodes. +Auxiliary CPU operations remain allowed when they are not among the requirements. + +For example, add `--require-provider-op Conv --require-provider-op MatMulInteger` +to a native QT model invocation to require both the backbone and integer head on +CUDA. This currently **fails** on the tested ONNX Runtime 1.29.0 CUDA build: its +native integer head executes on CPU. The calibrated QDQ artifact instead runs +floating Conv/Gemm on CUDA, so requiring integer operation types also fails. +See the [measured placement results](../../docs/benchmarks.md#onnx-cuda-provider-placement). +A successful default measurement means some work ran on the requested provider; +it is not an integer-kernel or quality acceptance gate. + +Failures after output creation retain a failed `report.json`; input/environment preflight failures leave no result directory. Timing covers warm `Session.run` with CPU NumPy inputs and outputs, including diff --git a/dev/benchmarks/onnx_inference.py b/dev/benchmarks/onnx_inference.py index 416ea1c..2367317 100644 --- a/dev/benchmarks/onnx_inference.py +++ b/dev/benchmarks/onnx_inference.py @@ -35,7 +35,15 @@ def visit(message): return [{"path": str(p), "sha256": file_hash(p), "bytes": p.stat().st_size} for p in sorted(paths)] -def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provider_options=None): +def require_operations(counts, provider, operations): + """Require observed runtime operations to execute entirely on the chosen EP.""" + for operation in operations: + placement = {ep: count for (op, ep), count in counts.items() if op == operation and count} + if set(placement) != {provider}: + raise RuntimeError(f"Required operation {operation} must execute exclusively on {provider}; observed: {placement}") + + +def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provider_options=None, required_ops=()): import onnx import onnxruntime as ort @@ -60,6 +68,7 @@ def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provi "runner_sha256": file_hash(__file__), "provider": provider, "provider_options": provider_options or {}, + "required_ops": list(required_ops), "available_providers": ort.get_available_providers(), "threads": threads, "warmup": warmup, @@ -108,6 +117,7 @@ def options(): record["profile"] = str(profile) if not any(ep == provider for _, ep in counts): raise RuntimeError(f"No profiled operation executed on requested provider {provider}") + require_operations(counts, provider, required_ops) if any(not np.isfinite(array).all() for array in predictions): raise ValueError(f"Nonfinite predictions from {path}") record["outputs"] = [node.name for node in session.get_outputs()] @@ -144,6 +154,13 @@ def main(): parser.add_argument("--output", type=Path, required=True) parser.add_argument("--provider", required=True) parser.add_argument("--provider-options", type=json.loads, default={}) + parser.add_argument( + "--require-provider-op", + dest="required_ops", + action="append", + default=[], + help="Require this profiled operation type to execute entirely on the requested provider (repeatable)", + ) parser.add_argument("--threads", type=int, default=1) parser.add_argument("--warmup", type=int, default=3) parser.add_argument("--repeats", type=int, default=11) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 9eae711..c562dd2 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1600,3 +1600,85 @@ Local probe and phase reports are retained under ignored `tmp-million-head/` as Validation passed static checks and 387 CPU-default tests (146 skips and the existing EMA expected failure), plus 13 focused CPU/CUDA preparation cases and 91 CUDA-enabled quantized-training/model regressions. + + +### ONNX CUDA provider placement + +A local follow-up on 2026-09-09 used an isolated ONNX Runtime GPU 1.29.0 wheel, +Python 3.13.7, ONNX 1.22.0 and NumPy 2.4.6 on the RTX 3080 Ti Laptop GPU. +The wheel links CUDA 13 libraries; the process used the existing CUDA 13/cuDNN 9 +library directories without replacing the working PyTorch or CPU ORT packages. +`use_tf32=0` was explicit. Runtime dependencies must match the actual installed +wheel; see the [ORT CUDA provider guide](https://onnxruntime.ai/docs/execution-providers/CUDA-ExecutionProvider.html). + +The portable runner profiled the retained EfficientNetV2-S exports using the same +eight preprocessed Blair validation images, batch eight, size 128. **Selecting +CUDA did not produce integer GPU execution for either quantized recipe.** + +| Export | Main CUDA operations | Main CPU operations | Profiled host-to-device / device-to-host copies | +| --- | --- | --- | ---: | +| Floating flat | 170 Conv, 2 Gemm | Auxiliary operations | 1 / 1 | +| Native QT flat | 170 Conv | 2 MatMulInteger | 3 / 3 | +| Native QT hierarchical | 170 Conv | 2 MatMulInteger | 3 / 3 | +| Calibrated unsigned Percentile flat QDQ | 170 floating Conv, 2 floating Gemm | 171 DequantizeLinear | 172 / 1 | + +The QDQ CUDA run additionally executed 340 QuantizeLinear and 650 DequantizeLinear +operations on GPU. This differs from its earlier CPU run, which fused integer +QLinearConv/QGemm operations. These counts describe the observed optimized runtime +graph, not a universal statement about every CUDA/TensorRT version or quantized +representation. The native head's CPU fallback is particularly relevant to the +large-class deployment objective, even though this quality model has only 25 leaves. +Single-process timings are retained in the reports; these placement probes do not +establish speedups on the intended target machines. + +An optional `--require-provider-op` check now requires each named runtime operation +type to occur and execute entirely on the requested provider. Requiring Conv and +MatMulInteger rejected the native CUDA run with its two CPU MatMulInteger nodes, +retaining a failed report and profile before timing. Missing/fused-away operation +types also fail; inspect actual profiles when selecting requirements. The default +runner still permits partial CPU execution and records it. + +On the eight images, float and both native heads passed CPU/CUDA score comparison +at rtol=1e-4, atol=1e-5 (maximum errors 1.24e-5, 1.34e-5 and 8.59e-6 respectively). +The calibrated flat QDQ export failed: maximum score error 0.45745 and one top-1 +change versus its own CPU output. That recipe needs a full CUDA-specific quality +study; its CPU results cannot serve as GPU quality evidence. + +The native flat/hierarchical exports were then evaluated on all **912** Blair +validation images, in manifest order, batch eight, against retained matched CPU +ORT and full-FP32 PyTorch references. Manifest/checkpoint hashes and class mappings +were checked; the baseline score file hashes are retained. All predictions stayed +the same. `mini_metrics` revision `70cc69adc05362863439277048e06386c1f885e1`, with +explicit MacroF1 selection criterion and the earlier threshold-zero, no-abstention +policy, gave exactly the same five metrics for all three execution paths: + +| Output | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat | 0.769112 | 0.758984 | 0.796193 | 1.000000 | 0.810716 | +| Hierarchical leaf | 0.730698 | 0.720275 | 0.799803 | 1.000000 | 0.787638 | +| Hierarchical parent | 0.867051 | 0.844461 | 0.913180 | 1.000000 | 0.857871 | + +Strict score parity remains false: versus full-FP32 PyTorch, 6 flat, 18 leaf and +19 parent images exceed tolerance, with maximum errors 0.014928, 0.012346 and +0.010621. Confidence-threshold equivalence and production acceptance remain open. +The initial probe completed flat evaluation then hit a cleanup NameError; after +fixing that probe, hierarchical evaluation completed while preserving the flat +result. No training or performance measurements were inferred from that retry. + +Profiles, hashes and output arrays are retained under ignored +`tmp-onnx-cuda-placement-flat/`, `tmp-onnx-cuda-placement-recipes/`, +`tmp-onnx-cuda-placement-cpu-reference/` and `tmp-onnx-cuda-required-int8/`. +Full-validation scores, metric CSVs and reports are under +`tmp-native-int8-cuda-validation/`; its probe is +`tmp-native-int8-validation/run_cuda.py`. + +Next, verify an integer GPU execution path with an appropriate provider/kernel +and repeat the quality protocol before making a GPU quantization recommendation. +The native dynamic quantizer and calibrated QDQ recipe have distinct numerical +contracts. HPC training performance, intended desktop/Spark GPU performance and +ARM CPU inference still require their actual hardware; laptop execution does not +close those requirements. + +The placement-check increment passed static checks and 390 CPU-default tests, +with 146 skips and the existing EMA expected failure. The real CUDA negative +probe also rejected native-head CPU fallback and retained its diagnostic report. diff --git a/docs/onnx.md b/docs/onnx.md index b01255c..64464f3 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -102,7 +102,10 @@ The actual EfficientNetV2-S backbone with symmetric normalized flat/hierarchical heads is also covered by offline dynamic-batch export tests. Trained Blair checkpoints passed ONNX Runtime CPU parity on real images; see the [deployment experiment](benchmarks.md#efficientnetv2-onnx-cpu-export-and-inference-quantization). -GPU providers, ARM execution and arbitrary spatial dimensions remain unvalidated. +Local [CUDA placement checks](benchmarks.md#onnx-cuda-provider-placement) expose +CPU fallback for native integer heads and floating execution for calibrated +convolutions. Target GPU hardware, ARM execution and arbitrary spatial dimensions +remain unvalidated. An initial signed MinMax INT8 recipe lost substantial accuracy and retained floating convolutions. A follow-up unsigned Percentile recipe executed all convolutions as QLinearConv and roughly halved warm local CPU inference latency, @@ -155,9 +158,11 @@ Tests cover normalized symmetric flat/hierarchical heads, EfficientNetV2-S, active-class filtering, dynamic batches, checkpoint CLI loading and numerical edge cases. Trained Blair checkpoints also passed checks on eight real validation images at batches 1, 2, 4 and 8. An exported graph still needs runtime profiling -and quality evaluation on the intended provider. CUDA/ARM ONNX execution, -million-class export capacity and production performance of this native path -remain unverified. On the full Blair validation split, top-1 predictions and the +and quality evaluation on the intended provider. Local CUDA-provider execution +retains CPU MatMulInteger operations; it does not establish integer GPU execution. +Target GPU/ARM deployment, million-class export capacity and production performance +of this native path remain unverified. On the full Blair validation split, +top-1 predictions and the requested macro metrics matched the full-FP32 CUDA reference, but some image scores exceeded the strict export tolerance; see the [full-validation results](benchmarks.md#native-onnx-full-validation-quality-and-numerical-limits). diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 5b270a9..33e80ac 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -47,8 +47,13 @@ measurements remain useful for debugging but cannot establish gains on these systems. Do not assume a single quantized artifact or kernel recipe works across CUDA PyTorch, ONNX GPU and ONNX ARM CPU. Native QT now has an opt-in [integer-forward ONNX export](onnx.md#native-int8-training-checkpoints), verified -against a full-FP32 CUDA reference on the local CPU provider. This does not -establish ONNX GPU/ARM execution or performance. ONNX Runtime training/fine-tuning +against a full-FP32 CUDA reference on the local CPU provider. Subsequent +[local CUDA-provider checks](benchmarks.md#onnx-cuda-provider-placement) retain +CPU execution for native integer heads; full Blair validation preserves the five +requested metrics, while strict score parity still fails. The calibrated QDQ +recipe uses floating Conv/Gemm on CUDA and fails the small CPU/CUDA parity probe. +Integer GPU execution, target GPU performance and ARM execution remain open. +ONNX Runtime training/fine-tuning would be a separate integration; an inference export does not provide it. The shared dataset harness now selects backbone and head independently while diff --git a/docs/roadmap.md b/docs/roadmap.md index cb4b80b..ab35705 100644 --- a/docs/roadmap.md +++ b/docs/roadmap.md @@ -92,8 +92,9 @@ An initial native QT and shared-loader milestone is implemented for the document CUDA Linear regime. The [validation audit](quantized-training-validation.md) records three-seed memory/speed benefits, real loading comparisons, synthetic and hierarchical checkpoint/inference coverage, allocation-aware worker defaults, -and installed-package checks. The primary EfficientNetV2 configuration with symmetric -hidden layers and normalized flat/hierarchical heads remains unvalidated. HPC +and installed-package checks. Local EfficientNetV2-S runs now cover symmetric +hidden layers and normalized flat/hierarchical heads, including 100k-class +synthetic capacity checks and Blair quality measurements. HPC PyTorch training, local GPU training/ONNX inference and ARM ONNX edge inference are required deployment targets, not extensions of a completed objective. Broader operator coverage, cold-start performance and quality studies remain open. The chronological diff --git a/tests/test_benchmark_onnx.py b/tests/test_benchmark_onnx.py index d67dee5..90d4ed6 100644 --- a/tests/test_benchmark_onnx.py +++ b/tests/test_benchmark_onnx.py @@ -3,7 +3,7 @@ import numpy as np import pytest -from dev.benchmarks.onnx_inference import run +from dev.benchmarks.onnx_inference import require_operations, run onnx = pytest.importorskip("onnx") pytest.importorskip("onnxruntime") @@ -29,9 +29,10 @@ def model_and_inputs(tmp_path): def test_cpu_measurement_records_external_weights_execution_and_outputs(model_and_inputs, tmp_path): model, inputs = model_and_inputs output = tmp_path / "measurement" - report = run([model], inputs, output, "CPUExecutionProvider", warmup=1, repeats=2) + report = run([model], inputs, output, "CPUExecutionProvider", warmup=1, repeats=2, required_ops=["MatMul"]) assert report == json.loads((output / "report.json").read_text()) assert report["status"] == "passed" + assert report["required_ops"] == ["MatMul"] record = report["models"][0] assert len(record["files"]) == 2 assert len(record["seconds"]) == 2 @@ -50,6 +51,29 @@ def test_unavailable_provider_does_not_fall_back(model_and_inputs, tmp_path): assert not (tmp_path / "missing").exists() +def test_missing_required_operation_retains_failed_profile(model_and_inputs, tmp_path): + model, inputs = model_and_inputs + output = tmp_path / "missing-operation" + with pytest.raises(RuntimeError, match="Required operation MatMulInteger"): + run([model], inputs, output, "CPUExecutionProvider", required_ops=["MatMulInteger"]) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" + assert report["required_ops"] == ["MatMulInteger"] + assert report["models"][0]["execution"] + assert report["models"][0]["seconds"] == [] + + +@pytest.mark.parametrize("gpu_count", [0, 1]) +def test_required_operation_rejects_partial_or_complete_cpu_fallback(gpu_count): + counts = {("Conv", "CUDAExecutionProvider"): 170, ("MatMulInteger", "CPUExecutionProvider"): 2} + if gpu_count: + counts[("MatMulInteger", "CUDAExecutionProvider")] = gpu_count + with pytest.raises(RuntimeError, match="Required operation MatMulInteger"): + require_operations(counts, "CUDAExecutionProvider", ["Conv", "MatMulInteger"]) + # Auxiliary CPU work is allowed when the required operations remain on GPU. + require_operations(counts, "CUDAExecutionProvider", ["Conv"]) + + def test_input_contract_failure_is_retained(model_and_inputs, tmp_path): model, inputs = model_and_inputs np.savez(inputs, wrong=np.ones((2, 2), dtype=np.float32)) From a04966813363d289790df89375c9ce5bb7320a28 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 13:36:33 +0200 Subject: [PATCH 058/155] test: validate TensorRT INT8 inference and expose graph optimization --- dev/benchmarks/README.md | 37 +++++++++++ dev/benchmarks/onnx_inference.py | 8 ++- docs/benchmarks.md | 90 +++++++++++++++++++++++++++ docs/quantized-training-validation.md | 8 ++- tests/test_benchmark_onnx.py | 12 +++- 5 files changed, 150 insertions(+), 5 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index f98c2c5..cf13a20 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -629,6 +629,13 @@ See the [measured placement results](../../docs/benchmarks.md#onnx-cuda-provider A successful default measurement means some work ran on the requested provider; it is not an integer-kernel or quality acceptance gate. +`--optimization disable|basic|extended|all` selects and records ONNX Runtime's +graph optimization level for both the profiling and timing sessions (`all` is the +unchanged default). This matters for provider partitioning: the tested TensorRT +QDQ graph acquired unsupported INT32 bias dequantization under the default +rewrites. With `disable`, its flat model ran as one TensorRT partition without +CPU operations. This setting does not disable TensorRT's own engine optimization. + Failures after output creation retain a failed `report.json`; input/environment preflight failures leave no result directory. @@ -648,3 +655,33 @@ power/thermal configuration and concurrent workload alongside the report. The unsigned activation recipe measured on x86 is a candidate to test, not a universal GPU/ARM recipe. This inference runner does not validate HPC PyTorch quantized training, native QT checkpoint export or million-class capacity. + +### TensorRT calibration candidate + +The local candidate uses signed, symmetric INT8 activation/weight QDQ, per-channel +weights and floating biases. It derives symmetric ranges from the retained +Percentile 99.9 training-split calibration cache. With ONNX Runtime 1.29.0's +`quantize_static`, the relevant arguments are `quant_format=QuantFormat.QDQ`, +`activation_type=QuantType.QInt8`, `weight_type=QuantType.QInt8`, `per_channel=True` +and `extra_options={"ActivationSymmetric": True, "WeightSymmetric": True, +"QuantizeBias": False}`. Supply a reviewed calibration reader/cache and use +separate output artifacts. This changes the quantization recipe; it does not +preserve the native training export's dynamic quantizer. + +In a separately prepared compatible TensorRT/ONNX Runtime GPU environment: + +```bash +python -m dev.benchmarks.onnx_inference \ + --model signed-symmetric-qdq/model.onnx --inputs preprocessed-batch.npz \ + --output /tmp/trt-placement-1 --provider TensorrtExecutionProvider \ + --optimization disable \ + --provider-options '{"trt_max_workspace_size":"1073741824","trt_engine_cache_enable":true,"trt_engine_cache_path":"/tmp/trt-cache-1"}' +``` + +TensorRT profiles show fused partition names rather than individual Conv/Gemm +operations. A partition on GPU alone does not prove integer arithmetic: inspect +engine layer input/weight types and tactics using detailed TensorRT engine +inspection. Direct hierarchical engine construction requires registering the +standard TensorRT plugins (`init_libnvinfer_plugins`) for scatter reductions. +Do not copy device-specific engines between target machines. See the +[TensorRT evidence and quality limits](../../docs/benchmarks.md#tensorrt-int8-calibration-candidate). diff --git a/dev/benchmarks/onnx_inference.py b/dev/benchmarks/onnx_inference.py index 2367317..ef8331b 100644 --- a/dev/benchmarks/onnx_inference.py +++ b/dev/benchmarks/onnx_inference.py @@ -43,10 +43,13 @@ def require_operations(counts, provider, operations): raise RuntimeError(f"Required operation {operation} must execute exclusively on {provider}; observed: {placement}") -def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provider_options=None, required_ops=()): +def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provider_options=None, required_ops=(), optimization="all"): import onnx import onnxruntime as ort + levels = {"disable": "ORT_DISABLE_ALL", "basic": "ORT_ENABLE_BASIC", "extended": "ORT_ENABLE_EXTENDED", "all": "ORT_ENABLE_ALL"} + if optimization not in levels: + raise ValueError(f"Unknown graph optimization level: {optimization}") if min(threads, warmup, repeats) < 1: raise ValueError("threads, warmup and repeats must be positive") if not models or len(set(models)) != len(models): @@ -69,6 +72,7 @@ def run(models, inputs, output, provider, threads=1, warmup=3, repeats=11, provi "provider": provider, "provider_options": provider_options or {}, "required_ops": list(required_ops), + "optimization": optimization, "available_providers": ort.get_available_providers(), "threads": threads, "warmup": warmup, @@ -91,6 +95,7 @@ def options(): opts = ort.SessionOptions() opts.intra_op_num_threads = threads opts.inter_op_num_threads = 1 + opts.graph_optimization_level = getattr(ort.GraphOptimizationLevel, levels[optimization]) return opts try: @@ -154,6 +159,7 @@ def main(): parser.add_argument("--output", type=Path, required=True) parser.add_argument("--provider", required=True) parser.add_argument("--provider-options", type=json.loads, default={}) + parser.add_argument("--optimization", choices=["disable", "basic", "extended", "all"], default="all") parser.add_argument( "--require-provider-op", dest="required_ops", diff --git a/docs/benchmarks.md b/docs/benchmarks.md index c562dd2..183b50e 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1682,3 +1682,93 @@ close those requirements. The placement-check increment passed static checks and 390 CPU-default tests, with 146 skips and the existing EMA expected failure. The real CUDA negative probe also rejected native-head CPU fallback and retained its diagnostic report. + + +### TensorRT INT8 calibration candidate + +The next local experiment used isolated TensorRT 10.16.1.11 CUDA 13 packages with +ONNX Runtime GPU 1.29.0 and the same RTX 3080 Ti Laptop GPU. The working `.venv` +was not changed. TensorRT is the documented +[ORT GPU quantization path](https://onnxruntime.ai/docs/performance/model-optimizations/quantization.html), +but its [standard quantizer](https://docs.nvidia.com/deeplearning/tensorrt/latest/_static/operators/Quantize.html) +requires constant scales and zero zero-points. Native QT uses dynamic row scales. + +Direct parsing rejected the native export at an unsigned intermediate Cast. A +small signed MatMulInteger probe was also rejected as unsupported, so changing +only activation signedness does not establish a native TensorRT export path. +Those failures are retained, not replaced by a calibrated graph with a different +numerical contract. + +The separate calibration candidate uses signed symmetric INT8 activation/weight +QDQ, per-channel weights and **floating biases**, applied to Conv/Gemm/MatMul. +Its symmetric ranges derive from the previously recorded asymmetric Percentile +99.9 ranges on 128 training images; it is not a newly fitted symmetric histogram. +Default INT32 bias dequantization was rejected by TensorRT. Disabling bias +quantization made the flat graph parse directly; registering TensorRT's standard +plugins also enabled the hierarchical scatter reductions. + +ONNX Runtime's default graph rewrites reintroduced INT32 bias dequantization, +leaving 171 DequantizeLinear operations on CPU and 171 host-to-device copy nodes +around a TensorRT partition. Repeating the same flat graph with graph optimization +disabled produced **one TensorRT partition and no profiled CPU operations**. +The runner now exposes `--optimization disable|basic|extended|all`, records it, +and applies it to both profiling and timing sessions. The default remains `all`. +This does not disable TensorRT engine optimization. The initial provider probe +also exposed a configuration error: this build requires True/False values for +TensorRT boolean options, rather than the string "1"; its failed report is retained. + +Directly built engines used a 1 GiB workspace limit, builder optimization level 1, +TF32 disabled, detailed layer inspection, and batch profiles 1–8 at image size 128. +Both normalized symmetric EfficientNetV2-S heads built and executed. Inspection +shows all **170 convolutions with INT8 inputs and weights**, and **both head GEMMs +with INT8 tactics**. Floating bias and auxiliary operations remain. The +hierarchical engine uses three standard scatter plugin layers. These are actual +integer GPU computation checks, not conclusions from QDQ nodes alone. The flat +ORT TensorRT output matched the direct engine exactly on the eight fixed images. +That is not a full ORT-versus-direct-engine equivalence claim. + +The direct engines were evaluated on all **912** Blair validation images in +manifest order, batch eight, alongside full-FP32 PyTorch and CPU ORT executions of +the same signed QDQ graphs. Checkpoint/export hashes and class order were checked; +graph/external-weight/calibration/engine hashes are retained. Metrics use local +`mini_metrics` revision `70cc69adc05362863439277048e06386c1f885e1`, explicit MacroF1 +selection criterion, threshold zero and no abstention or threshold tuning. + +| Output / execution | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat / Float | 0.762073 | 0.754415 | 0.789570 | 1.000000 | 0.814794 | +| Flat / TensorRT INT8 | 0.755863 | 0.744234 | 0.790002 | 1.000000 | 0.804222 | +| Hierarchical leaf / Float | 0.709222 | 0.695729 | 0.798771 | 1.000000 | 0.782579 | +| Hierarchical leaf / TensorRT INT8 | 0.710140 | 0.687581 | 0.775226 | 1.000000 | 0.769531 | +| Hierarchical parent / Float | 0.859365 | 0.827595 | 0.920382 | 1.000000 | 0.851394 | +| Hierarchical parent / TensorRT INT8 | 0.847096 | 0.805315 | 0.921725 | 1.000000 | 0.828815 | + +Relative to float, Macro-F1 changes are **−0.62 percentage points flat, +0.09 leaf, +and −1.23 parent**. Recall and Theil's U decrease for all three outputs; the small +leaf F1 increase is not evidence of generally improved quality. TensorRT versus +CPU QDQ changes 13 flat, 10 leaf and 7 parent predictions. Maximum score differences +are 0.46753, 0.52708 and 0.52582, and strict parity fails. This reinforces the need +for provider-specific quality evaluation. No production degradation tolerance has +been fixed. The user accepts degradation of a few percentage points when paired +with a significant inference speed/cost or memory improvement. These quality +results alone do not establish that joint trade-off; matched performance evidence +against a practical floating deployment baseline is required. + +This milestone establishes a **separate calibrated INT8 GPU inference candidate** +for both representative heads, not native QT export equivalence. It does not +establish target-machine performance, optimal TensorRT build settings, training +benefits, confidence-threshold parity or large-class deployment capacity. The +probes used a default CUDA stream and ran alongside CPU correctness checks; their +recorded timings are not benchmark evidence. Production measurements require +fresh, uncontended runs with the final stream, build and cache configuration. + +Ignored `tmp-trt-probe/` retains parser failures, the candidate manifest, graphs, +ORT reports, engines, detailed `direct-{flat,hierarchical}/layers.json`, the +`build_inspect.py` and `full_validation.py` probes, and the full-validation +scores/metric CSVs/report. Rebuild engines on each target device. The reproducible +runner options and recipe parameters are in the +[developer guide](../dev/benchmarks/README.md#tensorrt-calibration-candidate). + +Static checks and 393 CPU-default tests passed, with 146 skips and the known EMA +expected failure. Runtime regressions verify that the selected ORT optimization +level changes the profiled graph while preserving fixture outputs. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 33e80ac..20c0982 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -52,7 +52,13 @@ against a full-FP32 CUDA reference on the local CPU provider. Subsequent CPU execution for native integer heads; full Blair validation preserves the five requested metrics, while strict score parity still fails. The calibrated QDQ recipe uses floating Conv/Gemm on CUDA and fails the small CPU/CUDA parity probe. -Integer GPU execution, target GPU performance and ARM execution remain open. +The subsequent [TensorRT candidate](benchmarks.md#tensorrt-int8-calibration-candidate) +executes calibrated INT8 convolutions and head GEMMs for both representative heads. +It has a separate numerical contract and mixed metric changes. Degradation of a +few percentage points is acceptable to the user when paired with a significant +measured inference speed/cost or memory benefit; quality alone does not establish +acceptance. Native-export integer GPU execution, target GPU performance and ARM +execution remain open. ONNX Runtime training/fine-tuning would be a separate integration; an inference export does not provide it. diff --git a/tests/test_benchmark_onnx.py b/tests/test_benchmark_onnx.py index 90d4ed6..1a649b5 100644 --- a/tests/test_benchmark_onnx.py +++ b/tests/test_benchmark_onnx.py @@ -12,7 +12,10 @@ @pytest.fixture def model_and_inputs(tmp_path): graph = onnx.helper.make_graph( - [onnx.helper.make_node("MatMul", ["images", "weight"], ["scores"])], + [ + onnx.helper.make_node("MatMul", ["images", "weight"], ["raw_scores"]), + onnx.helper.make_node("Identity", ["raw_scores"], ["scores"]), + ], "linear", [onnx.helper.make_tensor_value_info("images", onnx.TensorProto.FLOAT, ["batch", 2])], [onnx.helper.make_tensor_value_info("scores", onnx.TensorProto.FLOAT, ["batch", 2])], @@ -26,18 +29,21 @@ def model_and_inputs(tmp_path): return path, inputs -def test_cpu_measurement_records_external_weights_execution_and_outputs(model_and_inputs, tmp_path): +@pytest.mark.parametrize("optimization", ["disable", "basic", "extended", "all"]) +def test_cpu_measurement_records_external_weights_execution_and_outputs(model_and_inputs, tmp_path, optimization): model, inputs = model_and_inputs output = tmp_path / "measurement" - report = run([model], inputs, output, "CPUExecutionProvider", warmup=1, repeats=2, required_ops=["MatMul"]) + report = run([model], inputs, output, "CPUExecutionProvider", warmup=1, repeats=2, required_ops=["MatMul"], optimization=optimization) assert report == json.loads((output / "report.json").read_text()) assert report["status"] == "passed" assert report["required_ops"] == ["MatMul"] + assert report["optimization"] == optimization record = report["models"][0] assert len(record["files"]) == 2 assert len(record["seconds"]) == 2 assert record["median_seconds"] > 0 assert any(e["op"] == "MatMul" and e["provider"] == "CPUExecutionProvider" for e in record["execution"]) + assert any(e["op"] == "Identity" for e in record["execution"]) == (optimization == "disable") with np.load(output / "model-0-outputs.npz") as actual, np.load(inputs) as expected: np.testing.assert_array_equal(actual["scores"], expected["images"]) with pytest.raises(FileExistsError): From 0f3e2a9d861910648ddbc84688a7fe69b70a1c4c Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 13:45:05 +0200 Subject: [PATCH 059/155] docs: measure TensorRT INT8 trade-offs against FP16 --- docs/benchmarks.md | 72 ++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 72 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 183b50e..0be6628 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1772,3 +1772,75 @@ runner options and recipe parameters are in the Static checks and 393 CPU-default tests passed, with 146 skips and the known EMA expected failure. Runtime regressions verify that the selected ORT optimization level changes the profiled graph while preserving fixture outputs. + + +### TensorRT INT8 versus FP16: initial paired trade-off + +The user's acceptance criterion is conditional: a few percentage points of metric +loss are tolerable with a substantial inference speed/cost or memory benefit. +The comparator therefore includes a TensorRT FP16 deployment baseline, not just +the full-FP32 validation reference. Both heads were built from the same floating +checkpoints and preprocessing contracts. FP16 was enabled for the baseline; +TF32 remained disabled. Workspace (1 GiB), builder optimization level (1), spatial +size (128) and batch profiles (1–8) match the INT8 candidate. These build settings +are a controlled initial comparison, not a tuning result. + +FP16 was checked on all 912 Blair validation images using the same five +`mini_metrics` metrics and fixed policy: + +| FP16 output | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat | 0.759822 | 0.751879 | 0.787498 | 1.000000 | 0.813357 | +| Hierarchical leaf | 0.709222 | 0.695729 | 0.798771 | 1.000000 | 0.782579 | +| Hierarchical parent | 0.859365 | 0.827595 | 0.920382 | 1.000000 | 0.851394 | + +FP16 changed three flat predictions relative to FP32 and no hierarchical +predictions. Strict score parity fails for both heads. Compared with this FP16 +baseline, INT8 Macro-F1 changes by approximately −0.40 percentage points flat, ++0.09 leaf and −1.23 parent. These remain single-checkpoint quality measurements. + +Timing then used three fresh processes per head with no concurrent validation or +benchmark jobs. Each process held both engines, used a non-default CUDA stream, +10 warmups and 31 measured repetitions per model/batch, alternated FP16/INT8 order +within repetitions, and reversed ordering in the middle process. Preallocated +CPU/GPU input and output buffers were reused. Host timing includes copies and +synchronized engine execution, excluding decoding, preprocessing, allocation, +engine loading/building and shape changes. + +The ranges below are the three process medians in milliseconds. Ratios are +computed **within each paired process**, then summarized by their median; lower +than one would favor INT8. Variation is retained rather than selecting one run. + +| Head | Batch | FP16 host ms range | INT8 host ms range | Median paired INT8 / FP16 | +| --- | ---: | ---: | ---: | ---: | +| flat | 1 | 3.444–3.887 | 4.288–4.384 | 1.233 | +| flat | 8 | 3.371–4.528 | 4.230–5.624 | 1.242 | +| hierarchical | 1 | 3.440–4.008 | 3.899–5.207 | 1.170 | +| hierarchical | 8 | 3.609–3.823 | 4.134–4.559 | 1.152 | + +There is **no observed host-latency improvement** in these local small-batch +probes. CUDA-event durations frequently exceeded their enclosing synchronized +host durations; those counters are retained for diagnosis but excluded from +GPU-only timing conclusions. No cause has been established for that discrepancy. +The host measurements are limited observations from this WSL laptop and build +configuration, not evidence about A100/B300, Spark/desktop GPU, or ARM performance. + +Serialized engines shrink from 46,089,148 to 26,406,668 bytes (flat) and 46,177,388 +to 26,519,700 bytes (hierarchical), approximately **43% smaller**. TensorRT's +reported execution-context memory for the profile changes from 5,669,888 to +5,586,944 bytes for both heads, only **1.46% lower**. These are engine-file size +and a runtime-reported context requirement, not total GPU residency or peak +process memory. They do not establish a substantial total-runtime-memory benefit. + +Thus quality is in the potentially tolerable range, but the desired speed or +runtime-memory trade-off is **not yet demonstrated** against FP16. The storage +benefit is real. Larger batches and large-class heads need separate paired probes, +and the requested target machines remain necessary for deployment conclusions. +The tested real-data heads have 25 leaves; they do not model 100k-class head costs. + +Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, +`full-validation-fp16/`, `paired-{flat,hierarchical}-{1,2,3}/`, +`paired-summary.json`, and the `build_fp16.py`, `full_validation_fp16.py`, and +`paired_inference.py` probes. Reports retain engine/source hashes and all timing +samples. This follow-up changes documentation only; all six benchmark processes +and both full-validation cases completed with finite outputs. From b362bd318755a526b6e7d4379613ea95d221dcc0 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 13:47:42 +0200 Subject: [PATCH 060/155] docs: assess quantization status and completion requirements --- docs/quantization-status.md | 107 ++++++++++++++++++++++++++ docs/quantized-training-validation.md | 3 + 2 files changed, 110 insertions(+) create mode 100644 docs/quantization-status.md diff --git a/docs/quantization-status.md b/docs/quantization-status.md new file mode 100644 index 0000000..54454f8 --- /dev/null +++ b/docs/quantization-status.md @@ -0,0 +1,107 @@ +# Quantization goal status — 2026-09-09 + +The goal remains **incomplete**. There are working native INT8 training and +calibrated INT8 inference paths, but the required benefit on the intended +deployment regimes has not been established. The latest completed comparison is +recorded in [the benchmark report](benchmarks.md#tensorrt-int8-versus-fp16-initial-paired-trade-off). + +The acceptance principle is a joint trade-off: the user tolerates metric losses +of a few percentage points when accompanied by a substantial inference speed/cost +or memory benefit. This does not make a quality-only result, smaller engine file, +or integer operator count sufficient evidence of production readiness. + +## Current evidence + +| Workstream | Verified locally | What remains unproven | +| --- | --- | --- | +| Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | +| Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained production preparation/inspection workflow | +| CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | +| Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | + +Native training currently leaves convolutions, gradients and optimizer states +floating. Calibrated TensorRT inference is a separate recipe derived from floating +checkpoints; it does not prove that the native dynamic training quantizer exports +unchanged into integer GPU execution. DDP/FSDP remain unsupported for native QT. +EMA repair is explicitly deferred and is not a prerequisite for this goal. + +The latest TensorRT comparison uses matched builder settings and three fresh +paired processes per head. At batches 1 and 8, INT8 did not beat FP16 in local +host latency. Engines are approximately 43% smaller, while reported execution +context memory is only 1.46% lower; total runtime memory has not been measured +reliably. CUDA-event durations conflicted with enclosing host timing and are +excluded from device-only performance conclusions. + +Against FP16, INT8 Macro-F1 changes are approximately −0.40 percentage points for +the flat classifier, +0.09 for hierarchical leaves and −1.23 for parents. Recall +and Theil's U also require consideration; a small leaf F1 improvement is not a +general quality improvement. All five requested metrics are recorded in the +benchmark report. These are single-checkpoint validation results, not a multi-seed +production acceptance study. + +## Remaining work, in practical order + +1. **Make the latest experiments reproducible from a clean checkout.** Promote the + useful preparation, engine inspection, paired inference and metric-evaluation + probes into maintained developer commands with explicit optional environments. + Preserve calibration records, class/preprocessing contracts, hashes, failures + and raw timing samples. Resolve or exclude inconsistent timing sources. The + current detailed probes and engines are retained locally under ignored `tmp-*` + directories; documentation alone does not make them a continuous pipeline. + +2. **Find and verify the beneficial workload regimes.** Pair INT8 with practical + FP16/BF16 baselines while varying batch size, resolution and head size in a + controlled way. Include 10k/100k-class synthetic heads alongside real-data + quality checks; keep full million-class models off this laptop. Measure cold + setup, steady-state throughput/latency, transfers, loading and runtime memory + separately. Optimize the dominant measured costs rather than assuming integer + arithmetic is faster. Include full and frozen-backbone training/fine-tuning. + +3. **Settle the trained-checkpoint-to-deployment contract.** Either provide a GPU + implementation preserving native dynamic quantization, or explicitly convert + and calibrate a native QT checkpoint into a distinct deployment artifact and + validate the resulting quality change. The existing float-checkpoint TensorRT + candidate does not close that loop. Keep model configurations generic and + verify normalized/masked/hierarchical output semantics and class ordering. + +4. **Close the training and target-hardware evidence gaps.** Run the paired + training profiles on the intended A40/A100/B300-class GPU and AMD EPYC systems, + including actual shared-storage/loading conditions and CPU allocation limits. + Validate convergence, time to useful quality, memory and resume behavior. + Broaden quantized operator coverage where required to achieve the real-model + efficiency objective. If the target workload requires distributed training, + implement and validate distributed QT explicitly; single-GPU results cannot + establish it. Repeat training/fine-tuning and ONNX inference on the intended + Spark/RTX desktop separately. + +5. **Validate the edge deployment on ARM.** Run the same versioned ONNX candidates + on Raspberry Pi or the intended comparable device. Measure batch-one latency, + sustained throughput and process memory with explicit threads and preprocessing, + and reevaluate all five metrics. Verify the actual runtime kernels and installed + dependency set; neither x86 CPU results nor CUDA results establish ARM support. + +6. **Turn the accepted trade-offs into continuous release evidence.** Select + concrete quality and benefit thresholds for each supported deployment profile + once paired measurements exist. Automate synthetic correctness checks on CPU, + representative quality/performance checks on appropriate runners, and retained + visible reports with immutable inputs and baselines. Document supported recipes, + limitations, installation, calibration, export and hardware-specific engine + rebuilding. Finish compatibility review without weakening the ordinary float + path or implying support for untested configurations. + +## Completion standard + +I would consider the goal finished only when the intended HPC training, local +training/fine-tuning and GPU inference, and ARM inference profiles each have a +reproducible supported path, measured worthwhile benefit against a practical +baseline, acceptable paired quality using Macro-F1/Recall/Precision, Coverage and +Theil's U, and repeatable deployment/regression reporting. Each claim must be +backed by the corresponding hardware and workload, including preprocessing, +class mappings and checkpoint behavior where applicable. + +There is useful local work remaining before target machines become available; +their absence does not prevent the next implementation steps. It does prevent +final performance certification of those targets. Birds/iNaturalist, unrelated +training-feature comparisons, additional dataset formats and EMA repair are not +new prerequisites for completing this quantization goal. diff --git a/docs/quantized-training-validation.md b/docs/quantized-training-validation.md index 20c0982..8acc4e5 100644 --- a/docs/quantized-training-validation.md +++ b/docs/quantized-training-validation.md @@ -6,6 +6,9 @@ and shared loading for float and quantized training/inference. The capability is opt-in; speedups are workload-dependent. Local validation was recorded on 2026-09-09 using Python 3.13.7, PyTorch 2.12.0/CUDA 13.0, TorchAO 0.17.0 and an RTX 3080 Ti Laptop GPU. +The [2026-09-09 goal status report](quantization-status.md) separates verified +capabilities from remaining work and defines the evidence required for completion. + ## Primary model and deployment targets The primary model is EfficientNetV2 with a symmetric hidden layer (`hidden=True`) From 456d9ab1de9297eab7b0406a7e5ca6e3bf52c821 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 14:01:55 +0200 Subject: [PATCH 061/155] feat: add reproducible TensorRT build and inspection command --- dev/benchmarks/README.md | 65 +++++++++ dev/benchmarks/tensorrt_build.py | 239 +++++++++++++++++++++++++++++++ docs/benchmarks.md | 36 +++++ docs/quantization-status.md | 10 +- tests/test_benchmark_tensorrt.py | 147 +++++++++++++++++++ 5 files changed, 493 insertions(+), 4 deletions(-) create mode 100644 dev/benchmarks/tensorrt_build.py create mode 100644 tests/test_benchmark_tensorrt.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index cf13a20..9ceebb6 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -685,3 +685,68 @@ inspection. Direct hierarchical engine construction requires registering the standard TensorRT plugins (`init_libnvinfer_plugins`) for scatter reductions. Do not copy device-specific engines between target machines. See the [TensorRT evidence and quality limits](../../docs/benchmarks.md#tensorrt-int8-calibration-candidate). + +### Maintained TensorRT build and inspection command + +`dev.benchmarks.tensorrt_build` builds an engine from an existing ONNX graph, +records detailed layer inspection, executes one supplied input, and optionally +checks its outputs against a named reference NPZ. It uses CUDA PyTorch for probe +buffers/stream management, plus explicitly installed compatible TensorRT and ONNX; +it never installs packages. This is a developer validation command, not a +PyTorch dependency for the deployed TensorRT engine. Module import and `--help` +do not load TensorRT or PyTorch. + +```bash +python -m dev.benchmarks.tensorrt_build \ + --model signed-symmetric-qdq/model.onnx --inputs preprocessed-batch.npz \ + --profiles profiles.json --output /tmp/trt-build-1 --optimization 3 + +# Practical floating baseline from the unquantized graph: +python -m dev.benchmarks.tensorrt_build \ + --model exported-float/model.onnx --inputs preprocessed-batch.npz \ + --profiles profiles.json --output /tmp/trt-fp16-build-1 --fp16 --optimization 3 +``` + +Every input name must occur in the profiles JSON, for example: + +```json +{"images": {"min": [1, 3, 128, 128], "opt": [8, 3, 128, 128], "max": [8, 3, 128, 128]}} +``` + +Without `--profiles`, all min/opt/max shapes equal the supplied input shapes. +The smoke-test input may be anywhere within an explicit profile. Fixed network +dimensions and input dtypes must match; arrays are not implicitly cast. Profiles +control the engine's supported input range, not a claim that every shape in that +range has been executed or validated. No model/head allowlist is used. + +The default workspace limit is 1024 MiB and builder optimization level is 3; +`--workspace-mib`, `--optimization` and `--device` are explicit controls. TF32 is +disabled unless `--tf32` is supplied. `--fp16` allows FP16 tactics. QDQ precision +comes from the graph; the command does not calibrate, insert QDQ or replace the +native training quantizer. Standard TensorRT plugins are registered before parsing. + +A new output directory receives `model.engine`, `layers.json`, `outputs.npz` and +`report.json`, including graph/external-weight/input hashes, SDK versions, GPU, +settings, engine hash/size and reported context-memory requirement. Failures after +output creation retain diagnostics, including parser errors and warning/error +messages. An existing directory is never overwritten. A built engine or smaller +file does not certify integer execution, speed, total GPU-memory reduction or +acceptable model quality; inspect layer types/tactics and run the separate studies. + +Optional `--reference expected-outputs.npz` requires exact output names, shapes +and dtypes. Floating parity defaults to `--rtol 1e-4 --atol 1e-5`; integer/bool +outputs compare exactly. Failed parity retains the engine, actual outputs and +comparison report. Absence of a reference is recorded and means parity was not +checked. A single input does not replace full dataset evaluation with mini_metrics. + +The probe currently supports ordinary execution inputs and linear device IO. +Shape-tensor input profiles, host/nonlinear IO bindings, unresolved data-dependent +output dimensions and empty outputs fail explicitly. These are probe limitations, +not blanket limitations of ONNX or TensorRT. Engines must be rebuilt and tested +for their actual destination device/software environment. + +CPU profile/reference contracts run in the normal suite. In the prepared GPU +environment, set `CUDA_VISIBLE_DEVICES` and `RUN_CUDA_TESTS=1`, then run +`bash dev/check.sh test tests/test_benchmark_tensorrt.py` to include real engine +construction, successful reference matching, retained mismatches and parser +failure diagnostics. diff --git a/dev/benchmarks/tensorrt_build.py b/dev/benchmarks/tensorrt_build.py new file mode 100644 index 0000000..b0e8a5c --- /dev/null +++ b/dev/benchmarks/tensorrt_build.py @@ -0,0 +1,239 @@ +"""Build, inspect and smoke-test an ONNX TensorRT engine in an explicit environment.""" + +import json +import math +import platform +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np + +from .onnx_inference import file_hash, model_files + + +def input_profiles(feeds, profiles=None): + """Validate explicit ranges; absent ranges use the supplied fixed input shapes.""" + if profiles is None: + profiles = {name: {key: list(value.shape) for key in ("min", "opt", "max")} for name, value in feeds.items()} + if set(profiles) != set(feeds): + raise ValueError("Profile names must match the named input arrays") + result = {} + for name, value in feeds.items(): + ranges = profiles[name] + if set(ranges) != {"min", "opt", "max"}: + raise ValueError(f"Profile {name} needs min, opt and max shapes") + for shape in ranges.values(): + if len(shape) != value.ndim or any(type(n) is not int or n < 1 for n in shape): + raise ValueError(f"Invalid dimensions in profile {name}") + for low, opt, high, sample in zip(ranges["min"], ranges["opt"], ranges["max"], value.shape, strict=True): + if not low <= opt <= high or not low <= sample <= high: + raise ValueError(f"Unordered ranges or sample outside profile {name}") + result[name] = {key: list(shape) for key, shape in ranges.items()} + return result + + +def compare_outputs(actual, expected, rtol, atol): + if set(actual) != set(expected): + raise ValueError("Reference output names do not match engine outputs") + result = {} + for name, value in actual.items(): + reference = expected[name] + if value.shape != reference.shape or value.dtype != reference.dtype: + raise ValueError(f"Reference shape/dtype does not match output {name}") + if not np.isfinite(reference).all(): + raise ValueError(f"Nonfinite reference output {name}") + close = np.isclose(value, reference, rtol=rtol, atol=atol) if value.dtype.kind == "f" else value == reference + result[name] = {"passed": bool(close.all()), "elements_outside_tolerance": int((~close).sum())} + return result + + +def build( + model, + inputs, + output, + profiles=None, + fp16=False, + tf32=False, + workspace_mib=1024, + optimization=3, + device=0, + reference=None, + rtol=1e-4, + atol=1e-5, +): + if workspace_mib < 1 or not 0 <= optimization <= 5 or device < 0: + raise ValueError("Require positive workspace, optimization 0–5 and nonnegative device index") + if any(not math.isfinite(x) or x < 0 for x in (rtol, atol)): + raise ValueError("Parity tolerances must be finite and nonnegative") + with np.load(inputs, allow_pickle=False) as data: + feeds = {name: data[name].copy(order="C") for name in data.files} + if not feeds or any(not np.isfinite(a).all() for a in feeds.values()): + raise ValueError("Supply finite named input arrays") + profiles = input_profiles(feeds, profiles) + try: + import onnx + import tensorrt as trt + import torch + except ImportError as error: + raise ImportError( + "Use an explicitly prepared compatible TensorRT, ONNX and CUDA PyTorch environment; no packages are installed by this command" + ) from error + + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "versions": {"tensorrt": trt.__version__, "torch": torch.__version__, "onnx": onnx.__version__, "numpy": np.__version__}, + "environment": {"platform": platform.platform(), "python": platform.python_version()}, + "settings": { + "profiles": profiles, + "fp16_allowed": fp16, + "tf32_allowed": tf32, + "workspace_bytes": workspace_mib * 1024**2, + "builder_optimization": optimization, + "device": device, + }, + "inputs": { + "path": str(inputs), + "sha256": file_hash(inputs), + "arrays": {name: {"shape": list(a.shape), "dtype": str(a.dtype)} for name, a in feeds.items()}, + }, + "reference": None, + "messages": [], + "scope": ( + "Engine build, detailed layer inspection and one input execution; not a quality or performance certification. " + "Workspace/context bytes are not total GPU memory." + ), + } + + class Logger(trt.ILogger): + def log(self, severity, message): + if severity <= trt.ILogger.WARNING: + report["messages"].append({"severity": str(severity), "message": message}) + + try: + report["model_files"] = model_files(Path(model), onnx) + with torch.cuda.device(device): + report["environment"]["gpu"] = torch.cuda.get_device_name(device) + report["environment"]["compute_capability"] = list(torch.cuda.get_device_capability(device)) + logger = Logger() + trt.init_libnvinfer_plugins(logger, "") + builder = trt.Builder(logger) + network = builder.create_network(0) + parser = trt.OnnxParser(network, logger) + if not parser.parse_from_file(str(model)): + report["parser_errors"] = [str(parser.get_error(i)) for i in range(parser.num_errors)] + raise RuntimeError("TensorRT could not parse the model; see parser_errors in report.json") + tensors = {network.get_input(i).name: network.get_input(i) for i in range(network.num_inputs)} + if set(tensors) != set(feeds): + raise ValueError("Input names do not match the parsed network") + profile = builder.create_optimization_profile() + for name, tensor in tensors.items(): + if tensor.is_shape_tensor: + raise ValueError(f"Shape-tensor input profiles are not supported by this probe: {name}") + if np.dtype(trt.nptype(tensor.dtype)) != feeds[name].dtype: + raise ValueError(f"Input dtype does not match network: {name}") + for shape in profiles[name].values(): + if len(shape) != len(tensor.shape) or any(n >= 0 and n != s for n, s in zip(tensor.shape, shape, strict=True)): + raise ValueError(f"Profile changes a fixed network dimension: {name}") + profile.set_shape(name, profiles[name]["min"], profiles[name]["opt"], profiles[name]["max"]) + config = builder.create_builder_config() + config.set_memory_pool_limit(trt.MemoryPoolType.WORKSPACE, workspace_mib * 1024**2) + config.builder_optimization_level = optimization + config.profiling_verbosity = trt.ProfilingVerbosity.DETAILED + config.clear_flag(trt.BuilderFlag.TF32) + if tf32: + config.set_flag(trt.BuilderFlag.TF32) + if fp16: + config.set_flag(trt.BuilderFlag.FP16) + if config.add_optimization_profile(profile) < 0: + raise ValueError("TensorRT rejected the optimization profile") + serialized = builder.build_serialized_network(network, config) + if serialized is None: + raise RuntimeError("TensorRT engine construction failed; see messages in report.json") + engine_path = output / "model.engine" + engine_path.write_bytes(serialized) + runtime = trt.Runtime(logger) + engine = runtime.deserialize_cuda_engine(serialized) + if engine is None: + raise RuntimeError("TensorRT could not deserialize the constructed engine") + layers_path = output / "layers.json" + layers_path.write_text(engine.create_engine_inspector().get_engine_information(trt.LayerInformationFormat.JSON)) + report["engine"] = { + "sha256": file_hash(engine_path), + "bytes": engine_path.stat().st_size, + "context_memory_bytes": engine.device_memory_size_v2, + "layers_sha256": file_hash(layers_path), + "num_layers": engine.num_layers, + } + context = engine.create_execution_context() + stream = torch.cuda.Stream(device=device) + buffers, outputs = {}, {} + with torch.cuda.stream(stream): + for name, value in feeds.items(): + buffers[name] = torch.from_numpy(value).to(device=f"cuda:{device}") + if not context.set_input_shape(name, value.shape): + raise ValueError(f"TensorRT rejected sample shape for {name}") + for index in range(engine.num_io_tensors): + name = engine.get_tensor_name(index) + if ( + engine.get_tensor_format(name) != trt.TensorFormat.LINEAR + or engine.get_tensor_location(name) != trt.TensorLocation.DEVICE + ): + raise ValueError(f"Probe requires linear device IO tensors: {name}") + if engine.get_tensor_mode(name) == trt.TensorIOMode.OUTPUT: + shape = tuple(context.get_tensor_shape(name)) + if any(n <= 0 for n in shape): + raise ValueError(f"Unresolved or empty output shape for {name}: {shape}") + dtype = torch.from_numpy(np.empty((), dtype=trt.nptype(engine.get_tensor_dtype(name)))).dtype + buffers[name] = torch.empty(shape, device=f"cuda:{device}", dtype=dtype) + outputs[name] = buffers[name] + if not context.set_tensor_address(name, buffers[name].data_ptr()): + raise RuntimeError(f"TensorRT rejected IO address for {name}") + if not context.execute_async_v3(stream.cuda_stream): + raise RuntimeError("TensorRT execution failed") + stream.synchronize() + actual = {name: value.cpu().numpy() for name, value in outputs.items()} + if any(not np.isfinite(a).all() for a in actual.values()): + raise ValueError("Engine produced nonfinite outputs") + np.savez(output / "outputs.npz", **actual) + report["outputs"] = {name: {"shape": list(a.shape), "dtype": str(a.dtype)} for name, a in actual.items()} + if reference is not None: + report["reference"] = {"path": str(reference), "sha256": file_hash(reference), "rtol": rtol, "atol": atol} + with np.load(reference, allow_pickle=False) as expected: + report["reference"]["comparison"] = compare_outputs(actual, expected, rtol, atol) + if not all(r["passed"] for r in report["reference"]["comparison"].values()): + raise ValueError("Engine outputs failed reference parity") + report["status"] = "passed" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--inputs", type=Path, required=True, help="Named preprocessed input arrays in NPZ format") + parser.add_argument("--output", type=Path, required=True, help="New directory for engine, inspection, outputs and report") + parser.add_argument("--profiles", type=Path, help="JSON object mapping each input name to min/opt/max shape lists") + parser.add_argument("--fp16", action="store_true", help="Allow FP16 tactics; QDQ quantization is defined by the graph") + parser.add_argument("--tf32", action="store_true") + parser.add_argument("--workspace-mib", type=int, default=1024) + parser.add_argument("--optimization", type=int, choices=range(6), default=3) + parser.add_argument("--device", type=int, default=0) + parser.add_argument("--reference", type=Path, help="Optional NPZ of expected outputs with exact names/shapes/dtypes") + parser.add_argument("--rtol", type=float, default=1e-4) + parser.add_argument("--atol", type=float, default=1e-5) + args = vars(parser.parse_args()) + if args["profiles"] is not None: + args["profiles"] = json.loads(args["profiles"].read_text()) + build(**args) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 0be6628..a0f1c63 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1844,3 +1844,39 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, `paired_inference.py` probes. Reports retain engine/source hashes and all timing samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. + +### Maintained TensorRT engine reproduction + +`dev.benchmarks.tensorrt_build` replaces the one-off engine build/inspection probe +with a documented command. It accepts arbitrary named execution inputs, explicit +min/opt/max profiles, precision flags and workspace/build settings. It records +model/external-weight/input hashes, environment, detailed layers, engine size and +context-memory requirement, and one set of actual outputs. An optional reference +checks exact names/shapes/dtypes and explicit numerical tolerances. Parser and +parity failures retain failed reports and available artifacts; a fresh output +directory is required. + +The command was exercised on both signed floating-bias EfficientNetV2-S QDQ +candidates above, using batch eight, profile 1–8, size 128, workspace 1 GiB, builder +optimization level 1, and TF32/FP16 flags disabled. All flat and hierarchical +outputs passed comparison to the retained direct-engine outputs at rtol=1e-4, +atol=1e-5. Inspection again shows 170 INT8 convolutions and two head GEMMs per +engine. Reports and inspection are retained under ignored +`tmp-trt-probe/maintained-{flat,hierarchical}/`. + +These checks establish reproduction and one-input execution, not new throughput, +full-dataset quality or every-shape parity results. TensorRT can choose different +tactics on rebuild; compare the exact engine hashes when interpreting timing +reports. The command is generic across model/head names but explicitly rejects +shape-tensor input profiles, host/nonlinear IO bindings and unresolved/empty +outputs that its smoke-test buffer handling does not yet support. + +CPU contracts cover profile ranges, exact integer reference comparison, floating +parity and CLI help without TensorRT/PyTorch imports. Intentional CUDA checks cover real multi-input +engine construction, successful parity, retained mismatches, parser failure +messages and external-weight provenance. See the +[command and environment instructions](../dev/benchmarks/README.md#maintained-tensorrt-build-and-inspection-command). + +Validation passed static checks and 406 CPU-default tests (149 skips and the +known EMA expected failure), plus all 16 focused checks in the prepared CUDA/ +TensorRT environment and both real-model reference comparisons above. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 54454f8..f67ad4d 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,7 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained production preparation/inspection workflow | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained build/inspection/smoke command | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained calibration, paired timing and dataset-quality commands | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | @@ -42,9 +42,11 @@ production acceptance study. ## Remaining work, in practical order -1. **Make the latest experiments reproducible from a clean checkout.** Promote the - useful preparation, engine inspection, paired inference and metric-evaluation - probes into maintained developer commands with explicit optional environments. +1. **Make the latest experiments reproducible from a clean checkout.** Engine + build, inspection and optional smoke-test parity now have a + [maintained command](../dev/benchmarks/README.md#maintained-tensorrt-build-and-inspection-command). + Promote calibration preparation, paired inference and dataset metric evaluation + into maintained developer commands with explicit optional environments. Preserve calibration records, class/preprocessing contracts, hashes, failures and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` diff --git a/tests/test_benchmark_tensorrt.py b/tests/test_benchmark_tensorrt.py new file mode 100644 index 0000000..b4e9c0b --- /dev/null +++ b/tests/test_benchmark_tensorrt.py @@ -0,0 +1,147 @@ +import json +import os +import subprocess +import sys + +import numpy as np +import pytest + +from dev.benchmarks.tensorrt_build import build, compare_outputs, input_profiles + + +def test_profiles_preserve_named_shapes_and_allow_sample_away_from_optimum(): + feeds = {"images": np.zeros((2, 3, 8, 8)), "offset": np.zeros(3)} + profiles = input_profiles(feeds) + profiles["images"] = {"min": [1, 3, 8, 8], "opt": [4, 3, 8, 8], "max": [8, 3, 8, 8]} + assert input_profiles(feeds, profiles) == profiles + assert input_profiles({"scalar": np.array(1.0)})["scalar"] == {"min": [], "opt": [], "max": []} + + +@pytest.mark.parametrize( + "ranges", + [ + {"min": [1], "opt": [1], "max": [1]}, + {"min": [3], "opt": [2], "max": [4]}, + {"min": [1], "opt": [5], "max": [4]}, + {"min": [0], "opt": [2], "max": [4]}, + {"min": [True], "opt": [2], "max": [4]}, + {"min": [1.0], "opt": [2], "max": [4]}, + {"min": [1, 1], "opt": [2, 1], "max": [4, 1]}, + {"min": [1], "max": [4]}, + ], +) +def test_profiles_reject_invalid_or_uncovered_shapes(ranges): + with pytest.raises(ValueError): + input_profiles({"x": np.zeros(2)}, {"x": ranges}) + + +def test_profiles_require_exact_input_names(): + with pytest.raises(ValueError, match="names"): + input_profiles({"x": np.zeros(2)}, {"y": {"min": [1], "opt": [2], "max": [4]}}) + + +def test_integer_reference_parity_is_exact_even_above_float_precision(): + actual = {"x": np.array([2**60], dtype=np.int64)} + expected = {"x": np.array([2**60 + 1], dtype=np.int64)} + assert compare_outputs(actual, expected, 1, 1)["x"] == {"passed": False, "elements_outside_tolerance": 1} + + +def test_reference_checks_float_tolerance_and_contract(): + actual = {"x": np.array([1.0, 2.0], dtype=np.float32)} + expected = {"x": np.array([1.01, 2.0], dtype=np.float32)} + assert compare_outputs(actual, expected, 0, 0.02)["x"]["passed"] + assert not compare_outputs(actual, expected, 0, 0.001)["x"]["passed"] + for wrong in ( + {"y": actual["x"]}, + {"x": np.zeros(2, dtype=np.float64)}, + {"x": np.zeros(3, dtype=np.float32)}, + {"x": np.full(2, np.nan, dtype=np.float32)}, + ): + with pytest.raises(ValueError): + compare_outputs(actual, wrong, 0, 0) + + +def test_help_does_not_import_tensorrt_or_torch(): + code = """ +import runpy, sys +sys.argv = ['tensorrt_build', '--help'] +try: + runpy.run_module('dev.benchmarks.tensorrt_build', run_name='__main__') +except SystemExit as error: + assert error.code == 0 +assert 'tensorrt' not in sys.modules +assert 'torch' not in sys.modules +""" + subprocess.run([sys.executable, "-c", code], check=True, capture_output=True, text=True) + + +def gpu_dependencies(): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 in a prepared TensorRT/CUDA environment") + pytest.importorskip("tensorrt") + import torch + + if not torch.cuda.is_available(): + pytest.fail("CUDA requested but unavailable") + return pytest.importorskip("onnx") + + +@pytest.mark.parametrize("wrong_reference", [False, True]) +def test_tensorrt_build_records_execution_and_reference_failure(tmp_path, wrong_reference): + onnx = gpu_dependencies() + graph = onnx.helper.make_graph( + [onnx.helper.make_node("MatMul", ["x", "weight"], ["product"]), onnx.helper.make_node("Add", ["product", "offset"], ["scores"])], + "two-input-linear", + [ + onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, ["batch", 2]), + onnx.helper.make_tensor_value_info("offset", onnx.TensorProto.FLOAT, [2]), + ], + [onnx.helper.make_tensor_value_info("scores", onnx.TensorProto.FLOAT, ["batch", 2])], + [onnx.numpy_helper.from_array(np.eye(2, dtype=np.float32), "weight")], + ) + model = onnx.helper.make_model(graph, opset_imports=[onnx.helper.make_opsetid("", 18)], ir_version=10) + path = tmp_path / "model.onnx" + onnx.save_model(model, path, save_as_external_data=True, all_tensors_to_one_file=True, location="weights.data", size_threshold=0) + x, offset = np.array([[1, 2], [3, 4]], dtype=np.float32), np.array([0.5, -0.5], dtype=np.float32) + inputs, reference, output = tmp_path / "inputs.npz", tmp_path / "reference.npz", tmp_path / "result" + np.savez(inputs, x=x, offset=offset) + np.savez(reference, scores=x + offset + (1 if wrong_reference else 0)) + profiles = {"x": {"min": [1, 2], "opt": [2, 2], "max": [4, 2]}, "offset": {"min": [2], "opt": [2], "max": [2]}} + if wrong_reference: + with pytest.raises(ValueError, match="failed reference parity"): + build(path, inputs, output, profiles=profiles, reference=reference, optimization=0) + else: + build(path, inputs, output, profiles=profiles, reference=reference, optimization=0) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == ("failed" if wrong_reference else "passed") + assert report["reference"]["comparison"]["scores"]["passed"] != wrong_reference + assert len(report["model_files"]) == 2 + assert report["engine"]["bytes"] > 0 + assert json.loads((output / "layers.json").read_text())["Layers"] + with np.load(output / "outputs.npz") as result: + np.testing.assert_allclose(result["scores"], x + offset, rtol=0, atol=0) + with pytest.raises(FileExistsError): + build(path, inputs, output) + + +def test_parser_failure_retains_diagnostics_without_engine(tmp_path): + onnx = gpu_dependencies() + graph = onnx.helper.make_graph( + [onnx.helper.make_node("MissingBenchmarkPlugin", ["x"], ["y"], domain="mini_trainer.test")], + "unsupported", + [onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, [2])], + [onnx.helper.make_tensor_value_info("y", onnx.TensorProto.FLOAT, [2])], + ) + model = onnx.helper.make_model( + graph, opset_imports=[onnx.helper.make_opsetid("", 18), onnx.helper.make_opsetid("mini_trainer.test", 1)], ir_version=10 + ) + path, inputs, output = tmp_path / "model.onnx", tmp_path / "inputs.npz", tmp_path / "failure" + onnx.save(model, path) + np.savez(inputs, x=np.ones(2, dtype=np.float32)) + with pytest.raises(RuntimeError, match="could not parse"): + build(path, inputs, output) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" + assert report["parser_errors"] + assert report["messages"] + assert not (output / "model.engine").exists() From e0d3f2b5617712caefb6e61c75076ecba9bedbeb Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 14:22:32 +0200 Subject: [PATCH 062/155] feat: add reproducible ONNX calibration with bounded collection --- dev/benchmarks/README.md | 81 ++++++++- dev/benchmarks/onnx_calibration.py | 259 ++++++++++++++++++++++++++++ docs/benchmarks.md | 42 +++++ docs/quantization-status.md | 10 +- tests/test_benchmark_calibration.py | 169 ++++++++++++++++++ 5 files changed, 556 insertions(+), 5 deletions(-) create mode 100644 dev/benchmarks/onnx_calibration.py create mode 100644 tests/test_benchmark_calibration.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 9ceebb6..e358aa8 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -656,6 +656,82 @@ unsigned activation recipe measured on x86 is a candidate to test, not a univers GPU/ARM recipe. This inference runner does not validate HPC PyTorch quantized training, native QT checkpoint export or million-class capacity. +### Maintained ONNX calibration command + +`dev.benchmarks.onnx_calibration` creates a fresh QDQ model from ordered, +preprocessed NPZ batches. Use an explicitly prepared ONNX/ONNX Runtime environment; +the command installs nothing and requires the calibration-cache API tested with +ONNX Runtime 1.29.0. Calibration and the final smoke execution run on CPU with +`--threads 1` by default, including the first session created by the calibrator. + +Supply a manifest such as: + +```json +{ + "schema_version": 1, + "split": "train", + "batch_input": "images", + "provenance": { + "dataset": "versioned dataset manifest and hash", + "checkpoint": "source checkpoint hash", + "selection": "sample selection algorithm and seed", + "preprocessing": "exact resize, channel order, normalization and dtype" + }, + "batches": [ + {"path": "batch-000.npz", "sample_ids": ["train/image-001", "train/image-002"]} + ] +} +``` + +Each NPZ must contain all model inputs under their ONNX names, with finite arrays +and matching dtypes/shapes. `batch_input` identifies the input whose leading +dimension matches `sample_ids`; other inputs can be static or unbatched. Paths +are relative to the manifest. Optional batch `sha256` values are checked before +use, and the command always records hashes of the bytes actually loaded. Sample +IDs must be unique across batches. The split must be `train` or `calibration`; +this declaration cannot prove separation from held-out data, so review the +selection and provenance. Preserve batch order and boundaries for reproducible +histogram collection. Prepare inputs using the source model's preprocessing; +this command does not infer image transforms or class mappings. + +```bash +# Unsigned activations, signed per-channel weights, quantized biases (CPU candidate): +python -m dev.benchmarks.onnx_calibration \ + --model exported-float/model.onnx --manifest calibration/manifest.json \ + --output /tmp/qdq-cpu-1 --threads 1 + +# Signed symmetric activations and floating biases (tested TensorRT candidate): +python -m dev.benchmarks.onnx_calibration \ + --model exported-float/model.onnx --manifest calibration/manifest.json \ + --output /tmp/qdq-trt-1 --threads 1 \ + --activation-type int8 --symmetric-activations --float-bias +``` + +Defaults are Percentile 99.9, asymmetric histogram collection, symmetric INT8 +per-channel weights, and Conv/Gemm/MatMul selection. `--method minmax`, +`--percentile`, `--symmetric-calibration`, `--per-tensor-weights` and repeated +`--op-type` flags make alternatives explicit. Histogram symmetry and activation +quantizer symmetry are separate settings: the tested TensorRT recipe derives +symmetric quantizer ranges from asymmetric histograms, so it does **not** use +`--symmetric-calibration`. + +A new output directory receives a private source-model snapshot, augmented +calibration graph, `ranges.json`, quantized `model.onnx` and its external weights, +`calibration-smoke.npz`, and `report.json`. Source/external-file hashes, recipe, +versions, ordered sample IDs, input hashes and failures after output creation are +retained. Existing output directories are refused. The private snapshot isolates +ORT's shape-inference sidecars from the source directory. Histogram collection +retains raw activations for one batch at a time; model loading, histograms and +graph construction still consume memory. The artifact directory includes research +intermediates, not only deployment files. + +The smoke run checks finite outputs on one calibration batch. It does not measure +held-out quality, integer execution coverage, speed or memory benefit. Continue +with provider/engine inspection and paired evaluation using mini_metrics. CPU +regressions cover both calibration methods, signed/unsigned recipes, multiple +inputs, thread limits, unchanged source files and retained failure reports in +`tests/test_benchmark_calibration.py`. + ### TensorRT calibration candidate The local candidate uses signed, symmetric INT8 activation/weight QDQ, per-channel @@ -664,8 +740,9 @@ Percentile 99.9 training-split calibration cache. With ONNX Runtime 1.29.0's `quantize_static`, the relevant arguments are `quant_format=QuantFormat.QDQ`, `activation_type=QuantType.QInt8`, `weight_type=QuantType.QInt8`, `per_channel=True` and `extra_options={"ActivationSymmetric": True, "WeightSymmetric": True, -"QuantizeBias": False}`. Supply a reviewed calibration reader/cache and use -separate output artifacts. This changes the quantization recipe; it does not +"QuantizeBias": False}`. The maintained calibration command above regenerates +this recipe from reviewed input batches into separate artifacts. This changes +the quantization recipe; it does not preserve the native training export's dynamic quantizer. In a separately prepared compatible TensorRT/ONNX Runtime GPU environment: diff --git a/dev/benchmarks/onnx_calibration.py b/dev/benchmarks/onnx_calibration.py new file mode 100644 index 0000000..0c2fe98 --- /dev/null +++ b/dev/benchmarks/onnx_calibration.py @@ -0,0 +1,259 @@ +"""Calibrate a fresh ONNX QDQ artifact from explicit, ordered input batches.""" + +import hashlib +import inspect +import io +import json +import math +import platform +from argparse import ArgumentParser +from collections import Counter +from pathlib import Path + +import numpy as np + +from .onnx_inference import file_hash, model_files + + +def calibration_manifest(path): + manifest = json.loads(Path(path).read_text()) + if manifest.get("schema_version") != 1 or manifest.get("split") not in ("train", "calibration"): + raise ValueError("Require schema_version=1 and a declared train/calibration split") + if not isinstance(manifest.get("batch_input"), str) or not manifest["batch_input"]: + raise ValueError("Declare the input whose leading dimension identifies samples: batch_input") + if not isinstance(manifest.get("provenance"), dict) or not manifest["provenance"]: + raise ValueError("Record dataset/preprocessing provenance") + if not isinstance(manifest.get("batches"), list) or not manifest["batches"]: + raise ValueError("Supply an ordered nonempty batches list") + seen = set() + for batch in manifest["batches"]: + if not isinstance(batch.get("path"), str) or not batch["path"]: + raise ValueError("Each batch needs an NPZ path") + ids = batch.get("sample_ids") + if not isinstance(ids, list) or not ids or any(not isinstance(x, str) or not x for x in ids): + raise ValueError("Each batch needs nonempty string sample_ids") + if len(set(ids)) != len(ids) or seen.intersection(ids): + raise ValueError("Calibration sample IDs must be unique") + seen.update(ids) + return manifest + + +def load_batch(manifest_path, manifest, batch): + path = (Path(manifest_path).parent / batch["path"]).resolve() + payload = path.read_bytes() + digest = hashlib.sha256(payload).hexdigest() + if batch.get("sha256") is not None and batch["sha256"] != digest: + raise ValueError(f"Calibration input hash mismatch: {path}") + with np.load(io.BytesIO(payload), allow_pickle=False) as archive: + feeds = {name: archive[name].copy(order="C") for name in archive.files} + if not feeds or any(not np.isfinite(a).all() for a in feeds.values()): + raise ValueError(f"Calibration arrays must be finite: {path}") + name = manifest["batch_input"] + if name not in feeds or feeds[name].ndim == 0 or feeds[name].shape[0] != len(batch["sample_ids"]): + raise ValueError(f"sample_ids do not match the batch_input leading dimension: {path}") + record = { + "path": str(path), + "sha256": digest, + "sample_ids": batch["sample_ids"], + "arrays": {name: {"shape": list(a.shape), "dtype": str(a.dtype)} for name, a in feeds.items()}, + } + return feeds, record + + +def calibrate( + model, + manifest, + output, + method="percentile", + percentile=99.9, + activation_type="uint8", + symmetric_activations=False, + symmetric_calibration=False, + float_bias=False, + per_channel=True, + op_types=("Conv", "Gemm", "MatMul"), + threads=1, +): + if method not in ("minmax", "percentile") or activation_type not in ("uint8", "int8"): + raise ValueError("Choose minmax/percentile calibration and uint8/int8 activations") + if not math.isfinite(percentile) or not 0 < percentile <= 100 or threads < 1 or not op_types: + raise ValueError("Require percentile in (0,100], positive threads and selected operator types") + metadata = calibration_manifest(manifest) + try: + import onnx + import onnxruntime as ort + from onnxruntime.quantization import CalibrationDataReader, CalibrationMethod, QuantFormat, QuantType, quantize_static + from onnxruntime.quantization.calibrate import MinMaxCalibrater, PercentileCalibrater, save_tensors_data + except ImportError as error: + raise ImportError("Use an explicitly prepared ONNX/ONNX Runtime quantization environment; this command installs nothing") from error + if "calibration_cache_path" not in inspect.signature(quantize_static).parameters: + raise RuntimeError("This command requires ONNX Runtime's calibration-cache API (tested with 1.29.0)") + + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "versions": {"onnx": onnx.__version__, "onnxruntime": ort.__version__, "numpy": np.__version__}, + "environment": {"platform": platform.platform(), "python": platform.python_version()}, + "calibration_manifest": {"path": str(manifest), "sha256": file_hash(manifest), "contents": metadata}, + "recipe": { + "format": "QDQ", + "method": method, + "percentile": percentile if method == "percentile" else None, + "histogram_symmetric": symmetric_calibration, + "activation_type": activation_type, + "activation_symmetric": symmetric_activations, + "weight_type": "int8", + "weight_symmetric": True, + "per_channel_weights": per_channel, + "quantize_bias": not float_bias, + "op_types": list(op_types), + }, + "execution": { + "provider": "CPUExecutionProvider", + "intra_op_threads": threads, + "inter_op_threads": 1, + "calibration_graph_optimization": "disabled", + "collection_batch_limit": 1, + }, + "batches": [], + "scope": ( + "Calibration from declared training/calibration inputs and one calibration-batch execution; " + "not held-out quality, integer-kernel placement or deployment acceptance." + ), + } + options = ort.SessionOptions() + options.intra_op_num_threads = threads + options.inter_op_num_threads = 1 + options.graph_optimization_level = ort.GraphOptimizationLevel.ORT_DISABLE_ALL + base = MinMaxCalibrater if method == "minmax" else PercentileCalibrater + + class Calibrator(base): + def create_inference_session(self): + self.infer_session = ort.InferenceSession(self.augmented_model_path, sess_options=options, providers=["CPUExecutionProvider"]) + self.infer_session.disable_fallback() + + class OneBatch(CalibrationDataReader): + def __init__(self, feeds): + self.feeds = feeds + + def get_next(self): + feeds, self.feeds = self.feeds, None + return feeds + + try: + report["source_files"] = model_files(Path(model), onnx) + source = onnx.load(model) + report["source_ops"] = dict(Counter(node.op_type for node in source.graph.node)) + if not set(op_types).intersection(report["source_ops"]): + raise ValueError("No selected operator types occur in the source graph") + # ORT writes inferred-model sidecars beside its input. Isolate those writes. + snapshot_dir = output / "source" + snapshot_dir.mkdir() + snapshot = snapshot_dir / "model.onnx" + onnx.save_model( + source, snapshot, save_as_external_data=True, all_tensors_to_one_file=True, location="weights.data", size_threshold=0 + ) + del source + report["snapshot_files"] = model_files(snapshot, onnx) + collector = Calibrator( + snapshot, + list(op_types), + augmented_model_path=str(output / "calibration.onnx"), + use_external_data_format=True, + symmetric=symmetric_calibration, + **({"percentile": percentile} if method == "percentile" else {}), + ) + collector.augment_graph() + collector.create_inference_session() + names = {node.name for node in collector.infer_session.get_inputs()} + for entry in metadata["batches"]: + feeds, record = load_batch(manifest, metadata, entry) + report["batches"].append(record) + if set(feeds) != names: + raise ValueError("Calibration input names do not match the source model") + collector.collect_data(OneBatch(feeds)) + del feeds + cache = output / "ranges.json" + ranges = collector.compute_data() + if not len(ranges.data): + raise ValueError("No tensor ranges were calibrated") + report["calibrated_tensors"] = len(ranges.data) + save_tensors_data(ranges, cache) + del collector, ranges + report["ranges_sha256"] = file_hash(cache) + quantized = output / "model.onnx" + quantize_static( + snapshot, + quantized, + None, + quant_format=QuantFormat.QDQ, + per_channel=per_channel, + activation_type=QuantType.QUInt8 if activation_type == "uint8" else QuantType.QInt8, + weight_type=QuantType.QInt8, + op_types_to_quantize=list(op_types), + calibration_cache_path=cache, + use_external_data_format=True, + calibrate_method=CalibrationMethod.MinMax if method == "minmax" else CalibrationMethod.Percentile, + extra_options={"ActivationSymmetric": symmetric_activations, "WeightSymmetric": True, "QuantizeBias": not float_bias}, + ) + onnx.checker.check_model(str(quantized)) + graph = onnx.load(quantized, load_external_data=False) + report["output_ops"] = dict(Counter(node.op_type for node in graph.graph.node)) + if not report["output_ops"].get("QuantizeLinear") or not report["output_ops"].get("DequantizeLinear"): + raise ValueError("The output graph contains no complete QDQ quantization") + del graph + report["output_files"] = model_files(quantized, onnx) + # A smoke execution of calibration data establishes loadability, not quality. + feeds, record = load_batch(manifest, metadata, metadata["batches"][0]) + if record["sha256"] != report["batches"][0]["sha256"]: + raise ValueError("First calibration batch changed before smoke execution") + smoke_options = ort.SessionOptions() + smoke_options.intra_op_num_threads = threads + smoke_options.inter_op_num_threads = 1 + session = ort.InferenceSession(str(quantized), sess_options=smoke_options, providers=["CPUExecutionProvider"]) + session.disable_fallback() + arrays = session.run(None, feeds) + if any(not np.isfinite(a).all() for a in arrays): + raise ValueError("Quantized graph produced nonfinite calibration-smoke outputs") + output_names = [node.name for node in session.get_outputs()] + np.savez(output / "calibration-smoke.npz", **dict(zip(output_names, arrays, strict=True))) + report["smoke"] = { + "batch_index": 0, + "graph_optimization": "all", + "outputs": output_names, + "sha256": file_hash(output / "calibration-smoke.npz"), + } + report["status"] = "passed" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--manifest", type=Path, required=True, help="Ordered calibration NPZ batches, sample IDs and provenance") + parser.add_argument("--output", type=Path, required=True, help="New artifact directory") + parser.add_argument("--method", choices=["minmax", "percentile"], default="percentile") + parser.add_argument("--percentile", type=float, default=99.9) + parser.add_argument("--activation-type", choices=["uint8", "int8"], default="uint8") + parser.add_argument("--symmetric-activations", action="store_true") + parser.add_argument("--symmetric-calibration", action="store_true", help="Collect symmetric calibration ranges/histograms") + parser.add_argument("--float-bias", action="store_true", help="Keep biases floating instead of quantizing them to INT32") + parser.add_argument("--per-tensor-weights", dest="per_channel", action="store_false") + parser.add_argument("--op-type", dest="op_types", action="append", help="Operator type to quantize; repeatable") + parser.add_argument("--threads", type=int, default=1) + args = vars(parser.parse_args()) + if args["op_types"] is None: + args["op_types"] = ["Conv", "Gemm", "MatMul"] + calibrate(**args) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index a0f1c63..b12159f 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1845,6 +1845,48 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. +### Maintained ONNX calibration reproduction + +`dev.benchmarks.onnx_calibration` now regenerates QDQ artifacts from explicit +ordered NPZ batches and a provenance manifest. It supports MinMax/Percentile +calibration, signed/unsigned activations, separate calibration/quantizer symmetry, +per-channel weights and floating/quantized biases. All calibration sessions use +explicit CPU thread limits from creation, and ORT's shape-inference sidecars stay +inside a private source snapshot in the output directory. See the +[manifest and commands](../dev/benchmarks/README.md#maintained-onnx-calibration-command). + +The local reproduction used ONNX 1.22.0 and ONNX Runtime 1.29.0, one CPU thread, +and the exact retained 128 Blair training samples in their original 16 batches +of eight. Inputs were regenerated with each floating checkpoint's preprocessing +at 128px. Neither validation nor test samples were used. Both heads used the +signed symmetric activation/per-channel weight, floating-bias TensorRT recipe +with asymmetric Percentile 99.9 histogram collection. + +| Head | Exactly equal tensor ranges | Exactly equal initializer arrays | Exactly equal graph nodes | +| --- | ---: | ---: | ---: | +| Flat | 339 | 1,374 | 1,366 | +| Hierarchical | 339 | 1,378 | 1,380 | + +Names, input/output definitions and node order also matched the earlier +`tmp-trt-probe/{head}-float-bias.onnx` candidates. Both new graphs passed ONNX +checking and finite-output CPU smoke execution. Snapshot/external-file paths +change serialization hashes, so this is graph/tensor reproduction, not a claim +of identical file bytes. No new engine timing or full-dataset quality measurement +was made for this command milestone; the preceding results remain attributed to +their original engines and runs. + +Input manifests and batches are retained under `tmp-onnx-calibration-inputs/`; +new models, caches, smoke outputs and reports are under +`tmp-onnx-calibration-{flat,hierarchical}/`. The maintained command validates a +declared calibration split, unique sample IDs, input hashes and dimensions; it +cannot independently prove that a caller's provenance excludes held-out data. +Dataset preparation, paired timing and full mini_metrics evaluation still need +to be connected into the maintained continuous pipeline. + +Validation passed static checks, all ten focused calibration tests, and the full +CPU-default suite: 416 passed, 149 skipped and the known EMA expected failure. +This milestone did not rerun GPU tests or establish new target-hardware results. + ### Maintained TensorRT engine reproduction `dev.benchmarks.tensorrt_build` replaces the one-off engine build/inspection probe diff --git a/docs/quantization-status.md b/docs/quantization-status.md index f67ad4d..3ee2c5e 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,7 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained build/inspection/smoke command | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained calibration, paired timing and dataset-quality commands | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration and build/inspection/smoke commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained paired timing and dataset-quality commands | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | @@ -45,8 +45,12 @@ production acceptance study. 1. **Make the latest experiments reproducible from a clean checkout.** Engine build, inspection and optional smoke-test parity now have a [maintained command](../dev/benchmarks/README.md#maintained-tensorrt-build-and-inspection-command). - Promote calibration preparation, paired inference and dataset metric evaluation - into maintained developer commands with explicit optional environments. + [Calibration](../dev/benchmarks/README.md#maintained-onnx-calibration-command) + also has a maintained command, verified to reproduce both candidates' ranges, + initializer arrays and graph nodes from the retained 128 training samples. + Promote paired inference and dataset metric evaluation into maintained developer + commands with explicit optional environments; connect dataset preparation to + the calibration manifest contract. Preserve calibration records, class/preprocessing contracts, hashes, failures and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` diff --git a/tests/test_benchmark_calibration.py b/tests/test_benchmark_calibration.py new file mode 100644 index 0000000..484c86c --- /dev/null +++ b/tests/test_benchmark_calibration.py @@ -0,0 +1,169 @@ +import hashlib +import json +from pathlib import Path + +import numpy as np +import pytest + +from dev.benchmarks.onnx_calibration import calibrate, calibration_manifest, load_batch + + +@pytest.fixture +def inputs(tmp_path): + entries = [] + for index, x in enumerate(([[0, 1], [1, 2]], [[-4, 10], [6, 20]])): + path = tmp_path / f"batch-{index}.npz" + np.savez(path, x=np.array(x, dtype=np.float32), offset=np.array([0.5, -0.5], dtype=np.float32)) + entries.append( + { + "path": path.name, + "sha256": hashlib.sha256(path.read_bytes()).hexdigest(), + "sample_ids": [f"train-{2 * index}", f"train-{2 * index + 1}"], + } + ) + metadata = { + "schema_version": 1, + "split": "train", + "batch_input": "x", + "provenance": {"dataset": "deterministic fixture", "preprocessing": "float features"}, + "batches": entries, + } + path = tmp_path / "calibration.json" + path.write_text(json.dumps(metadata)) + return path, metadata + + +@pytest.mark.parametrize("split", ["val", "test", ""]) +def test_calibration_rejects_declared_heldout_or_unknown_split(inputs, split): + path, metadata = inputs + metadata["split"] = split + path.write_text(json.dumps(metadata)) + with pytest.raises(ValueError, match="train/calibration"): + calibration_manifest(path) + + +def test_manifest_rejects_duplicate_sample_ids(inputs): + path, metadata = inputs + metadata["batches"][1]["sample_ids"][0] = metadata["batches"][0]["sample_ids"][0] + path.write_text(json.dumps(metadata)) + with pytest.raises(ValueError, match="unique"): + calibration_manifest(path) + + +def test_batch_identity_hash_and_dimensions_are_checked(inputs): + path, metadata = inputs + batch = metadata["batches"][0] + feeds, record = load_batch(path, metadata, batch) + assert record["sha256"] == batch["sha256"] + assert feeds["x"].shape == (2, 2) and feeds["offset"].shape == (2,) + with pytest.raises(ValueError, match="hash mismatch"): + load_batch(path, metadata, {**batch, "sha256": "wrong"}) + with pytest.raises(ValueError, match="leading dimension"): + load_batch(path, metadata, {**batch, "sample_ids": ["only-one"]}) + + +@pytest.fixture +def model(tmp_path): + onnx = pytest.importorskip("onnx") + graph = onnx.helper.make_graph( + [ + onnx.helper.make_node("Gemm", ["x", "weight", "bias"], ["product"]), + onnx.helper.make_node("Add", ["product", "offset"], ["scores"]), + ], + "linear", + [ + onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, ["batch", 2]), + onnx.helper.make_tensor_value_info("offset", onnx.TensorProto.FLOAT, [2]), + ], + [onnx.helper.make_tensor_value_info("scores", onnx.TensorProto.FLOAT, ["batch", 2])], + [ + onnx.numpy_helper.from_array(np.eye(2, dtype=np.float32), "weight"), + onnx.numpy_helper.from_array(np.array([0.25, -0.25], dtype=np.float32), "bias"), + ], + ) + parent = tmp_path / "source" + parent.mkdir() + path = parent / "model.onnx" + onnx.save_model( + onnx.helper.make_model(graph, opset_imports=[onnx.helper.make_opsetid("", 18)], ir_version=10), + path, + save_as_external_data=True, + all_tensors_to_one_file=True, + location="weights.data", + size_threshold=0, + ) + return path + + +@pytest.mark.parametrize("method", ["minmax", "percentile"]) +@pytest.mark.parametrize("signed", [False, True]) +def test_calibration_recipes_keep_all_batches_source_and_thread_limits(model, inputs, tmp_path, monkeypatch, method, signed): + onnx = pytest.importorskip("onnx") + ort = pytest.importorskip("onnxruntime") + from onnxruntime.quantization.calibrate import load_tensors_data + + manifest, metadata = inputs + output = tmp_path / "output" + original = {p.name: p.read_bytes() for p in model.parent.iterdir()} + sessions, inference_paths = [], [] + real_session, real_infer = ort.InferenceSession, onnx.shape_inference.infer_shapes_path + + def session(*args, **kwargs): + options = kwargs["sess_options"] + sessions.append((options.intra_op_num_threads, options.inter_op_num_threads)) + return real_session(*args, **kwargs) + + def infer(source, target, *args, **kwargs): + inference_paths.append(Path(target)) + return real_infer(source, target, *args, **kwargs) + + monkeypatch.setattr(ort, "InferenceSession", session) + monkeypatch.setattr(onnx.shape_inference, "infer_shapes_path", infer) + report = calibrate( + model, + manifest, + output, + method=method, + activation_type="int8" if signed else "uint8", + symmetric_activations=signed, + float_bias=signed, + threads=2, + ) + assert report["status"] == "passed" + assert report == json.loads((output / "report.json").read_text()) + assert sessions == [(2, 1), (2, 1)] + assert inference_paths and all(output in p.parents for p in inference_paths) + assert {p.name: p.read_bytes() for p in model.parent.iterdir()} == original + assert [b["sample_ids"] for b in report["batches"]] == [b["sample_ids"] for b in metadata["batches"]] + assert report["output_ops"]["QuantizeLinear"] > 0 + if method == "minmax": + low, high = load_tensors_data(output / "ranges.json")["x"].range_value + np.testing.assert_array_equal(low, np.array(-4, dtype=np.float32)) + np.testing.assert_array_equal(high, np.array(20, dtype=np.float32)) + graph = onnx.load(output / "model.onnx") + initializers = {x.name: x for x in graph.graph.initializer} + assert initializers["x_zero_point"].data_type == (onnx.TensorProto.INT8 if signed else onnx.TensorProto.UINT8) + if signed: + assert not onnx.numpy_helper.to_array(initializers["x_zero_point"]).any() + assert initializers["bias"].data_type == onnx.TensorProto.FLOAT + else: + assert initializers["bias_quantized"].data_type == onnx.TensorProto.INT32 + with np.load(output / "calibration-smoke.npz") as smoke: + assert np.isfinite(smoke["scores"]).all() + assert smoke["scores"].shape == (2, 2) + with pytest.raises(FileExistsError): + calibrate(model, manifest, output) + + +def test_calibration_batch_failure_retains_failed_report(model, inputs, tmp_path): + pytest.importorskip("onnxruntime") + path, metadata = inputs + metadata["batches"][1]["sha256"] = "changed" + path.write_text(json.dumps(metadata)) + output = tmp_path / "failed" + with pytest.raises(ValueError, match="hash mismatch"): + calibrate(model, path, output, method="minmax") + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" + assert len(report["batches"]) == 1 + assert not (output / "model.onnx").exists() From 810471eefb82da3c7306bb2bce261e8729042322 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 14:35:14 +0200 Subject: [PATCH 063/155] feat: add paired TensorRT latency benchmark with retained trials --- dev/benchmarks/README.md | 63 ++++++++ dev/benchmarks/tensorrt_pair.py | 213 ++++++++++++++++++++++++++ docs/benchmarks.md | 40 +++++ docs/quantization-status.md | 10 +- tests/test_benchmark_tensorrt_pair.py | 109 +++++++++++++ 5 files changed, 431 insertions(+), 4 deletions(-) create mode 100644 dev/benchmarks/tensorrt_pair.py create mode 100644 tests/test_benchmark_tensorrt_pair.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index e358aa8..9b01b65 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -827,3 +827,66 @@ environment, set `CUDA_VISIBLE_DEVICES` and `RUN_CUDA_TESTS=1`, then run `bash dev/check.sh test tests/test_benchmark_tensorrt.py` to include real engine construction, successful reference matching, retained mismatches and parser failure diagnostics. + +### Maintained paired TensorRT timing command + +`dev.benchmarks.tensorrt_pair` compares two existing engines on the same named +NPZ input arrays. It has no model/head or batch-size assumptions; prepare a +separate NPZ for each workload shape, including all static inputs. Both engines +must accept the same input dtypes/shapes in profile 0 and produce the same output +names/shapes. Output precisions may differ and are recorded. It uses the explicit +TensorRT/CUDA PyTorch environment from the build command and installs nothing. + +```bash +OMP_NUM_THREADS=1 python -m dev.benchmarks.tensorrt_pair \ + --baseline fp16/model.engine --candidate int8/model.engine \ + --inputs preprocessed-batch.npz --output /tmp/pair-1 +OMP_NUM_THREADS=1 python -m dev.benchmarks.tensorrt_pair \ + --baseline fp16/model.engine --candidate int8/model.engine \ + --inputs preprocessed-batch.npz --output /tmp/pair-2 --reverse +OMP_NUM_THREADS=1 python -m dev.benchmarks.tensorrt_pair \ + --baseline fp16/model.engine --candidate int8/model.engine \ + --inputs preprocessed-batch.npz --output /tmp/pair-3 +``` + +Run processes sequentially, with no concurrent benchmarks, tests or training. +Record power/thermal settings and concurrent system load alongside the reports. +Use matched builder settings and practical baselines for the target hardware; +these commands do not infer whether an engine uses INT8, FP16 or another precision. +Rebuild engines for each destination device. Repeat batch sizes/resolutions and +10k/100k-class local head workloads explicitly; reserve million-class full models +for suitably sized systems. + +Defaults are 10 warmup pairs and 31 measured pairs. `--warmup`, `--repeats`, +`--reverse` and `--device` are explicit controls. Each pair runs baseline and +candidate adjacently; their order alternates, and `--reverse` reverses that order. +The summary reports the median of **paired candidate/baseline ratios**, where +less than one means lower candidate latency. Retain per-process results rather +than treating repetitions from one process as independent device replications. + +Host wall time covers input copies to GPU, execution, output copies to CPU and +stream synchronization, with preallocated buffers and a nondefault CUDA stream. +Default host buffers are pageable; `--pinned` uses pinned buffers and nonblocking +copies, with the same final synchronization. Compare these as separate IO regimes. +No CUDA-event timing is used: earlier WSL event readings were inconsistent with +enclosing host durations. This command excludes preprocessing, allocations, +shape changes, engine loading and construction from steady-state measurements. +It does not use CUDA graph capture. File reading and deserialization/context +creation have separate observations, but these are not whole-process cold-start +measurements and may benefit from caches. + +Both engines, contexts and IO buffers coexist. Engine file sizes and reported +context-memory requirements are recorded, **not total runtime memory savings**. +Profile shape tensors, nonlinear/host engine IO and unresolved/empty outputs are +explicitly unsupported by this probe. This restriction concerns engine bindings; +the command's host transfer buffers are separate from those device bindings. + +A new directory receives `report.json`, raw paired host durations, engine/input +hashes, versions, GPU/thread information and final named output NPZs. Completed +pairs and diagnostics survive later failures; existing output directories are +refused. Final outputs must be finite, but this is not quality or numerical-parity +acceptance. Use the separate full-dataset mini_metrics evaluation and engine +inspection to establish the quality/efficiency trade-off. CPU tests cover pairing, +summary arithmetic and invalid clocks; optional GPU tests cover actual multi-input +execution, both transfer modes and retained failures in +`tests/test_benchmark_tensorrt_pair.py`. diff --git a/dev/benchmarks/tensorrt_pair.py b/dev/benchmarks/tensorrt_pair.py new file mode 100644 index 0000000..7748c61 --- /dev/null +++ b/dev/benchmarks/tensorrt_pair.py @@ -0,0 +1,213 @@ +"""Measure paired TensorRT engine latency with explicit inputs and retained trials.""" + +import hashlib +import io +import json +import math +import platform +import statistics +import time +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np + +from .onnx_inference import file_hash + + +def paired_trials(execute, warmup, repeats, reverse=False): + """Alternate adjacent measurements; keep each ratio paired within its trial.""" + if warmup < 0 or repeats < 1: + raise ValueError("Require nonnegative warmup and positive repeats") + for index in range(warmup + repeats): + order = ["baseline", "candidate"] + if bool(index % 2) ^ reverse: + order.reverse() + seconds = {name: execute(name) for name in order} + if any(not math.isfinite(value) or value <= 0 for value in seconds.values()): + raise ValueError("Measured durations must be finite and positive") + if index >= warmup: + yield {"index": index - warmup, "order": order, "seconds": seconds} + + +def summarize(trials): + ratios = [trial["seconds"]["candidate"] / trial["seconds"]["baseline"] for trial in trials] + return { + "median_seconds": {name: statistics.median(trial["seconds"][name] for trial in trials) for name in ("baseline", "candidate")}, + "candidate_over_baseline_ratios": ratios, + "median_paired_ratio": statistics.median(ratios), + } + + +def buffers_for(engine, context, feeds, torch, trt, device, pinned): + names = [engine.get_tensor_name(i) for i in range(engine.num_io_tensors)] + inputs = {name for name in names if engine.get_tensor_mode(name) == trt.TensorIOMode.INPUT} + if inputs != set(feeds): + raise ValueError("Input names do not match engine") + for name in names: + if engine.get_tensor_format(name) != trt.TensorFormat.LINEAR or engine.get_tensor_location(name) != trt.TensorLocation.DEVICE: + raise ValueError(f"Require linear device IO: {name}") + for name, value in feeds.items(): + if engine.is_shape_inference_io(name): + raise ValueError(f"Shape-tensor inputs are not supported by this probe: {name}") + if np.dtype(trt.nptype(engine.get_tensor_dtype(name))) != value.dtype: + raise ValueError(f"Input dtype does not match engine: {name}") + if not context.set_input_shape(name, value.shape): + raise ValueError(f"Input shape is outside engine profile 0: {name}") + unresolved = context.infer_shapes() + if unresolved: + raise ValueError(f"Unresolved engine shapes: {unresolved}") + buffers = {} + for name in names: + shape = tuple(context.get_tensor_shape(name)) + if any(n <= 0 for n in shape): + raise ValueError(f"Unresolved or empty tensor: {name}") + dtype = torch.from_numpy(np.empty((), dtype=trt.nptype(engine.get_tensor_dtype(name)))).dtype + host = torch.empty(shape, dtype=dtype, pin_memory=pinned) + if name in inputs: + host.copy_(torch.from_numpy(feeds[name])) + gpu = torch.empty(shape, dtype=dtype, device=f"cuda:{device}") + if not context.set_tensor_address(name, gpu.data_ptr()): + raise RuntimeError(f"TensorRT rejected IO address: {name}") + buffers[name] = (gpu, host) + return buffers, tuple(sorted(inputs)) + + +def benchmark(baseline, candidate, inputs, output, warmup=10, repeats=31, reverse=False, device=0, pinned=False): + if warmup < 0 or repeats < 1 or device < 0: + raise ValueError("Require nonnegative warmup/device and positive repeats") + payload = Path(inputs).read_bytes() + input_hash = hashlib.sha256(payload).hexdigest() + with np.load(io.BytesIO(payload), allow_pickle=False) as data: + feeds = {name: data[name].copy(order="C") for name in data.files} + del payload + if not feeds or any(not np.isfinite(a).all() for a in feeds.values()): + raise ValueError("Supply finite named input arrays") + try: + import tensorrt as trt + import torch + except ImportError as error: + raise ImportError("Use an explicitly prepared TensorRT and CUDA PyTorch environment; this command installs nothing") from error + + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "versions": {"tensorrt": trt.__version__, "torch": torch.__version__, "numpy": np.__version__}, + "environment": {"platform": platform.platform(), "python": platform.python_version()}, + "settings": {"warmup": warmup, "repeats": repeats, "reverse": reverse, "device": device, "pinned_host_io": pinned, "profile": 0}, + "inputs": {"path": str(inputs), "sha256": input_hash}, + "models": {}, + "messages": [], + "scope": ( + "Synchronized host latency: preallocated host-to-device inputs, engine execution and device-to-host outputs. " + "Excludes decoding/preprocessing, allocation, deserialization and shape changes. Both engines/contexts coexist. " + "No CUDA-event timing or CUDA graph capture. Not a quality, integer-placement or total-memory acceptance test." + ), + } + + class Logger(trt.ILogger): + def log(self, severity, message): + if severity <= trt.ILogger.WARNING: + report["messages"].append({"severity": str(severity), "message": message}) + + try: + with torch.cuda.device(device): + report["environment"].update( + gpu=torch.cuda.get_device_name(device), compute_capability=list(torch.cuda.get_device_capability(device)) + ) + report["environment"].update(torch_threads=torch.get_num_threads(), torch_interop_threads=torch.get_num_interop_threads()) + logger = Logger() + trt.init_libnvinfer_plugins(logger, "") + runtime = trt.Runtime(logger) + stream = torch.cuda.Stream(device=device) + models = {} + for name, path in (("baseline", baseline), ("candidate", candidate)): + before = time.perf_counter() + serialized = Path(path).read_bytes() + read_seconds = time.perf_counter() - before + info = { + "path": str(path), + "sha256": hashlib.sha256(serialized).hexdigest(), + "bytes": len(serialized), + "read_seconds": read_seconds, + } + report["models"][name] = info + torch.cuda.synchronize(device) + before = time.perf_counter() + engine = runtime.deserialize_cuda_engine(serialized) + if engine is None: + raise RuntimeError(f"Could not deserialize {name} engine") + context = engine.create_execution_context() + if context is None: + raise RuntimeError(f"Could not create {name} execution context") + torch.cuda.synchronize(device) + info["deserialize_and_context_seconds"] = time.perf_counter() - before + del serialized + with torch.cuda.stream(stream): + buffers, input_names = buffers_for(engine, context, feeds, torch, trt, device, pinned) + info["context_memory_bytes"] = engine.device_memory_size_v2 + info["io"] = { + key: {"shape": list(host.shape), "dtype": str(host.numpy().dtype), "input": key in input_names} + for key, (_, host) in buffers.items() + } + models[name] = engine, context, buffers, input_names + # Outputs can use different precisions but must describe the same named shapes. + contracts = [ + {key: value["shape"] for key, value in info["io"].items() if not value["input"]} for info in report["models"].values() + ] + if not contracts[0] or contracts[0] != contracts[1]: + raise ValueError("Engine output names/shapes must match") + stream.synchronize() + + def execute(name): + _, context, buffers, input_names = models[name] + with torch.cuda.stream(stream): + before = time.perf_counter() + for key in input_names: + gpu, host = buffers[key] + gpu.copy_(host, non_blocking=pinned) + if not context.execute_async_v3(stream.cuda_stream): + raise RuntimeError(f"Execution failed: {name}") + for key, (gpu, host) in buffers.items(): + if key not in input_names: + host.copy_(gpu, non_blocking=pinned) + stream.synchronize() + return time.perf_counter() - before + + report["trials"] = [] + report["trials"].extend(paired_trials(execute, warmup, repeats, reverse)) + report["summary"] = summarize(report["trials"]) + for name, (_, _, buffers, input_names) in models.items(): + arrays = {key: host.numpy() for key, (_, host) in buffers.items() if key not in input_names} + np.savez(output / f"{name}-outputs.npz", **arrays) + report["models"][name]["outputs_sha256"] = file_hash(output / f"{name}-outputs.npz") + if any(not np.isfinite(a).all() for a in arrays.values()): + raise ValueError(f"Nonfinite final outputs: {name}") + report["status"] = "passed" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--baseline", type=Path, required=True) + parser.add_argument("--candidate", type=Path, required=True) + parser.add_argument("--inputs", type=Path, required=True, help="Named preprocessed NPZ inputs with exact shapes and dtypes") + parser.add_argument("--output", type=Path, required=True, help="New report/output directory") + parser.add_argument("--warmup", type=int, default=10) + parser.add_argument("--repeats", type=int, default=31) + parser.add_argument("--reverse", action="store_true", help="Reverse the alternating execution order") + parser.add_argument("--device", type=int, default=0) + parser.add_argument("--pinned", action="store_true", help="Use preallocated pinned host IO and nonblocking copies") + benchmark(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index b12159f..d357e37 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1845,6 +1845,46 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. +### Maintained paired TensorRT timing reproduction + +`dev.benchmarks.tensorrt_pair` now measures arbitrary existing engine pairs with +named preprocessed inputs, alternating adjacent execution order, raw host durations +and median paired candidate/baseline ratios. It supports pageable or pinned host +IO, records hashes and final outputs, and retains completed pairs on later errors. +It deliberately omits CUDA-event timing. See the +[commands and measurement scope](../dev/benchmarks/README.md#maintained-paired-tensorrt-timing-command). + +The local validation reused the retained FP16 and signed-QDQ INT8 engines on the +RTX 3080 Ti Laptop, TensorRT 10.16.1.11 and PyTorch 2.12.0+cu130. Each head used +three fresh sequential processes, reversing execution order in the second run, +with 10 warmup pairs and 31 measured pairs per process. Inputs were the same eight +128px images, with pageable preallocated host IO. Both engines/contexts coexisted; +the CPU and GPU test suites had finished before timing began. + +| Head | Run 1 median paired INT8/FP16 ratio | Run 2 | Run 3 | +| --- | ---: | ---: | ---: | +| Flat | 1.270 | 1.247 | 1.169 | +| Hierarchical | 1.247 | 1.210 | 1.189 | + +Every ratio exceeded one: these local runs again found no INT8 host-latency +advantage. These are transfer/execution/synchronization observations, not +device-only time or evidence about the intended target GPUs. Unlike the earlier +scratch runner, this command inserts no timing events, so the measurement code +is not identical. Do not interpret differences from the earlier table as model +performance regressions. No new full-dataset quality measurement was made. + +All final named outputs from all six processes matched the respective retained +FP16/INT8 engine smoke references exactly. Raw trials, hashes, loading observations +and outputs are in `tmp-trt-probe/pair-maintained-{flat,hierarchical}-{1,2,3}/`; +the collected summaries are in `tmp-trt-probe/pair-maintained-summary.json`. +No total runtime-memory reduction was established. + +Static checks and the full CPU-default suite passed: 425 passed, 151 skipped, +and the known EMA expected failure. All 11 focused tests passed in the prepared +GPU environment, including actual multiple-input execution with pageable and +pinned buffers and retained deserialization failures. Pinned IO correctness was +tested; the real-model timing table above uses pageable IO only. + ### Maintained ONNX calibration reproduction `dev.benchmarks.onnx_calibration` now regenerates QDQ artifacts from explicit diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 3ee2c5e..a7e58fb 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,7 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration and build/inspection/smoke commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained paired timing and dataset-quality commands | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration, build/inspection/smoke and paired timing commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained dataset-quality command | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | @@ -48,9 +48,11 @@ production acceptance study. [Calibration](../dev/benchmarks/README.md#maintained-onnx-calibration-command) also has a maintained command, verified to reproduce both candidates' ranges, initializer arrays and graph nodes from the retained 128 training samples. - Promote paired inference and dataset metric evaluation into maintained developer - commands with explicit optional environments; connect dataset preparation to - the calibration manifest contract. + [Paired inference timing](../dev/benchmarks/README.md#maintained-paired-tensorrt-timing-command) + now retains alternating host-time pairs in a maintained command, with explicit + pageable/pinned IO regimes. Promote dataset metric evaluation into a maintained + developer command with an explicit optional environment; connect dataset + preparation to the calibration manifest contract. Preserve calibration records, class/preprocessing contracts, hashes, failures and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` diff --git a/tests/test_benchmark_tensorrt_pair.py b/tests/test_benchmark_tensorrt_pair.py new file mode 100644 index 0000000..6a55e78 --- /dev/null +++ b/tests/test_benchmark_tensorrt_pair.py @@ -0,0 +1,109 @@ +import json +import os +import subprocess +import sys + +import numpy as np +import pytest + +from dev.benchmarks.tensorrt_pair import benchmark, paired_trials, summarize + + +@pytest.mark.parametrize("reverse", [False, True]) +def test_measurements_alternate_and_exclude_warmup(reverse): + calls = [] + + def execute(name): + calls.append(name) + return float(len(calls)) + + trials = list(paired_trials(execute, 1, 3, reverse)) + assert len(calls) == 8 and len(trials) == 3 + expected = ["baseline", "candidate", "candidate", "baseline"] * 2 + assert calls == (["candidate" if x == "baseline" else "baseline" for x in expected] if reverse else expected) + assert [t["index"] for t in trials] == [0, 1, 2] + assert min(trials[0]["seconds"].values()) == 3 + + +def test_summary_uses_paired_ratios_not_ratio_of_medians(): + trials = [{"seconds": {"baseline": a, "candidate": b}} for a, b in [(1, 2), (2, 200), (100, 100)]] + result = summarize(trials) + assert result["median_paired_ratio"] == 2 + assert result["candidate_over_baseline_ratios"] == [2, 100, 1] + assert result["median_seconds"] == {"baseline": 2, "candidate": 100} + + +@pytest.mark.parametrize("duration", [0, -1, float("nan"), float("inf")]) +def test_invalid_clock_readings_are_rejected(duration): + with pytest.raises(ValueError, match="finite and positive"): + list(paired_trials(lambda _: duration, 0, 1)) + + +def test_complete_trials_survive_later_execution_failure(): + trials = [] + times = iter([1, 2, 3]) + + def execute(_): + return next(times, 0) + + with pytest.raises(ValueError): + trials.extend(paired_trials(execute, 0, 2)) + assert len(trials) == 1 + + +def test_help_does_not_load_gpu_libraries(): + code = """ +import runpy, sys +sys.argv = ['tensorrt_pair', '--help'] +try: + runpy.run_module('dev.benchmarks.tensorrt_pair', run_name='__main__') +except SystemExit as error: + assert error.code == 0 +assert 'tensorrt' not in sys.modules and 'torch' not in sys.modules +""" + subprocess.run([sys.executable, "-c", code], check=True, capture_output=True) + + +@pytest.mark.parametrize("pinned", [False, True]) +def test_real_engine_pair_preserves_named_outputs_and_failures(tmp_path, pinned): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 in an explicitly prepared GPU environment") + pytest.importorskip("tensorrt") + import torch + + assert torch.cuda.is_available(), "CUDA requested but unavailable" + onnx = pytest.importorskip("onnx") + from dev.benchmarks.tensorrt_build import build + + graph = onnx.helper.make_graph( + [onnx.helper.make_node("Add", ["x", "offset"], ["scores"])], + "two-input", + [ + onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, [2, 2]), + onnx.helper.make_tensor_value_info("offset", onnx.TensorProto.FLOAT, [2]), + ], + [onnx.helper.make_tensor_value_info("scores", onnx.TensorProto.FLOAT, [2, 2])], + ) + model, inputs = tmp_path / "model.onnx", tmp_path / "inputs.npz" + onnx.save(onnx.helper.make_model(graph, opset_imports=[onnx.helper.make_opsetid("", 18)], ir_version=10), model) + x, offset = np.arange(4, dtype=np.float32).reshape(2, 2), np.array([0.5, -0.5], dtype=np.float32) + np.savez(inputs, x=x, offset=offset) + build(model, inputs, tmp_path / "build", optimization=0) + engine = tmp_path / "build/model.engine" + output = tmp_path / "pair" + report = benchmark(engine, engine, inputs, output, warmup=1, repeats=3, pinned=pinned) + assert report["status"] == "passed" and len(report["trials"]) == 3 + assert report == json.loads((output / "report.json").read_text()) + assert report["models"]["baseline"]["sha256"] == report["models"]["candidate"]["sha256"] + for mode in ("baseline", "candidate"): + with np.load(output / f"{mode}-outputs.npz") as result: + np.testing.assert_array_equal(result["scores"], x + offset) + with pytest.raises(FileExistsError): + benchmark(engine, engine, inputs, output) + bad = tmp_path / "bad.engine" + bad.write_bytes(b"not an engine") + failed = tmp_path / "failure" + with pytest.raises(RuntimeError, match="deserialize candidate"): + benchmark(engine, bad, inputs, failed, warmup=0, repeats=1) + retained = json.loads((failed / "report.json").read_text()) + assert retained["status"] == "failed" and retained["messages"] From e3a24f0520e4e6843e459271548b1c0c3396979e Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 14:49:30 +0200 Subject: [PATCH 064/155] feat: validate paired prediction quality with mini_metrics --- dev/benchmarks/README.md | 85 ++++++++++++ dev/benchmarks/quality_compare.py | 221 ++++++++++++++++++++++++++++++ docs/benchmarks.md | 47 +++++++ docs/quantization-status.md | 11 +- tests/test_benchmark_quality.py | 152 ++++++++++++++++++++ 5 files changed, 512 insertions(+), 4 deletions(-) create mode 100644 dev/benchmarks/quality_compare.py create mode 100644 tests/test_benchmark_quality.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 9b01b65..0e8bff5 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -890,3 +890,88 @@ inspection to establish the quality/efficiency trade-off. CPU tests cover pairin summary arithmetic and invalid clocks; optional GPU tests cover actual multi-input execution, both transfer modes and retained failures in `tests/test_benchmark_tensorrt_pair.py`. + +### Maintained paired quality comparison + +`dev.benchmarks.quality_compare` evaluates baseline and candidate prediction CSVs +using mini_metrics Macro-F1, Macro-Recall, Macro-Precision, Coverage and Theil's U +independently at each level. It uses an explicit held-out manifest to check sample +identity, labels, complete level coverage and declared class mappings. It does +not run model inference; connect the CSVs produced by a dataset inference pass +to this contract. This format avoids requiring complete score matrices merely +to compare classification metrics for large heads. + +Prepare mini_metrics explicitly in the evaluation environment. The command does +not install it or assume a sibling checkout. It records the imported package path +and Python source hashes separately from installed distribution metadata: an +explicit source checkout on `PYTHONPATH` can differ from the installed release. + +```bash +PYTHONHASHSEED=0 OMP_NUM_THREADS=1 python -m dev.benchmarks.quality_compare \ + --manifest heldout-comparison.json --output /tmp/quality-1 +``` + +The CSV columns are exactly `instance_id,filename,level,label,prediction,confidence,threshold`. +IDs are nonnegative INT64 integers; levels are zero-based in manifest order. +Class names remain literal strings, including numeric-looking names such as +`001`. Confidence must be finite in [0,1] and threshold must be zero. Rows may +arrive in any order; duplicate/missing samples or levels, changed filenames/labels, +unknown predictions and hash mismatches are rejected. Optional derived metric +columns are deliberately excluded so they cannot override the evaluated labels +or prediction policy. + +A minimal manifest has this structure (include every evaluated sample): + +```json +{ + "schema_version": 1, + "split": "val", + "provenance": {"dataset": "manifest hash and split selection", "preprocessing": "exact inference transforms"}, + "levels": [{"name": "leaf", "classes": ["cat", "dog"]}], + "samples": [ + {"instance_id": 0, "filename": "cat.jpg", "labels": ["cat"]}, + {"instance_id": 1, "filename": "dog.jpg", "labels": ["dog"]} + ], + "baseline": { + "path": "baseline.csv", "classes": [["cat", "dog"]], + "provenance": {"model": "checkpoint/export/engine hashes", "preprocessing": "baseline transforms"} + }, + "candidate": { + "path": "candidate.csv", "classes": [["cat", "dog"]], + "provenance": {"model": "checkpoint/export/engine hashes", "preprocessing": "candidate transforms"} + } +} +``` + +Paths are relative to the manifest. Baseline/candidate may each include `sha256` +to require exact input bytes. Each artifact's `classes` must match the ordered +level mappings used to translate its score columns to labels. These are explicit +declarations; the evaluator cannot prove how an external producer translated +scores or whether a declared `val`/`test` split is independent of training and +calibration. Review the dataset, preprocessing and model provenance. Hierarchical +models add levels and one label per level to each sample; this command computes +ordinary per-level metrics, not hierarchical path metrics. + +Policy is fixed: `optimal=False`, threshold zero, no abstention, `known_only=False`, +`per_class=False`, and an explicit `opt_crit=MacroF1`. No evaluation subset is +removed for threshold tuning. Coverage is therefore one for these complete, +known-label predictions; abstention/unknown-label policies need separate studies. +The package's own observed-class and undefined-value semantics are preserved; +declaring a class does not force it into mini_metrics' macro denominator. + +The report retains both sets of metrics, candidate-minus-baseline differences in +their original units, prediction-change counts, provenance and canonical CSVs. +Multiply differences by 100 to express percentage-point changes where appropriate. +Undefined values become explicit JSON `null` entries and appear in +`undefined_metrics`; they are not silently replaced by zero. `status=evaluated` +means calculation completed, not that production quality is acceptable. Failures +after output creation retain completed results; existing directories are refused. + +The tested mini_metrics F1 implementation iterates an unordered class set, so +last-bit summation differences can occur across Python hash seeds. Set +`PYTHONHASHSEED` before starting Python for repeatable runs, retain its recorded +value, and use a numerical tolerance when comparing independently recorded +floating results. Do not alter metric definitions to force exact historical +bytes. Correctness tests cover synthetic oracle metrics, literal class identities, +reordered predictions, invalid contracts and undefined Theil's U; tests requiring +mini_metrics skip explicitly when that optional evaluation package is absent. diff --git a/dev/benchmarks/quality_compare.py b/dev/benchmarks/quality_compare.py new file mode 100644 index 0000000..28d0c4b --- /dev/null +++ b/dev/benchmarks/quality_compare.py @@ -0,0 +1,221 @@ +"""Compare paired prediction tables with explicit held-out identity and mini_metrics.""" + +import csv +import hashlib +import importlib.metadata +import io +import json +import math +import os +import platform +from argparse import ArgumentParser +from pathlib import Path + +from .onnx_inference import file_hash + +METRICS = ("f1", "recall", "precision", "coverage", "theilU") +COLUMNS = ("instance_id", "filename", "level", "label", "prediction", "confidence", "threshold") + + +def read_manifest(path): + payload = Path(path).read_bytes() + manifest = json.loads(payload) + if manifest.get("schema_version") != 1 or manifest.get("split") not in ("val", "test"): + raise ValueError("Require schema_version=1 and a declared val/test split") + if not isinstance(manifest.get("provenance"), dict) or not manifest["provenance"]: + raise ValueError("Record dataset and preprocessing provenance") + levels = manifest.get("levels") + if not isinstance(levels, list) or not levels: + raise ValueError("Supply ordered level names and class mappings") + names = set() + for level in levels: + name, classes = level.get("name"), level.get("classes") + if not isinstance(name, str) or not name or name in names: + raise ValueError("Level names must be nonempty and unique") + names.add(name) + if ( + not isinstance(classes, list) + or not classes + or any(not isinstance(c, str) or not c for c in classes) + or len(set(classes)) != len(classes) + ): + raise ValueError("Each level requires unique nonempty class names in score-column order") + samples = manifest.get("samples") + if not isinstance(samples, list) or not samples: + raise ValueError("Declare the complete held-out sample set") + ids = set() + class_sets = [set(level["classes"]) for level in levels] + for sample in samples: + identifier = sample.get("instance_id") + if type(identifier) is not int or not 0 <= identifier < 2**63 or identifier in ids: + raise ValueError("Sample IDs must be unique nonnegative INT64 integers") + ids.add(identifier) + if not isinstance(sample.get("filename"), str) or not sample["filename"]: + raise ValueError("Each sample needs its filename/identity") + labels = sample.get("labels") + if not isinstance(labels, list) or len(labels) != len(levels): + raise ValueError("Each sample needs one label per level") + if any(label not in classes for label, classes in zip(labels, class_sets, strict=True)): + raise ValueError("Sample labels must belong to the declared classes") + for mode in ("baseline", "candidate"): + artifact = manifest.get(mode, {}) + if not isinstance(artifact.get("path"), str) or not artifact["path"] or not artifact.get("provenance"): + raise ValueError(f"Declare prediction path and model/preprocessing provenance for {mode}") + if artifact.get("classes") != [level["classes"] for level in levels]: + raise ValueError(f"{mode} class mappings must match the ordered level mappings") + return manifest, hashlib.sha256(payload).hexdigest() + + +def read_predictions(path, artifact, manifest): + payload = Path(path).read_bytes() + digest = hashlib.sha256(payload).hexdigest() + if artifact.get("sha256") is not None and artifact["sha256"] != digest: + raise ValueError(f"Prediction hash mismatch: {path}") + reader = csv.DictReader(io.StringIO(payload.decode("utf-8-sig"))) + if reader.fieldnames is None or set(reader.fieldnames) != set(COLUMNS) or len(reader.fieldnames) != len(COLUMNS): + raise ValueError("Prediction CSV must contain exactly the documented seven columns") + samples = {sample["instance_id"]: sample for sample in manifest["samples"]} + classes = [set(level["classes"]) for level in manifest["levels"]] + rows = {} + for row in reader: + if None in row or any(value is None for value in row.values()): + raise ValueError("Malformed prediction CSV row") + identifier, level = int(row["instance_id"]), int(row["level"]) + key = identifier, level + if identifier not in samples or not 0 <= level < len(classes) or key in rows: + raise ValueError("Unknown or duplicate sample/level in predictions") + sample = samples[identifier] + if row["filename"] != sample["filename"] or row["label"] != sample["labels"][level]: + raise ValueError("Prediction filename/label does not match held-out manifest") + if row["prediction"] not in classes[level]: + raise ValueError("Prediction is outside declared classes") + confidence, threshold = float(row["confidence"]), float(row["threshold"]) + if not math.isfinite(confidence) or not 0 <= confidence <= 1 or threshold != 0: + raise ValueError("Require confidence in [0,1] and threshold zero; this comparison does not tune or abstain") + rows[key] = {**row, "instance_id": identifier, "level": level, "confidence": confidence, "threshold": 0.0} + if len(rows) != len(samples) * len(classes): + raise ValueError("Predictions do not cover every held-out sample at every level") + # Canonicalize rows so CSV order cannot silently change the comparison. + ordered = [rows[(sample["instance_id"], level)] for level in range(len(classes)) for sample in manifest["samples"]] + return {column: [row[column] for row in ordered] for column in COLUMNS}, digest + + +def compare(manifest, output): + metadata, digest = read_manifest(manifest) + try: + import mini_metrics + from mini_metrics.data import MetricDF + from mini_metrics.metrics import MacroF1, evaluate_file + except ImportError as error: + raise ImportError("Prepare mini_metrics explicitly in the evaluation environment; this command installs nothing") from error + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "environment": { + "python": platform.python_version(), + "python_hash_seed": os.environ.get("PYTHONHASHSEED"), + }, + "manifest": {"path": str(manifest), "sha256": digest, "contents": metadata}, + "policy": { + "threshold": 0, + "optimal": False, + "known_only": False, + "per_class": False, + "criterion": "MacroF1", + "hierarchical_metrics": False, + }, + "scope": ( + "Fixed predictions evaluated independently at each declared level; " + "not threshold tuning, score parity, inference performance or production acceptance." + ), + "models": {}, + "undefined_metrics": [], + } + try: + root = Path(mini_metrics.__file__).parent + try: + version = importlib.metadata.version("mini_metrics") + except importlib.metadata.PackageNotFoundError: + version = None + report["mini_metrics"] = { + "installed_distribution_version": version, + "imported_package": str(root), + "source_hashes": {str(path.relative_to(root)): file_hash(path) for path in sorted(root.rglob("*.py"))}, + } + tables = {} + for mode in ("baseline", "candidate"): + artifact = metadata[mode] + path = Path(manifest).parent / artifact["path"] + table, source_hash = read_predictions(path, artifact, metadata) + tables[mode] = table + info = {"source_sha256": source_hash, "source_path": str(path)} + report["models"][mode] = info + normalized = output / f"{mode}.csv" + with normalized.open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=COLUMNS) + writer.writeheader() + writer.writerows(dict(zip(COLUMNS, row, strict=True)) for row in zip(*(table[c] for c in COLUMNS), strict=True)) + info["normalized_csv_sha256"] = file_hash(normalized) + # Construct directly to preserve literal string labels such as "001". + result = evaluate_file( + MetricDF(table), + optimal=False, + threshold=0, + known_only=False, + per_class=False, + simple=True, + hierarchical=False, + pattern=r"^(f1|recall|precision|coverage|theilU)$", + opt_crit=MacroF1, + verbose=0, + ) + if set(result) != set(METRICS): + raise ValueError("mini_metrics did not return exactly the requested five metrics") + info["metrics"] = {} + for metric in METRICS: + values = {int(level): float(value) for level, value in result[metric].items()} + if set(values) != set(range(len(metadata["levels"]))): + raise ValueError("mini_metrics returned unexpected levels") + info["metrics"][metric] = {str(level): value if math.isfinite(value) else None for level, value in values.items()} + report["undefined_metrics"].extend( + {"model": mode, "metric": metric, "level": level} for level, value in values.items() if not math.isfinite(value) + ) + report["levels"] = [] + for level, spec in enumerate(metadata["levels"]): + delta = {} + for metric in METRICS: + a, b = [report["models"][mode]["metrics"][metric][str(level)] for mode in ("baseline", "candidate")] + delta[metric] = None if a is None or b is None else b - a + predictions = [ + [pred for pred, lvl in zip(tables[mode]["prediction"], tables[mode]["level"], strict=True) if lvl == level] + for mode in ("baseline", "candidate") + ] + report["levels"].append( + { + "name": spec["name"], + "samples": len(metadata["samples"]), + "candidate_minus_baseline": delta, + "prediction_changes": sum(a != b for a, b in zip(*predictions, strict=True)), + } + ) + report["status"] = "evaluated" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2, allow_nan=False) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--manifest", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True, help="New directory for canonical CSVs and metric report") + compare(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index d357e37..009b4ae 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1845,6 +1845,53 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. +### Maintained paired mini_metrics reproduction + +`dev.benchmarks.quality_compare` now validates prediction CSVs against an explicit +held-out sample/level/class manifest and evaluates Macro-F1, Macro-Recall, +Macro-Precision, Coverage and Theil's U. It preserves literal class names, +canonicalizes reordered rows, rejects missing/duplicate samples and changed labels, +and retains model/dataset provenance plus imported mini_metrics source hashes. +See the [input contract and commands](../dev/benchmarks/README.md#maintained-paired-quality-comparison). + +The real-data replay compared the retained TensorRT INT8 predictions with the +retained TensorRT FP16 baseline for both heads across all 912 Blair validation +images. It reused existing inference CSVs; no new model execution, score-parity +check or timing measurement was performed. Fixed threshold zero, no abstention, +no threshold optimization, ordinary per-level metrics and explicit `MacroF1` +selection preserve the previous evaluation policy. + +| Level | F1 delta | Recall delta | Precision delta | Coverage delta | Theil's U delta | Changed predictions | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Flat leaf | −0.003960 | −0.007645 | +0.002504 | 0 | −0.009136 | 61 | +| Hierarchical leaf | +0.000918 | −0.008148 | −0.023545 | 0 | −0.013048 | 67 | +| Hierarchical parent | −0.012269 | −0.022280 | +0.001343 | 0 | −0.022578 | 25 | + +Deltas are candidate minus baseline in the original metric units; an F1 delta +of −0.003960 is approximately −0.396 percentage points. Coverage remains one for +all rows under this complete known-label, no-abstention policy. These quality +changes must be considered alongside measured efficiency, not accepted alone. + +An initial exact-equality replay failed only on last-bit Macro-F1 differences. +Inspection of the imported mini_metrics implementation showed unordered class-set +iteration followed by floating summation. The maximum difference from historical +results was 3.33e−16; every result was within the explicit 1e−12 reproduction +tolerance. No metric implementation or expected historical result was changed. +Two fresh processes with `PYTHONHASHSEED=0` reproduced all metrics and deltas +exactly. The runner records the hash-seed setting. Undefined metrics, such as +Theil's U with a single observation, are explicit JSON nulls with an accompanying +list, not invalid JSON NaNs or silently substituted zeros. + +The initial replay is retained under `tmp-quality-compare/`; fixed-seed manifests, +canonical CSVs, package hashes and reports are under `tmp-quality-compare-fixed/`, +including both heads and their fresh-process repeats. The local sibling package +was selected explicitly through `PYTHONPATH`; the command itself assumes no local +checkout layout. All 14 focused tests passed with both the installed mini_metrics +release and the local checkout. Static checks and the full CPU-default suite +passed: 439 passed, 151 skipped and the known EMA expected failure. +Full-dataset inference collection and calibration +input preparation still need to be connected to the maintained evaluation commands. + ### Maintained paired TensorRT timing reproduction `dev.benchmarks.tensorrt_pair` now measures arbitrary existing engine pairs with diff --git a/docs/quantization-status.md b/docs/quantization-status.md index a7e58fb..23cec94 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,7 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration, build/inspection/smoke and paired timing commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained dataset-quality command | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration, build/inspection/smoke, paired timing and paired prediction-quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained full-dataset inference collection | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | @@ -50,9 +50,12 @@ production acceptance study. initializer arrays and graph nodes from the retained 128 training samples. [Paired inference timing](../dev/benchmarks/README.md#maintained-paired-tensorrt-timing-command) now retains alternating host-time pairs in a maintained command, with explicit - pageable/pinned IO regimes. Promote dataset metric evaluation into a maintained - developer command with an explicit optional environment; connect dataset - preparation to the calibration manifest contract. + pageable/pinned IO regimes. A maintained + [quality comparison](../dev/benchmarks/README.md#maintained-paired-quality-comparison) + validates paired prediction CSVs against an explicit held-out manifest and + evaluates all five requested mini_metrics metrics. Connect full-dataset + inference collection to that contract and dataset preparation to the calibration + manifest contract; the evaluator itself does not run model inference. Preserve calibration records, class/preprocessing contracts, hashes, failures and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` diff --git a/tests/test_benchmark_quality.py b/tests/test_benchmark_quality.py new file mode 100644 index 0000000..c24eb4b --- /dev/null +++ b/tests/test_benchmark_quality.py @@ -0,0 +1,152 @@ +import csv +import json + +import pytest + +from dev.benchmarks.quality_compare import COLUMNS, METRICS, compare, read_manifest, read_predictions + + +@pytest.fixture +def example(tmp_path): + classes = [["001", "1"], ["alpha", "beta"]] + labels = [["001", "alpha"], ["001", "alpha"], ["1", "beta"], ["1", "beta"]] + samples = [{"instance_id": i, "filename": f"image-{i}.jpg", "labels": labs} for i, labs in enumerate(labels)] + metadata = { + "schema_version": 1, + "split": "val", + "provenance": {"dataset": "oracle fixture"}, + "levels": [{"name": name, "classes": cs} for name, cs in zip(("leaf", "parent"), classes, strict=True)], + "samples": samples, + } + for mode in ("baseline", "candidate"): + path = tmp_path / f"{mode}.csv" + metadata[mode] = {"path": path.name, "classes": classes, "provenance": {"model": mode}} + rows = [] + for level in range(2): + for sample in samples: + prediction = sample["labels"][level] + if mode == "candidate" and level == 0 and sample["instance_id"] == 1: + prediction = "1" + rows.append( + dict( + instance_id=sample["instance_id"], + filename=sample["filename"], + level=level, + label=sample["labels"][level], + prediction=prediction, + confidence=0.8, + threshold=0, + ) + ) + if mode == "candidate": + rows.reverse() + with path.open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=COLUMNS) + writer.writeheader() + writer.writerows(rows) + manifest = tmp_path / "manifest.json" + manifest.write_text(json.dumps(metadata)) + return manifest, metadata + + +@pytest.mark.parametrize("kind", ["train", "mapping", "duplicate"]) +def test_manifest_rejects_invalid_evaluation_contract(example, kind): + path, metadata = example + if kind == "train": + metadata["split"] = "train" + elif kind == "mapping": + metadata["candidate"]["classes"] = [["1", "001"], ["alpha", "beta"]] + else: + metadata["samples"].append(metadata["samples"][0]) + path.write_text(json.dumps(metadata)) + with pytest.raises(ValueError): + read_manifest(path) + + +@pytest.mark.parametrize("kind", ["missing", "duplicate", "label", "filename", "class", "threshold", "nonfinite", "hash"]) +def test_predictions_reject_mismatched_samples_and_policy(example, kind): + path, metadata = example + artifact = metadata["candidate"] + source = path.parent / artifact["path"] + with source.open() as stream: + rows = list(csv.DictReader(stream)) + if kind == "missing": + rows.pop() + elif kind == "duplicate": + rows.append(rows[0]) + elif kind in ("label", "filename"): + rows[0][kind] = "changed" + elif kind == "class": + rows[0]["prediction"] = "not a class" + elif kind == "threshold": + rows[0]["threshold"] = "0.5" + elif kind == "nonfinite": + rows[0]["confidence"] = "nan" + else: + artifact["sha256"] = "wrong" + with source.open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=COLUMNS) + writer.writeheader() + writer.writerows(rows) + with pytest.raises(ValueError): + read_predictions(source, artifact, metadata) + + +def test_real_metrics_preserve_literal_labels_and_pair_reordered_rows(example, tmp_path, monkeypatch): + metrics = pytest.importorskip("mini_metrics.metrics") + path, _ = example + original = metrics.evaluate_file + + def evaluate(source, **kwargs): + assert kwargs["opt_crit"] is metrics.MacroF1 + assert kwargs["optimal"] is False and kwargs["threshold"] == 0 + return original(source, **kwargs) + + monkeypatch.setattr(metrics, "evaluate_file", evaluate) + report = compare(path, tmp_path / "result") + assert report["status"] == "evaluated" and not report["undefined_metrics"] + assert report == json.loads((tmp_path / "result/report.json").read_text()) + baseline = report["models"]["baseline"]["metrics"] + candidate = report["models"]["candidate"]["metrics"] + assert set(baseline) == set(METRICS) + assert all(value == pytest.approx(1) for levels in baseline.values() for value in levels.values()) + assert candidate["f1"]["0"] == pytest.approx(11 / 15) + assert candidate["recall"]["0"] == pytest.approx(0.75) + assert candidate["precision"]["0"] == pytest.approx(5 / 6) + assert candidate["coverage"]["0"] == 1 + assert candidate["theilU"]["0"] == pytest.approx(0.31127812445913283) + assert report["levels"][0]["prediction_changes"] == 1 + assert report["levels"][1]["prediction_changes"] == 0 + with pytest.raises(FileExistsError): + compare(path, tmp_path / "result") + + +def test_invalid_candidate_retains_baseline_evaluation(example, tmp_path): + pytest.importorskip("mini_metrics") + path, metadata = example + metadata["candidate"]["sha256"] = "changed" + path.write_text(json.dumps(metadata)) + output = tmp_path / "failure" + with pytest.raises(ValueError, match="hash mismatch"): + compare(path, output) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" and report["models"]["baseline"]["metrics"] + + +def test_undefined_theil_u_is_explicit_null_not_invalid_json(example, tmp_path): + pytest.importorskip("mini_metrics") + path, metadata = example + metadata["samples"] = metadata["samples"][:1] + path.write_text(json.dumps(metadata)) + for mode in ("baseline", "candidate"): + source = path.parent / metadata[mode]["path"] + with source.open() as stream: + rows = [r for r in csv.DictReader(stream) if r["instance_id"] == "0"] + with source.open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=COLUMNS) + writer.writeheader() + writer.writerows(rows) + report = compare(path, tmp_path / "undefined") + assert report["models"]["baseline"]["metrics"]["theilU"]["0"] is None + assert report["levels"][0]["candidate_minus_baseline"]["theilU"] is None + assert len(report["undefined_metrics"]) == 4 From 8f72f87f125af34160528c2e725189f8296901ab Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 15:04:26 +0200 Subject: [PATCH 065/155] feat: connect dataset inference to paired quality evaluation --- dev/benchmarks/README.md | 90 +++++++ dev/benchmarks/dataset_inference.py | 276 ++++++++++++++++++++++ dev/benchmarks/quality_compare.py | 19 +- docs/benchmarks.md | 54 +++++ docs/quantization-status.md | 11 +- tests/test_benchmark_dataset_inference.py | 166 +++++++++++++ 6 files changed, 605 insertions(+), 11 deletions(-) create mode 100644 dev/benchmarks/dataset_inference.py create mode 100644 tests/test_benchmark_dataset_inference.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 0e8bff5..09b6339 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -975,3 +975,93 @@ floating results. Do not alter metric definitions to force exact historical bytes. Correctness tests cover synthetic oracle metrics, literal class identities, reordered predictions, invalid contracts and undefined Theil's U; tests requiring mini_metrics skip explicitly when that optional evaluation package is absent. + +### Maintained full-dataset prediction collection + +`dev.benchmarks.dataset_inference` connects preprocessed held-out input batches to +the paired quality evaluator. ONNX Runtime supports the CPU/edge collection path +and explicit GPU providers; a separate TensorRT backend executes built engines. +The CPU path imports neither PyTorch, TensorRT nor mini_metrics. Prepare ONNX, +ONNX Runtime and NumPy explicitly for that path; TensorRT collection uses the +existing compatible TensorRT/CUDA PyTorch environment. The command installs nothing. + +Extend the quality manifest's `schema_version`, `split`, `provenance`, `samples` +and `levels` fields with input batches and explicit output bindings. Baseline and +candidate artifact fields are produced by collection rather than supplied here: + +```json +{ + "schema_version": 1, + "split": "val", + "provenance": {"dataset": "manifest hash and held-out selection", "preprocessing": "exact transforms"}, + "levels": [{"name": "leaf", "classes": ["cat", "dog"], "output": "output_0", "score_semantics": "logits"}], + "samples": [ + {"instance_id": 0, "filename": "cat.jpg", "labels": ["cat"]}, + {"instance_id": 1, "filename": "dog.jpg", "labels": ["dog"]} + ], + "batch_input": "images", + "batches": [{"path": "batch-000.npz", "sample_ids": ["0", "1"]}] +} +``` + +Each NPZ contains all named model inputs; static/unbatched inputs are permitted. +The leading dimension of `batch_input` must match the batch's `sample_ids`, which +are canonical string forms of the manifest's integer IDs. Batches must cover every +sample exactly once. Paths are relative to the manifest, optional batch `sha256` +values are checked, and hashes of the loaded bytes are always recorded. Batch +order may differ from sample declaration order. Different batch sizes, including +the final partial batch, must fit the runtime graph/engine profile. + +Each mapped output must be finite floating scores of shape `[batch, classes]`. +Declare either `logits` (confidence computed by stable softmax) or `probabilities` +(checked to lie in [0,1] and sum to one). Class lists follow score-column order; +the collector cannot infer a trustworthy mapping from an opaque engine or prove +the input preprocessing matches the checkpoint. Bind these from reviewed export +metadata and preserve provenance. Add one mapping per hierarchical level; no +backbone/head allowlist is used. + +```bash +# CPU/edge collection, using an explicit ONNX candidate: +OMP_NUM_THREADS=1 python -m dev.benchmarks.dataset_inference \ + --model calibrated/model.onnx --manifest heldout-inputs/manifest.json \ + --output /tmp/heldout-cpu-1 --threads 1 + +# GPU baseline, then candidate and a ready-to-evaluate comparison manifest: +OMP_NUM_THREADS=1 python -m dev.benchmarks.dataset_inference \ + --backend tensorrt --model fp16/model.engine \ + --manifest heldout-inputs/manifest.json --output /tmp/heldout-fp16-1 +OMP_NUM_THREADS=1 python -m dev.benchmarks.dataset_inference \ + --backend tensorrt --model int8/model.engine \ + --manifest heldout-inputs/manifest.json --output /tmp/heldout-int8-1 \ + --baseline-bundle /tmp/heldout-fp16-1/evaluation.json +PYTHONHASHSEED=0 OMP_NUM_THREADS=1 python -m dev.benchmarks.quality_compare \ + --manifest /tmp/heldout-int8-1/comparison.json --output /tmp/heldout-quality-1 +``` + +ONNX collection defaults to `CPUExecutionProvider`, one intra-op thread, one +inter-op thread and full graph optimization. `--provider`, JSON +`--provider-options` and `--optimization disable` support explicit alternatives. +The requested provider must be available and activated, but registered providers +do not establish per-operator placement; use the maintained placement/inspection +commands separately. CPU fallback within a GPU partitioned graph is not ruled out. +TensorRT uses `--device` and profile 0, with standard plugins registered. Its +current linear device-IO and shape-tensor limitations match the paired runner. + +The new output directory receives `predictions.csv`, `evaluation.json` and +`report.json`. Supplying `--baseline-bundle` additionally creates `comparison.json` +after checking matching split, samples, labels and ordered level mappings. +Different backend artifacts can be compared this way. CSV paths in generated +bundles are absolute; update them if moving artifacts, retaining their hashes. +`--save-scores` retains each batch's named arrays and hashes for numerical +reproduction checks. Without it, only predictions and provenance are retained, +avoiding a full-dataset score matrix for large heads. + +Collection holds one batch's inputs/outputs at a time and streams CSV rows. It is +a correctness/quality path with per-batch buffer allocation, not an inference +performance benchmark. Confidence calculation may use higher precision than a +previous producer; raw scores and discrete predictions are separate contracts. +Failures retain completed batches and partial CSVs, while incomplete collection +does not publish an evaluation bundle. Existing directories are refused. Tests +cover multiple inputs, mapped levels, changing batch sizes, full CPU collection +through mini_metrics, saved scores, failures and CPU dependency isolation; the +intentional GPU test checks both batch sizes against CPU outputs. diff --git a/dev/benchmarks/dataset_inference.py b/dev/benchmarks/dataset_inference.py new file mode 100644 index 0000000..9ed5918 --- /dev/null +++ b/dev/benchmarks/dataset_inference.py @@ -0,0 +1,276 @@ +"""Collect held-out ONNX/TensorRT predictions for the paired quality evaluator.""" + +import csv +import hashlib +import json +import platform +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np + +from .onnx_calibration import load_batch +from .onnx_inference import file_hash, model_files +from .quality_compare import COLUMNS, validate_dataset + + +def inference_manifest(path): + payload = Path(path).read_bytes() + manifest = json.loads(payload) + validate_dataset(manifest) + if not isinstance(manifest.get("batch_input"), str) or not manifest["batch_input"]: + raise ValueError("Declare batch_input for sample identity") + outputs = [] + for level in manifest["levels"]: + if ( + not isinstance(level.get("output"), str) + or not level["output"] + or level.get("score_semantics") not in ("logits", "probabilities") + ): + raise ValueError("Each level needs an output name and logits/probabilities score_semantics") + outputs.append(level["output"]) + if len(set(outputs)) != len(outputs): + raise ValueError("Output bindings must be unique") + batches = manifest.get("batches") + if not isinstance(batches, list) or not batches: + raise ValueError("Supply ordered input batches") + ids = [] + for batch in batches: + if not isinstance(batch.get("path"), str) or not batch["path"]: + raise ValueError("Each batch needs an NPZ path") + if not isinstance(batch.get("sample_ids"), list) or not batch["sample_ids"]: + raise ValueError("Each batch needs sample_ids as canonical integer strings") + ids.extend(batch["sample_ids"]) + if ( + any(not isinstance(i, str) for i in ids) + or len(set(ids)) != len(ids) + or set(ids) != {str(s["instance_id"]) for s in manifest["samples"]} + ): + raise ValueError("Batches must cover every declared sample exactly once using canonical string IDs") + return manifest, hashlib.sha256(payload).hexdigest() + + +def predictions(values, level, count): + values = np.asarray(values) + if values.shape != (count, len(level["classes"])) or values.dtype.kind != "f" or not np.isfinite(values).all(): + raise ValueError(f"Expected finite floating [batch,classes] scores for {level['name']}") + indices = values.argmax(axis=1) + if level["score_semantics"] == "probabilities": + if (values < 0).any() or (values > 1).any() or not np.allclose(values.sum(axis=1, dtype=np.float64), 1, rtol=1e-5, atol=1e-5): + raise ValueError("Declared probabilities must lie in [0,1] and sum to one") + confidence = values[np.arange(count), indices] + else: + shifted = values.astype(np.float64) - values.max(axis=1, keepdims=True) + confidence = 1 / np.exp(shifted).sum(axis=1) + return indices, confidence + + +class OnnxPredictor: + def __init__(self, model, report, provider, provider_options, threads, optimization): + import onnx + import onnxruntime as ort + + if provider not in ort.get_available_providers(): + raise ValueError(f"Requested provider unavailable: {provider}") + report["model_files"] = model_files(Path(model), onnx) + options = ort.SessionOptions() + options.intra_op_num_threads, options.inter_op_num_threads = threads, 1 + options.graph_optimization_level = getattr( + ort.GraphOptimizationLevel, f"ORT_{'DISABLE_ALL' if optimization == 'disable' else 'ENABLE_ALL'}" + ) + self.session = ort.InferenceSession(str(model), sess_options=options, providers=[provider], provider_options=[provider_options]) + self.session.disable_fallback() + if provider not in self.session.get_providers(): + raise RuntimeError("Requested provider was not activated") + self.names = [node.name for node in self.session.get_outputs()] + report["runtime"] = { + "onnx": onnx.__version__, + "onnxruntime": ort.__version__, + "providers": self.session.get_providers(), + "provider_options": self.session.get_provider_options(), + } + + def __call__(self, feeds): + return dict(zip(self.names, self.session.run(None, feeds), strict=True)) + + +class TensorRTPredictor: + def __init__(self, model, report, device): + import tensorrt as trt + import torch + + self.torch, self.trt, self.device = torch, trt, device + self.logger = trt.Logger(trt.Logger.WARNING) + trt.init_libnvinfer_plugins(self.logger, "") + self.runtime = trt.Runtime(self.logger) + payload = Path(model).read_bytes() + report["model_files"] = [{"path": str(Path(model).resolve()), "sha256": hashlib.sha256(payload).hexdigest(), "bytes": len(payload)}] + with torch.cuda.device(device): + self.engine = self.runtime.deserialize_cuda_engine(payload) + if self.engine is None: + raise RuntimeError("Could not deserialize TensorRT engine") + self.context = self.engine.create_execution_context() + if self.context is None: + raise RuntimeError("Could not create TensorRT execution context") + self.stream = torch.cuda.Stream(device=device) + report["runtime"] = { + "tensorrt": trt.__version__, + "torch": torch.__version__, + "gpu": torch.cuda.get_device_name(device), + "device": device, + "profile": 0, + } + + def __call__(self, feeds): + from .tensorrt_pair import buffers_for + + torch = self.torch + with torch.cuda.device(self.device), torch.cuda.stream(self.stream): + buffers, inputs = buffers_for(self.engine, self.context, feeds, torch, self.trt, self.device, False) + for name in inputs: + gpu, host = buffers[name] + gpu.copy_(host) + if not self.context.execute_async_v3(self.stream.cuda_stream): + raise RuntimeError("TensorRT dataset execution failed") + for name, (gpu, host) in buffers.items(): + if name not in inputs: + host.copy_(gpu) + self.stream.synchronize() + return {name: host.numpy() for name, (_, host) in buffers.items() if name not in inputs} + + +def pair_bundle(baseline, candidate): + old = json.loads(Path(baseline).read_text()) + for key in ("schema_version", "split", "levels", "samples"): + if old[key] != candidate[key]: + raise ValueError(f"Baseline bundle differs in held-out {key}") + return { + **{key: candidate[key] for key in ("schema_version", "split", "levels", "samples", "provenance")}, + "baseline": old["artifact"], + "candidate": candidate["artifact"], + } + + +def collect( + model, + manifest, + output, + backend="onnx", + provider="CPUExecutionProvider", + provider_options=None, + threads=1, + optimization="all", + device=0, + save_scores=False, + baseline_bundle=None, +): + if backend not in ("onnx", "tensorrt") or threads < 1 or device < 0 or optimization not in ("all", "disable"): + raise ValueError("Invalid backend, thread count, device or graph optimization") + metadata, digest = inference_manifest(manifest) + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "environment": {"platform": platform.platform(), "python": platform.python_version(), "numpy": np.__version__}, + "manifest": {"path": str(manifest), "sha256": digest, "contents": metadata}, + "settings": { + "backend": backend, + "provider": provider if backend == "onnx" else None, + "threads": threads if backend == "onnx" else None, + "optimization": optimization if backend == "onnx" else None, + "save_scores": save_scores, + }, + "batches": [], + "scope": ( + "Held-out prediction collection from explicit preprocessed inputs; " + "not speed, integer placement, score parity or deployment acceptance." + ), + } + try: + predictor = ( + OnnxPredictor(model, report, provider, provider_options or {}, threads, optimization) + if backend == "onnx" + else TensorRTPredictor(model, report, device) + ) + samples = {str(s["instance_id"]): s for s in metadata["samples"]} + path = output / "predictions.csv" + with path.open("w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=COLUMNS) + writer.writeheader() + for index, batch in enumerate(metadata["batches"]): + feeds, record = load_batch(manifest, metadata, batch) + arrays = predictor(feeds) + rows = [] + for level, spec in enumerate(metadata["levels"]): + if spec["output"] not in arrays: + raise ValueError(f"Missing mapped output: {spec['output']}") + predicted, confidence = predictions(arrays[spec["output"]], spec, len(batch["sample_ids"])) + for identifier, pred, conf in zip(batch["sample_ids"], predicted, confidence, strict=True): + sample = samples[identifier] + rows.append( + dict( + instance_id=sample["instance_id"], + filename=sample["filename"], + level=level, + label=sample["labels"][level], + prediction=spec["classes"][int(pred)], + confidence=float(conf), + threshold=0, + ) + ) + if save_scores: + score_path = output / f"scores-{index:05d}.npz" + np.savez(score_path, **arrays) + record["scores"] = {"path": score_path.name, "sha256": file_hash(score_path)} + writer.writerows(rows) + report["batches"].append(record) + del feeds, arrays + artifact = { + "path": str(path.resolve()), + "sha256": file_hash(path), + "classes": [level["classes"] for level in metadata["levels"]], + "provenance": { + "model_files": report["model_files"], + "inference_manifest_sha256": digest, + "runner_sha256": report["runner_sha256"], + "runtime": report["runtime"], + "report": str((output / "report.json").resolve()), + }, + } + bundle = { + **{key: metadata[key] for key in ("schema_version", "split", "samples", "provenance")}, + "levels": [{key: spec[key] for key in ("name", "classes")} for spec in metadata["levels"]], + "artifact": artifact, + } + (output / "evaluation.json").write_text(json.dumps(bundle, indent=2) + "\n") + if baseline_bundle is not None: + (output / "comparison.json").write_text(json.dumps(pair_bundle(baseline_bundle, bundle), indent=2) + "\n") + report.update(status="inferred", predictions=artifact) + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--model", type=Path, required=True) + parser.add_argument("--manifest", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--backend", choices=["onnx", "tensorrt"], default="onnx") + parser.add_argument("--provider", default="CPUExecutionProvider") + parser.add_argument("--provider-options", type=json.loads) + parser.add_argument("--threads", type=int, default=1) + parser.add_argument("--optimization", choices=["all", "disable"], default="all") + parser.add_argument("--device", type=int, default=0) + parser.add_argument("--save-scores", action="store_true") + parser.add_argument("--baseline-bundle", type=Path, help="Create comparison.json against a previous inference evaluation.json") + collect(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/dev/benchmarks/quality_compare.py b/dev/benchmarks/quality_compare.py index 28d0c4b..d5d8618 100644 --- a/dev/benchmarks/quality_compare.py +++ b/dev/benchmarks/quality_compare.py @@ -20,6 +20,18 @@ def read_manifest(path): payload = Path(path).read_bytes() manifest = json.loads(payload) + validate_dataset(manifest) + for mode in ("baseline", "candidate"): + artifact = manifest.get(mode, {}) + if not isinstance(artifact.get("path"), str) or not artifact["path"] or not artifact.get("provenance"): + raise ValueError(f"Declare prediction path and model/preprocessing provenance for {mode}") + if artifact.get("classes") != [level["classes"] for level in manifest["levels"]]: + raise ValueError(f"{mode} class mappings must match the ordered level mappings") + return manifest, hashlib.sha256(payload).hexdigest() + + +def validate_dataset(manifest): + """Validate the shared held-out identity contract before inference or evaluation.""" if manifest.get("schema_version") != 1 or manifest.get("split") not in ("val", "test"): raise ValueError("Require schema_version=1 and a declared val/test split") if not isinstance(manifest.get("provenance"), dict) or not manifest["provenance"]: @@ -57,13 +69,6 @@ def read_manifest(path): raise ValueError("Each sample needs one label per level") if any(label not in classes for label, classes in zip(labels, class_sets, strict=True)): raise ValueError("Sample labels must belong to the declared classes") - for mode in ("baseline", "candidate"): - artifact = manifest.get(mode, {}) - if not isinstance(artifact.get("path"), str) or not artifact["path"] or not artifact.get("provenance"): - raise ValueError(f"Declare prediction path and model/preprocessing provenance for {mode}") - if artifact.get("classes") != [level["classes"] for level in levels]: - raise ValueError(f"{mode} class mappings must match the ordered level mappings") - return manifest, hashlib.sha256(payload).hexdigest() def read_predictions(path, artifact, manifest): diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 009b4ae..86132ca 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1845,6 +1845,60 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. +### Maintained full-dataset inference reproduction + +`dev.benchmarks.dataset_inference` now collects complete held-out prediction +tables with ONNX Runtime or TensorRT and produces the paired quality evaluator's +manifest directly. It validates the shared dataset identity contract before +execution, binds named outputs to ordered class lists and explicit score semantics, +checks every batch's identity/hash/shape, and optionally retains batch scores. +The CPU path imports neither PyTorch, TensorRT nor mini_metrics. See the +[input schema and composed commands](../dev/benchmarks/README.md#maintained-full-dataset-prediction-collection). + +For the Blair replay, all 912 validation source images were checked against their +recorded dataset hashes. Inputs were regenerated with each retained floating +checkpoint's preprocessing at 128px, after checking checkpoint class mappings +against the dataset manifest. Each head produced 114 batches of eight; the +synthetic regression separately exercises a final partial batch and static +secondary inputs. Real inputs/manifests are retained in `tmp-heldout-inputs/`. + +Both heads were then executed through the maintained collector using the retained +FP16 TensorRT engine, retained signed-QDQ INT8 TensorRT engine, and the maintained +calibration command's signed-QDQ ONNX model on CPU. All six runs evaluated all +912 images. They were correctness runs; CPU tests were active during part of the +work, so no timing or memory-benefit claim is made. + +| Head/backend | Discrete predictions versus retained results | Maximum leaf score difference | Maximum parent score difference | +| --- | --- | ---: | ---: | +| Flat TensorRT FP16 | All identical | 0 | — | +| Flat TensorRT INT8 | All identical | 0 | — | +| Flat ONNX CPU INT8 | All identical | 0 | — | +| Hierarchical TensorRT FP16 | All identical | 0 | 1.19e−7 | +| Hierarchical TensorRT INT8 | All identical | 0 | 1.19e−7 | +| Hierarchical ONNX CPU INT8 | All identical | 0 | 0 | + +The small TensorRT parent-score differences mean full bitwise score reproduction +is not established; they remain below the earlier absolute parity tolerance of +1e−5 and cause no argmax changes. Four generated comparison manifests (INT8 GPU +and CPU versus FP16 for each head) ran through the maintained mini_metrics +evaluator. Candidate metrics matched the retained reports within 3.33e−16. +Confidence is now computed with a float64 softmax for declared logits; CSV bytes +are therefore a separate contract from raw scores and discrete predictions. + +Collection reports, scores, bundles, paired metric reports and reproduction +checks are retained under `tmp-heldout-collection/{flat,hierarchical}-{fp16,int8,cpu_int8}/`. +These results verify the maintained collection/evaluation connection on the local +x86 CPU and laptop GPU. They do not establish ARM runtime behavior, target GPU +performance, native QT checkpoint deployment or production acceptance. Source +image preprocessing/batch preparation for this replay still used a local script; +promoting that preparation and composing continuous jobs remain necessary. + +Static checks and the full CPU-default suite passed: 447 passed, 152 skipped and +the known EMA expected failure. A subsequent focused run including the new CPU +dependency-isolation regression passed 23 tests with one GPU skip. All nine +focused tests present in the prepared TensorRT run passed, including real +multi-input/two-level execution at both batch sizes. + ### Maintained paired mini_metrics reproduction `dev.benchmarks.quality_compare` now validates prediction CSVs against an explicit diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 23cec94..9bcc8d2 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,7 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration, build/inspection/smoke, paired timing and paired prediction-quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; maintained full-dataset inference collection | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; automated source-dataset preparation and orchestration | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | @@ -53,9 +53,12 @@ production acceptance study. pageable/pinned IO regimes. A maintained [quality comparison](../dev/benchmarks/README.md#maintained-paired-quality-comparison) validates paired prediction CSVs against an explicit held-out manifest and - evaluates all five requested mini_metrics metrics. Connect full-dataset - inference collection to that contract and dataset preparation to the calibration - manifest contract; the evaluator itself does not run model inference. + evaluates all five requested mini_metrics metrics. Maintained + [full-dataset collection](../dev/benchmarks/README.md#maintained-full-dataset-prediction-collection) + now feeds that evaluator for ONNX Runtime and TensorRT, with complete Blair + replays for both heads. Connect source-dataset preparation to the calibration + and held-out input contracts and automate the composed pipeline. Input batches + for the real-data replay were still prepared with a local script. Preserve calibration records, class/preprocessing contracts, hashes, failures and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` diff --git a/tests/test_benchmark_dataset_inference.py b/tests/test_benchmark_dataset_inference.py new file mode 100644 index 0000000..e8575a3 --- /dev/null +++ b/tests/test_benchmark_dataset_inference.py @@ -0,0 +1,166 @@ +import json +import os +import subprocess +import sys + +import numpy as np +import pytest + +from dev.benchmarks.dataset_inference import collect, inference_manifest, pair_bundle, predictions +from dev.benchmarks.quality_compare import read_manifest, read_predictions + + +@pytest.fixture +def example(tmp_path): + onnx = pytest.importorskip("onnx") + graph = onnx.helper.make_graph( + [onnx.helper.make_node("Add", ["x", "offset"], ["leaf"]), onnx.helper.make_node("Neg", ["leaf"], ["parent"])], + "two-input-two-level", + [ + onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, ["batch", 2]), + onnx.helper.make_tensor_value_info("offset", onnx.TensorProto.FLOAT, [2]), + ], + [onnx.helper.make_tensor_value_info(name, onnx.TensorProto.FLOAT, ["batch", 2]) for name in ("leaf", "parent")], + ) + model = tmp_path / "model.onnx" + onnx.save(onnx.helper.make_model(graph, opset_imports=[onnx.helper.make_opsetid("", 18)], ir_version=10), model) + for i, values in enumerate(([[-2, 2], [2, -2]], [[3, -3]])): + np.savez(tmp_path / f"batch-{i}.npz", x=np.array(values, dtype=np.float32), offset=np.zeros(2, dtype=np.float32)) + metadata = { + "schema_version": 1, + "split": "val", + "provenance": {"dataset": "two-level oracle", "preprocessing": "identity"}, + "levels": [ + {"name": "leaf", "classes": ["001", "1"], "output": "leaf", "score_semantics": "logits"}, + {"name": "parent", "classes": ["a", "b"], "output": "parent", "score_semantics": "logits"}, + ], + "samples": [ + {"instance_id": i, "filename": f"image-{i}", "labels": labs} + for i, labs in [(7, ["001", "b"]), (3, ["001", "b"]), (9, ["1", "a"])] + ], + "batch_input": "x", + "batches": [{"path": "batch-0.npz", "sample_ids": ["9", "7"]}, {"path": "batch-1.npz", "sample_ids": ["3"]}], + } + manifest = tmp_path / "manifest.json" + manifest.write_text(json.dumps(metadata)) + return model, manifest, metadata + + +@pytest.mark.parametrize("kind", ["duplicate", "missing", "noncanonical", "output"]) +def test_inference_manifest_rejects_incomplete_or_ambiguous_contract(example, kind): + _, path, metadata = example + if kind == "duplicate": + metadata["batches"][1]["sample_ids"] = ["7"] + elif kind == "missing": + metadata["batches"].pop() + elif kind == "noncanonical": + metadata["batches"][1]["sample_ids"] = ["03"] + else: + metadata["levels"][1]["output"] = "leaf" + path.write_text(json.dumps(metadata)) + with pytest.raises(ValueError): + inference_manifest(path) + + +def test_score_semantics_and_shape_are_explicit(): + level = {"name": "leaf", "classes": ["a", "b"], "score_semantics": "logits"} + pred, confidence = predictions(np.array([[10000, 9999]], dtype=np.float32), level, 1) + assert pred.tolist() == [0] + assert confidence[0] == pytest.approx(1 / (1 + np.exp(-1))) + probability = {**level, "score_semantics": "probabilities"} + assert predictions(np.array([[0.25, 0.75]]), probability, 1)[0].tolist() == [1] + for values in (np.array([[1.0, 1.0]]), np.array([[-0.1, 1.1]]), np.array([[np.nan, 1]]), np.zeros((1, 3)), np.ones((1, 2), dtype=int)): + with pytest.raises(ValueError): + predictions(values, probability, 1) + + +def test_cpu_collection_handles_multiple_inputs_levels_and_partial_batch(example, tmp_path): + pytest.importorskip("onnxruntime") + model, manifest, metadata = example + baseline, candidate = tmp_path / "baseline", tmp_path / "candidate" + report = collect(model, manifest, baseline, save_scores=True) + assert report["status"] == "inferred" and len(report["batches"]) == 2 + assert [b["sample_ids"] for b in report["batches"]] == [["9", "7"], ["3"]] + with np.load(baseline / "scores-00001.npz") as data: + np.testing.assert_array_equal(data["leaf"], [[3, -3]]) + np.testing.assert_array_equal(data["parent"], [[-3, 3]]) + collect(model, manifest, candidate, baseline_bundle=baseline / "evaluation.json") + pair, _ = read_manifest(candidate / "comparison.json") + table, _ = read_predictions(candidate / "predictions.csv", pair["candidate"], pair) + assert table["label"] == table["prediction"] + assert len(table["label"]) == 6 + with pytest.raises(FileExistsError): + collect(model, manifest, baseline) + metadata["batches"][1]["sha256"] = "wrong" + manifest.write_text(json.dumps(metadata)) + with pytest.raises(ValueError, match="hash mismatch"): + collect(model, manifest, tmp_path / "failure") + failed = json.loads((tmp_path / "failure/report.json").read_text()) + assert failed["status"] == "failed" and len(failed["batches"]) == 1 + assert not (tmp_path / "failure/evaluation.json").exists() + + +def test_bundle_rejects_different_labels(example, tmp_path): + _, _, metadata = example + bundle = {**metadata, "artifact": {"path": "file.csv"}} + baseline = tmp_path / "bundle.json" + baseline.write_text(json.dumps(bundle)) + bundle["samples"][0]["labels"][0] = "changed" + with pytest.raises(ValueError, match="samples"): + pair_bundle(baseline, bundle) + + +def test_cpu_inference_to_real_mini_metrics(example, tmp_path): + pytest.importorskip("onnxruntime") + pytest.importorskip("mini_metrics") + from dev.benchmarks.quality_compare import compare + + model, manifest, _ = example + collect(model, manifest, tmp_path / "baseline") + collect(model, manifest, tmp_path / "candidate", baseline_bundle=tmp_path / "baseline/evaluation.json") + result = compare(tmp_path / "candidate/comparison.json", tmp_path / "metrics") + assert all(v == pytest.approx(1) for levels in result["models"]["candidate"]["metrics"].values() for v in levels.values()) + + +def test_cpu_collection_does_not_import_training_or_gpu_packages(example, tmp_path): + pytest.importorskip("onnxruntime") + model, manifest, _ = example + code = """ +import sys +class Reject: + def find_spec(self, fullname, path=None, target=None): + if fullname.split('.')[0] in {'torch', 'tensorrt', 'mini_metrics'}: + raise AssertionError('Unnecessary CPU collection dependency: ' + fullname) +sys.meta_path.insert(0, Reject()) +from dev.benchmarks.dataset_inference import collect +collect(sys.argv[1], sys.argv[2], sys.argv[3]) +""" + subprocess.run([sys.executable, "-c", code, str(model), str(manifest), str(tmp_path / "isolated")], check=True, capture_output=True) + + +def test_tensorrt_dataset_outputs_match_cpu_at_both_batch_sizes(example, tmp_path): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 in an explicitly prepared TensorRT environment") + pytest.importorskip("tensorrt") + import torch + + assert torch.cuda.is_available(), "CUDA requested but unavailable" + from dev.benchmarks.tensorrt_build import build + + model, manifest, _ = example + profiles = {"x": {"min": [1, 2], "opt": [2, 2], "max": [2, 2]}, "offset": {"min": [2], "opt": [2], "max": [2]}} + build(model, tmp_path / "batch-0.npz", tmp_path / "build", profiles=profiles, optimization=0) + collect(model, manifest, tmp_path / "cpu", save_scores=True) + result = collect( + tmp_path / "build/model.engine", + manifest, + tmp_path / "gpu", + backend="tensorrt", + save_scores=True, + baseline_bundle=tmp_path / "cpu/evaluation.json", + ) + assert result["status"] == "inferred" and result["runtime"]["profile"] == 0 + for index in range(2): + with np.load(tmp_path / f"cpu/scores-{index:05d}.npz") as cpu, np.load(tmp_path / f"gpu/scores-{index:05d}.npz") as gpu: + for name in ("leaf", "parent"): + np.testing.assert_array_equal(cpu[name], gpu[name]) From 3986babe592526d9bef133fa3bab99d8dc735283 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 15:22:19 +0200 Subject: [PATCH 066/155] feat: prepare reproducible calibration and held-out image inputs --- dev/benchmarks/README.md | 80 +++++++ dev/benchmarks/prepare_inputs.py | 301 +++++++++++++++++++++++++ docs/benchmarks.md | 45 ++++ docs/quantization-status.md | 12 +- tests/test_benchmark_prepare_inputs.py | 132 +++++++++++ 5 files changed, 566 insertions(+), 4 deletions(-) create mode 100644 dev/benchmarks/prepare_inputs.py create mode 100644 tests/test_benchmark_prepare_inputs.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 09b6339..fa15e71 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1065,3 +1065,83 @@ does not publish an evaluation bundle. Existing directories are refused. Tests cover multiple inputs, mapped levels, changing batch sizes, full CPU collection through mini_metrics, saved scores, failures and CPU dependency isolation; the intentional GPU test checks both batch sizes against CPU outputs. + +### Maintained image input preparation + +`dev.benchmarks.prepare_inputs` replaces the local image-to-NPZ preparation +scripts. It reads a version-1 benchmark dataset manifest and an ONNX export +manifest, checks class ordering and source image hashes, and writes the calibration +or held-out input contract used by the commands above. It uses the existing +repository image reader on CPU with **zero workers** and an explicit thread limit. + +```bash +# Reproduce the training-split calibration selection: +OMP_NUM_THREADS=1 python -m dev.benchmarks.prepare_inputs \ + --dataset-manifest benchmark/dataset_manifest.json --data-root examples/blair \ + --export-manifest exported-float/manifest.json --output /tmp/calibration-inputs-1 \ + --split train --count 128 --seed 42 --batch-size 8 --score-semantics logits + +# Complete held-out validation input set, without subsampling: +OMP_NUM_THREADS=1 python -m dev.benchmarks.prepare_inputs \ + --dataset-manifest benchmark/dataset_manifest.json --data-root examples/blair \ + --export-manifest exported-float/manifest.json --output /tmp/validation-inputs-1 \ + --split val --batch-size 8 --score-semantics logits +``` + +Pass the resulting `manifest.json` to `onnx_calibration` for `train` or +`dataset_inference` for `val`/`test`. Omitting `--count` preserves the complete +split in dataset-manifest order. Supplying it uses +`random.Random(seed).sample(split_records, count)` with explicit cardinality +checks. IDs are assigned in selected-record order; filenames and labels preserve +source identity. The full source records, their hashes and selection are retained +in `report.json`. Declared cross-split duplicate image hashes are rejected, and +each selected source image is rehashed before and after preparation. This does +not independently re-inventory unselected files or prove the supplied split policy. + +The source manifest is the existing benchmark format: `class_spec.cls2idx` is a +flat mapping or numbered level mappings, and each record includes `path`, `split`, +`targets` (one integer per source level) and `sha256`. Paths must resolve under +`--data-root`. Export metadata must contain matching class indices, the image +input shape/dtype, ordered outputs, and the image-reader `resize_size`. Ambiguous +classifier metadata requires `--classifier-module`. Default source levels are +the first N levels for N exported levels; `--source-levels` selects others +explicitly. Class ordering must match exactly. `--level-names` replaces the default +`level_0`, `level_1`, ... names. + +`--score-semantics` is required, because arbitrary exported forward outputs do +not establish whether they are logits or probabilities. Supply one value for all +levels or one per level. The exporter must declare preprocessing outside the +graph. Prepared batches must match its floating image-input shape and dtype, +including fixed batch dimensions; a partial batch is rejected if the graph does +not support it. + +The default factory is +`dev.benchmarks.prepare_inputs:repository_preprocess`. It uses the existing +architecture resolver and recorded resize/preprocessing dtype, with pretrained +weights disabled and local-only configuration loading where the backend supports +it. It does not load checkpoint tensors or construct the trained classifier head; +the architecture getter may still construct a backbone to resolve its transforms. +It does not interpret the export recipe's descriptive text as executable transforms. + +For custom transforms/configuration, provide a trusted importable Python factory +with `--preprocess-factory package.module:function` and JSON `--factory-args`. +The factory receives classifier metadata as its first argument and returns a +callable from the repository's resized RGB uint8 batch to one image tensor. +The callable must preserve row identity. The default factory accepts `model_args` +inside factory arguments for existing backend options. Factory and transform +execution each receive the declared seed for Python, legacy NumPy and CPU Torch +RNGs, with caller RNGs restored; custom generators or external randomness remain +the factory's responsibility. Use deterministic inference preprocessing and verify +input hashes. Factory source hash, arguments, representation and Torch/NumPy +versions are recorded. Current architecture defaults are not proof that an +unrecorded custom training transform has been reconstructed correctly. + +A new directory receives batch NPZs, `manifest.json` and `report.json`. Failures +after output creation retain diagnostics and any completed batches; existing +directories are refused. Preparation processes one batch at a time, but its +architecture loader and source inventory still consume memory. This image helper +currently produces one floating image input; the lower-level calibration and +collection manifests continue to support manually prepared multiple inputs. +The default factory requires the repository's PyTorch/Torchvision environment. +The resulting NPZs can be transferred to the lighter ONNX CPU collection environment +without reconstructing preprocessing on the edge device. diff --git a/dev/benchmarks/prepare_inputs.py b/dev/benchmarks/prepare_inputs.py new file mode 100644 index 0000000..e18b1e8 --- /dev/null +++ b/dev/benchmarks/prepare_inputs.py @@ -0,0 +1,301 @@ +"""Prepare calibration or held-out image batches from explicit export/dataset contracts.""" + +import hashlib +import importlib +import inspect +import json +import random +from argparse import ArgumentParser +from contextlib import contextmanager +from pathlib import Path + +import numpy as np + +from .dataset_inference import inference_manifest +from .onnx_calibration import calibration_manifest +from .onnx_inference import file_hash + + +@contextmanager +def seeded(seed): + """Restore caller RNGs; factories and transforms get explicit independent seeds.""" + import torch + + python_state, numpy_state = random.getstate(), np.random.get_state() + try: + with torch.random.fork_rng(devices=[]): + random.seed(seed) + np.random.seed(seed % 2**32) + torch.manual_seed(seed) + yield + finally: + random.setstate(python_state) + np.random.set_state(numpy_state) + + +def ordered_classes(mapping): + if not isinstance(mapping, dict) or not mapping or any(not isinstance(k, str) or not k for k in mapping): + raise ValueError("Require nonempty class-name mappings") + if all(type(v) is int for v in mapping.values()): + if sorted(mapping.values()) != list(range(len(mapping))): + raise ValueError("Class indices must be unique and contiguous from zero") + return [sorted(mapping, key=mapping.get)] + if set(mapping) != {str(i) for i in range(len(mapping))}: + raise ValueError("Hierarchical level indices must be contiguous from zero") + result = [] + for i in range(len(mapping)): + level = ordered_classes(mapping[str(i)]) + if len(level) != 1: + raise ValueError("Expected one class mapping per level") + result.extend(level) + return result + + +def select_records(dataset, split, count, seed): + if dataset.get("schema_version") != 1 or split not in ("train", "val", "test"): + raise ValueError("Require a version-1 dataset and train/val/test split") + seen, hashes, selected = set(), {}, [] + for record in dataset["records"]: + path, digest, source_split = record["path"], record["sha256"], record["split"] + if not isinstance(path, str) or not path or path in seen: + raise ValueError("Dataset paths must be nonempty and unique") + seen.add(path) + if not isinstance(digest, str) or len(digest) != 64 or any(c not in "0123456789abcdef" for c in digest): + raise ValueError("Require lowercase SHA256 source image hashes") + if source_split not in ("train", "val", "test"): + raise ValueError("Unknown dataset split") + if digest in hashes and hashes[digest] != source_split: + raise ValueError("Dataset declares byte-identical images across splits") + hashes[digest] = source_split + if source_split == split: + selected.append(record) + if not selected: + raise ValueError("Selected split is empty") + if count is not None: + if not 0 < count <= len(selected): + raise ValueError("count must fit the selected split") + selected = random.Random(seed).sample(selected, count) + return selected + + +def repository_preprocess(metadata, model_args=None): + """Resolve preprocessing through existing loaders without constructing the trained head.""" + from mini_trainer.modeling.architectures.load import get_dynamic_model, get_model, resolve_backbone_getter + from mini_trainer.utils import string_to_dtype + + name = metadata["backbone_class"] + getter, _ = resolve_backbone_getter(name) + args = {} if getter is get_dynamic_model else {"pretrained": False, "local_files_only": True} + args.update(model_args or {}) + args["resize_size"] = metadata["resize_size"] + dtype = string_to_dtype(metadata.get("preprocess_dtype") or metadata.get("_dtype", "float32")) + _, _, preprocess, _, _ = get_model(name, model_args=args, preprocess_dtype=dtype) + return preprocess + + +def prepare( + dataset_manifest, + data_root, + export_manifest, + output, + split, + count=None, + seed=42, + batch_size=8, + source_levels=None, + level_names=None, + classifier_module=None, + preprocess_factory="dev.benchmarks.prepare_inputs:repository_preprocess", + factory_args=None, + threads=1, + score_semantics=None, +): + import torch + + from mini_trainer.data import get_inference_dataloader + + if batch_size < 1 or threads < 1: + raise ValueError("Require positive batch size and thread count") + dataset_bytes, export_bytes = Path(dataset_manifest).read_bytes(), Path(export_manifest).read_bytes() + dataset, exported = json.loads(dataset_bytes), json.loads(export_bytes) + records = select_records(dataset, split, count, seed) + heads = exported["classifiers"] + if classifier_module is not None: + heads = [head for head in heads if head["module"] == classifier_module] + if len(heads) != 1: + raise ValueError("Select one unambiguous classifier metadata entry with classifier_module") + metadata = heads[0]["metadata"] + classes = ordered_classes(metadata["cls2idx"]) + if score_semantics is None: + raise ValueError("Declare logits/probabilities score semantics explicitly") + semantics = [score_semantics] if isinstance(score_semantics, str) else list(score_semantics) + if len(semantics) == 1: + semantics *= len(classes) + if len(semantics) != len(classes) or any(s not in ("logits", "probabilities") for s in semantics): + raise ValueError("Supply score semantics for every exported level") + dataset_classes = ordered_classes(dataset["class_spec"]["cls2idx"]) + source_levels = list(range(len(classes))) if source_levels is None else source_levels + if ( + len(source_levels) != len(classes) + or len(set(source_levels)) != len(source_levels) + or any(type(i) is not int or not 0 <= i < len(dataset_classes) for i in source_levels) + ): + raise ValueError("Supply one distinct valid source level per exported classifier level") + if [dataset_classes[i] for i in source_levels] != classes: + raise ValueError("Export class ordering differs from the selected dataset levels") + names = [f"level_{i}" for i in range(len(classes))] if level_names is None else level_names + if len(names) != len(classes) or len(set(names)) != len(names) or any(not isinstance(n, str) or not n for n in names): + raise ValueError("Supply unique names for every exported level") + outputs = exported["outputs"] + if len(outputs) != len(classes): + raise ValueError("Export outputs do not match the selected classifier's levels") + if exported["preprocessing"]["in_graph"]: + raise ValueError("This preparation command expects preprocessing outside the graph") + input_spec = exported["input"] + if input_spec["dtype"] not in ("float16", "float32", "float64"): + raise ValueError("This image preparation path requires a NumPy-compatible floating ONNX input") + if "resize_size" not in metadata: + raise ValueError("Export metadata must declare the repository image-reader resize_size") + root = Path(data_root).resolve() + paths = [(root / record["path"]).resolve() for record in records] + if any(not path.is_relative_to(root) for path in paths): + raise ValueError("Source images must remain under data_root") + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "dataset_manifest_sha256": hashlib.sha256(dataset_bytes).hexdigest(), + "export_manifest_sha256": hashlib.sha256(export_bytes).hexdigest(), + "export_source": exported.get("source"), + "classifier_metadata": metadata, + "selection": { + "split": split, + "count": count, + "seed": seed, + "method": "manifest order" if count is None else "random.Random(seed).sample", + }, + "settings": {"batch_size": batch_size, "threads": threads, "workers": 0, "source_levels": source_levels}, + "source_records": records, + "batches": [], + "scope": "Explicit image preprocessing and input artifacts; not calibration, model inference or target-hardware acceptance.", + } + previous_threads = torch.get_num_threads() + try: + torch.set_num_threads(threads) + for path, record in zip(paths, records, strict=True): + if file_hash(path) != record["sha256"]: + raise ValueError(f"Source image hash mismatch: {path}") + factory = preprocess_factory + if isinstance(factory, str): + module, name = factory.split(":", 1) + factory = getattr(importlib.import_module(module), name) + factory_source = inspect.getsourcefile(factory) + report["preprocessing"] = { + "factory": f"{factory.__module__}:{factory.__qualname__}", + "factory_args": factory_args or {}, + "factory_source_sha256": file_hash(factory_source) if factory_source else None, + "torch": torch.__version__, + "numpy": np.__version__, + } + with seeded(seed): + preprocess = factory(metadata, **(factory_args or {})) + report["preprocessing"]["description"] = repr(preprocess) + _, loader = get_inference_dataloader( + images=[str(path) for path in paths], + resize_size=metadata["resize_size"], + batch_size=batch_size, + num_workers=0, + device="cpu", + dtype=torch.float32, + ) + samples = [] + for index, record in enumerate(records): + targets = record["targets"] + labels = [] + for classes_at_level, source_level in zip(classes, source_levels, strict=True): + target = targets[source_level] + if type(target) is not int or not 0 <= target < len(classes_at_level): + raise ValueError("Dataset target index is outside the declared class mapping") + labels.append(classes_at_level[target]) + samples.append({"instance_id": index, "filename": record["path"], "labels": labels}) + offset = 0 + with torch.inference_mode(), seeded(seed): + for index, images in enumerate(loader): + transformed = preprocess(images) + if not isinstance(transformed, torch.Tensor): + raise ValueError("Image preprocessing must return one tensor for the exported image input") + array = transformed.detach().cpu().to(getattr(torch, input_spec["dtype"])).numpy() + shape = input_spec["shape"] + if array.ndim != len(shape) or any(type(n) is int and n != actual for n, actual in zip(shape, array.shape, strict=True)): + raise ValueError("Prepared tensor shape does not match the exported input") + if len(array) != len(images) or not np.isfinite(array).all(): + raise ValueError("Preprocessing must preserve batch identity and produce finite values") + path = output / f"batch-{index:05d}.npz" + np.savez(path, **{input_spec["name"]: array}) + ids = [str(i) for i in range(offset, offset + len(array))] + report["batches"].append({"path": path.name, "sha256": file_hash(path), "sample_ids": ids}) + offset += len(array) + if offset != len(records): + raise ValueError("Preparation did not cover all selected records") + for path, record in zip(paths, records, strict=True): + if file_hash(path) != record["sha256"]: + raise ValueError(f"Source image changed during preparation: {path}") + manifest = { + "schema_version": 1, + "split": split, + "provenance": { + key: report[key] + for key in ( + "dataset_manifest_sha256", + "export_manifest_sha256", + "export_source", + "selection", + "preprocessing", + "runner_sha256", + ) + }, + "batch_input": input_spec["name"], + "batches": report["batches"], + "levels": [ + {"name": name, "classes": cls, "output": binding["name"], "score_semantics": semantics[i]} + for i, (name, cls, binding) in enumerate(zip(names, classes, outputs, strict=True)) + ], + "samples": samples, + } + manifest_path = output / "manifest.json" + manifest_path.write_text(json.dumps(manifest, indent=2) + "\n") + (calibration_manifest if split == "train" else inference_manifest)(manifest_path) + report.update(status="prepared", manifest_sha256=file_hash(manifest_path)) + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + torch.set_num_threads(previous_threads) + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--dataset-manifest", type=Path, required=True) + parser.add_argument("--data-root", type=Path, required=True) + parser.add_argument("--export-manifest", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + parser.add_argument("--split", choices=["train", "val", "test"], required=True) + parser.add_argument("--count", type=int) + parser.add_argument("--seed", type=int, default=42) + parser.add_argument("--batch-size", type=int, default=8) + parser.add_argument("--source-levels", type=int, nargs="+") + parser.add_argument("--level-names", nargs="+") + parser.add_argument("--classifier-module") + parser.add_argument("--preprocess-factory", default="dev.benchmarks.prepare_inputs:repository_preprocess") + parser.add_argument("--factory-args", type=json.loads) + parser.add_argument("--threads", type=int, default=1) + parser.add_argument("--score-semantics", choices=["logits", "probabilities"], nargs="+", required=True) + prepare(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 86132ca..e26b68a 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1845,6 +1845,51 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. +### Maintained image preparation reproduction + +`dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ +manifests from source images, the benchmark dataset inventory and export metadata. +It uses existing image decoding/resizing with zero workers and one CPU thread by +default. Class ordering, selected source levels, source hashes, declared split +separation, output bindings, tensor shape/dtype and explicit score semantics are +checked. A trusted preprocessing factory handles custom transforms; the default +uses existing architecture loaders without loading the trained classifier head. +See the [commands and scope](../dev/benchmarks/README.md#maintained-image-input-preparation). + +The real reproduction used the retained EfficientNetV2-S flat/hierarchical export +manifests and Blair source inventory. Calibration selection used +`random.Random(42).sample(train_records, 128)`; validation retained all 912 records +in manifest order. Source image hashes were checked before and after preparation. + +| Head / split | Images | Batches of eight | Retained NPZ file hashes reproduced | +| --- | ---: | ---: | --- | +| Flat / calibration train | 128 | 16 | Every batch | +| Flat / validation | 912 | 114 | Every batch | +| Hierarchical / calibration train | 128 | 16 | Every batch | +| Hierarchical / validation | 912 | 114 | Every batch | + +Calibration filenames and selection order matched the historical calibration +records. Validation samples, class labels and level/output bindings matched the +maintained collector's previous input manifests exactly. Calibration IDs are now +canonical integer strings; that metadata change does not change the input tensors. +The new files are retained in `tmp-prepared-real/{flat,hierarchical}-{train,val}/`, +with reports and reproduction checks. These exact input bytes already passed the +preceding calibration and full-dataset inference replays, so those downstream +GPU/CPU runs were not repeated for this preparation-only milestone. + +Ten focused regressions passed, including byte-reproducible preparation, a partial +final batch, class-order and source-hash failures, declared cross-split duplicates, +and preservation of caller RNG state. The default loader receives backbone +metadata without constructing a million-class trained head; the regression checks +that loader boundary without allocating such a head. This does not remove the +backbone construction cost of the current architecture getters or establish +reproduction of an unrecorded custom preprocessing transform. Continuous job +orchestration and target-hardware verification remain outstanding. + +Static checks and the full CPU-default suite passed: 458 passed, 152 skipped and +the known EMA expected failure. No new GPU correctness or performance claim is +made by this input-preparation milestone. + ### Maintained full-dataset inference reproduction `dev.benchmarks.dataset_inference` now collects complete held-out prediction diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 9bcc8d2..9c64f2d 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,7 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; automated source-dataset preparation and orchestration | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | @@ -56,9 +56,13 @@ production acceptance study. evaluates all five requested mini_metrics metrics. Maintained [full-dataset collection](../dev/benchmarks/README.md#maintained-full-dataset-prediction-collection) now feeds that evaluator for ONNX Runtime and TensorRT, with complete Blair - replays for both heads. Connect source-dataset preparation to the calibration - and held-out input contracts and automate the composed pipeline. Input batches - for the real-data replay were still prepared with a local script. + replays for both heads. Maintained + [image preparation](../dev/benchmarks/README.md#maintained-image-input-preparation) + now reproduces every retained calibration and validation NPZ hash for both + heads from source images and export metadata. Automate the composed pipeline + and package its explicit optional environments. The default preparation factory + uses current architecture-loader transforms; custom preprocessing still needs + an explicit reviewed factory and input verification. Preserve calibration records, class/preprocessing contracts, hashes, failures and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` diff --git a/tests/test_benchmark_prepare_inputs.py b/tests/test_benchmark_prepare_inputs.py new file mode 100644 index 0000000..7033152 --- /dev/null +++ b/tests/test_benchmark_prepare_inputs.py @@ -0,0 +1,132 @@ +import json +import random + +import numpy as np +import pytest +import torch +from PIL import Image + +from dev.benchmarks.onnx_inference import file_hash +from dev.benchmarks.prepare_inputs import ordered_classes, prepare, repository_preprocess, seeded, select_records + + +def identity_factory(metadata): + return lambda images: images.float() / 255 + + +@pytest.fixture +def example(tmp_path): + root = tmp_path / "images" + root.mkdir() + records = [] + for i, split in enumerate(["train"] * 4 + ["val"] * 3 + ["test"] * 2): + path = root / f"{i}.png" + Image.fromarray(np.full((3, 4, 3), i * 20, dtype=np.uint8)).save(path) + records.append({"path": path.name, "sha256": file_hash(path), "split": split, "targets": [i % 2, i % 2]}) + classes = {"a": 0, "b": 1} + dataset = {"schema_version": 1, "class_spec": {"cls2idx": {"0": classes, "1": classes}}, "records": records} + export = { + "schema_version": 1, + "source": {"checkpoint_sha256": "fixture"}, + "input": {"name": "images", "dtype": "float32", "shape": ["batch", 3, 2, 2]}, + "outputs": [{"name": "scores"}], + "classifiers": [{"module": "head", "metadata": {"cls2idx": classes, "resize_size": 2, "backbone_class": "unused"}}], + "preprocessing": {"in_graph": False}, + } + dataset_path, export_path = tmp_path / "dataset.json", tmp_path / "export.json" + dataset_path.write_text(json.dumps(dataset)) + export_path.write_text(json.dumps(export)) + return dataset_path, root, export_path, dataset, export + + +def test_preparation_repeats_exact_bytes_and_publishes_both_contracts(example, tmp_path): + dataset, root, export, _, _ = example + hashes = [] + for index in range(2): + out = tmp_path / f"train-{index}" + report = prepare( + dataset, root, export, out, "train", count=3, batch_size=2, preprocess_factory=identity_factory, score_semantics="logits" + ) + assert report["status"] == "prepared" + assert report["settings"]["workers"] == 0 + manifest = json.loads((out / "manifest.json").read_text()) + assert manifest["split"] == "train" + hashes.append([b["sha256"] for b in manifest["batches"]]) + assert [b["sample_ids"] for b in manifest["batches"]] == [["0", "1"], ["2"]] + assert hashes[0] == hashes[1] + result = prepare( + dataset, root, export, tmp_path / "val", "val", batch_size=2, preprocess_factory=identity_factory, score_semantics="logits" + ) + assert result["selection"]["method"] == "manifest order" + with np.load(tmp_path / "val/batch-00000.npz") as arrays: + np.testing.assert_array_equal(arrays["images"][:, 0, 0, 0], np.array([80, 100], dtype=np.float32) / 255) + with pytest.raises(FileExistsError): + prepare(dataset, root, export, tmp_path / "val", "val", preprocess_factory=identity_factory, score_semantics="logits") + + +def test_source_hash_failure_is_retained_before_any_batches(example, tmp_path): + dataset, root, export, _, _ = example + (root / "4.png").write_bytes(b"changed") + with pytest.raises(ValueError, match="Source image hash mismatch"): + prepare(dataset, root, export, tmp_path / "failed", "val", preprocess_factory=identity_factory, score_semantics="logits") + report = json.loads((tmp_path / "failed/report.json").read_text()) + assert report["status"] == "failed" and not report["batches"] + assert not (tmp_path / "failed/manifest.json").exists() + + +def test_cross_split_duplicate_bytes_are_rejected(example): + dataset = example[3] + dataset["records"][4]["sha256"] = dataset["records"][0]["sha256"] + with pytest.raises(ValueError, match="across splits"): + select_records(dataset, "train", None, 42) + + +def test_class_order_mismatch_and_implicit_semantics_are_rejected(example, tmp_path): + dataset, root, export, _, metadata = example + with pytest.raises(ValueError, match="semantics explicitly"): + prepare(dataset, root, export, tmp_path / "implicit", "val") + metadata["classifiers"][0]["metadata"]["cls2idx"] = {"b": 0, "a": 1} + export.write_text(json.dumps(metadata)) + with pytest.raises(ValueError, match="class ordering"): + prepare(dataset, root, export, tmp_path / "mismatch", "val", score_semantics="logits") + + +@pytest.mark.parametrize("mapping", [{"a": 1}, {"a": 0, "b": 0}, {"a": True}, {"1": {"a": 0}}]) +def test_invalid_class_indices(mapping): + with pytest.raises(ValueError): + ordered_classes(mapping) + + +def test_factory_uses_existing_loader_without_building_trained_head(monkeypatch): + from mini_trainer.modeling.architectures import load + + calls = [] + sentinel = object() + + def get_model(name, **kwargs): + calls.append((name, kwargs)) + return object(), "head", sentinel, 123, 224 + + monkeypatch.setattr(load, "get_model", get_model) + result = repository_preprocess( + {"backbone_class": "efficientnet_v2_s", "resize_size": 128, "preprocess_dtype": "float32", "out_features": 1000000} + ) + assert result is sentinel + assert calls == [ + ( + "efficientnet_v2_s", + {"model_args": {"pretrained": False, "local_files_only": True, "resize_size": 128}, "preprocess_dtype": torch.float32}, + ) + ] + + +def test_rng_context_is_repeatable_and_restores_callers(): + python_state, numpy_state, torch_state = random.getstate(), np.random.get_state(), torch.random.get_rng_state() + values = [] + for _ in range(2): + with seeded(42): + values.append((random.random(), np.random.rand(), torch.rand(()).item())) + assert values[0] == values[1] + assert random.getstate() == python_state + np.testing.assert_array_equal(np.random.get_state()[1], numpy_state[1]) + assert torch.equal(torch.random.get_rng_state(), torch_state) From 304f0f42842433ed86b5b3b9fc33776a5e1c7730 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 15:38:47 +0200 Subject: [PATCH 067/155] feat: materialize native INT8 checkpoints for deployment calibration --- docs/onnx.md | 42 +++++++++ mini_trainer/export.py | 22 ++++- mini_trainer/modeling/quantized_training.py | 75 ++++++++++++++++ tests/test_quantized_materialization.py | 97 +++++++++++++++++++++ 4 files changed, 234 insertions(+), 2 deletions(-) create mode 100644 tests/test_quantized_materialization.py diff --git a/docs/onnx.md b/docs/onnx.md index 64464f3..a3e6c40 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -171,3 +171,45 @@ separately; passing the default sample checks does not establish universal score parity or confidence-threshold equivalence. The generic exporter does not impose a model allowlist; configurations outside this tested coverage must pass the same export and parity checks before a bundle is published. + +## Explicit materialization for deployment calibration + +`mt_export --materialize-int8-training` exports a **separate floating model** from +a native INT8 training checkpoint. This is an opt-in route to subsequent static +calibration for runtimes that cannot execute the native dynamic quantizer. The +ordinary export path above remains unchanged. + +```bash +mt_export --weights native-int8-last.pt --output materialized-onnx \ + --input-shape 3 128 128 --materialize-int8-training +``` + +The conversion checks the recorded native recipes, copies the state, materializes +INT8 weight representations, and removes the recipe that would restore native +training tensor types. It preserves class metadata, active-class masks and other +buffers. Normalized directions use their integer codes with scale signs absorbed +into the floating magnitudes. This follows the native effective-weight formula +and handles zero scales without introducing an ordinary weight-normalization +divide-by-zero. Undefined zero-code directions and invalid/nonfinite states fail. +The source checkpoint is never rewritten, and neither optimizers nor training +state are carried into this deployment artifact. + +**Dynamic activation quantization is removed.** ONNX verification compares against +the materialized floating model, not against the native training forward. Its +manifest records the original checkpoint hash, the floating state hash and an +explicit `source.quantized_training_materialization` recipe, including +`dynamic_activation_quantization_preserved=false` and +`training_resume_supported=false`. The original checkpoint remains the source +for native training/resume. Materialization holds floating weights in memory; +it is not a training-memory optimization. + +Use the maintained [input preparation](../dev/benchmarks/README.md#maintained-image-input-preparation) +and [calibration](../dev/benchmarks/README.md#maintained-onnx-calibration-command) +commands on this artifact. For the tested TensorRT recipe, choose signed symmetric +activations, signed per-channel weights and floating biases, with training-only +calibration data. Build and inspect a new engine for its destination device. +Compare native predictions, materialized predictions and the calibrated deployment +candidate on the same held-out samples using all five requested mini_metrics +metrics. Use a practical FP16 baseline for efficiency comparisons; successful +materialization/export alone does not establish acceptable quality, integer GPU +execution or a worthwhile cost reduction. diff --git a/mini_trainer/export.py b/mini_trainer/export.py index b07c9fe..535f9ed 100644 --- a/mini_trainer/export.py +++ b/mini_trainer/export.py @@ -10,7 +10,7 @@ from mini_trainer.modeling import Classifier, classification_module from mini_trainer.modeling.architectures.load import get_dynamic_model, resolve_backbone_getter from mini_trainer.modeling.onnx import export_onnx -from mini_trainer.modeling.quantized_training import load_training_weights +from mini_trainer.modeling.quantized_training import load_training_weights, materialize_quantized_training_state def main( @@ -22,6 +22,7 @@ def main( dynamic_batch: bool = True, batch_size: int = 2, reference_device: str = "cpu", + materialize_int8_training: bool = False, ): if batch_size < 1: raise ValueError("batch_size must be positive.") @@ -29,6 +30,9 @@ def main( checkpoint_hash = hashlib.file_digest(handle, "sha256").hexdigest() state = load_training_weights(weights, map_location="cpu") state = state.get("model", state) + conversion = None + if materialize_int8_training: + state, conversion = materialize_quantized_training_state(state, dtype=torch.float32) metadata = Classifier.extract_metadata(state) model_type = metadata.get("backbone_class") if not model_type: @@ -46,7 +50,7 @@ def main( if not input_shape or any(size < 1 for size in input_shape): raise ValueError("input_shape must contain positive non-batch dimensions.") sample = torch.zeros(batch_size, *input_shape) - return export_onnx( + bundle = export_onnx( model, sample, output, @@ -55,6 +59,12 @@ def main( checkpoint_sha256=checkpoint_hash, reference_device=reference_device, ) + if conversion is not None: + manifest_path = bundle / "manifest.json" + manifest = json.loads(manifest_path.read_text()) + manifest["source"]["quantized_training_materialization"] = conversion + manifest_path.write_text(json.dumps(manifest, indent=2, allow_nan=False) + "\n") + return bundle def run(): @@ -67,6 +77,14 @@ def run(): parser.add_argument("--static-batch", action="store_false", dest="dynamic_batch", help="Export a fixed batch size.") parser.add_argument("--batch-size", type=int, default=2, help="Example/static batch size (default: 2).") parser.add_argument("--reference-device", default="cpu", help="PyTorch parity device; native INT8 training requires cuda.") + parser.add_argument( + "--materialize-int8-training", + action="store_true", + help=( + "Explicit floating export for subsequent calibration; " + "removes native dynamic activation quantization and requires quality validation." + ), + ) args = vars(parser.parse_args()) if args["preprocessing"] is not None: args["preprocessing"] = json.loads(args["preprocessing"].read_text()) diff --git a/mini_trainer/modeling/quantized_training.py b/mini_trainer/modeling/quantized_training.py index dc66dd9..14a8f46 100644 --- a/mini_trainer/modeling/quantized_training.py +++ b/mini_trainer/modeling/quantized_training.py @@ -164,3 +164,78 @@ def load_training_weights(path, *, map_location="cpu"): raise with torch.serialization.safe_globals([_backend().TrainingWeight]): return torch.load(path, map_location=map_location, weights_only=True) + + +def materialize_quantized_training_state(state_dict: dict, *, dtype=torch.float32): + """Create independent floating deployment weights from native INT8 state. + + This removes dynamic activation quantization and is not a training-resume + conversion or a promise of native forward parity. Normalized directions use + integer codes as direction and absorb scale signs into their magnitudes, + including zero scales. The caller's tensors and recipes remain untouched. + """ + import copy + from collections import OrderedDict + + if dtype not in (torch.float16, torch.bfloat16, torch.float32, torch.float64): + raise ValueError("Materialization requires a floating dtype") + recipes = {key: value for key, value in state_dict.items() if key == "_quantized_training" or key.endswith("._quantized_training")} + weights = {key: value for key, value in state_dict.items() if getattr(value, "_is_quantized_training", False)} + if not recipes or not weights: + raise ValueError("Expected native INT8 training weights and their recipes") + backend = _backend() + selected = set() + for key, recipe in recipes.items(): + if not isinstance(recipe, dict) or recipe.get("schema_version") != 1 or recipe.get("backend") != "cuda-int8-linear": + raise ValueError("Unsupported quantized training recipe for materialization") + prefix = key.removesuffix("_quantized_training") + for module in recipe["quantized_modules"]: + base = prefix + module + ("." if module else "") + matches = {base + "weight", base + "parametrizations.weight.original1"} & weights.keys() + if len(matches) != 1: + raise ValueError(f"Missing or ambiguous native weight for {base}") + selected.update(matches) + if selected != weights.keys() or any(not isinstance(value, backend.TrainingWeight) for value in weights.values()): + raise ValueError("Native weights do not match the recorded training recipes") + result = OrderedDict((key, copy.deepcopy(value)) for key, value in state_dict.items() if key not in recipes and key not in weights) + if hasattr(state_dict, "_metadata"): + result._metadata = copy.deepcopy(state_dict._metadata) + converted = [] + for key, weight in weights.items(): + codes, scales = weight.int_data, weight.scale + if ( + codes.ndim != 2 + or not codes.numel() + or codes.dtype != torch.int8 + or scales.shape != (codes.shape[0],) + or not torch.isfinite(scales).all() + ): + raise ValueError(f"Invalid native weight representation: {key}") + values = codes.detach().to(dtype=dtype) + normalized = key.endswith("parametrizations.weight.original1") + if normalized: + magnitude_key = key.removesuffix("original1") + "original0" + magnitude = result.get(magnitude_key) + if not isinstance(magnitude, torch.Tensor) or magnitude.shape != (codes.shape[0], 1) or not torch.isfinite(magnitude).all(): + raise ValueError(f"Invalid normalization magnitude: {magnitude_key}") + if (torch.linalg.vector_norm(values.float(), dim=1) == 0).any(): + raise ValueError(f"Native normalization has an undefined zero-code direction: {key}") + result[magnitude_key] = magnitude.to(dtype=dtype) * scales.sign().to(dtype=dtype).view(-1, 1) + if not torch.isfinite(result[magnitude_key]).all(): + raise ValueError(f"Normalization magnitude overflows the materialization dtype: {key}") + else: + values.mul_(scales.to(dtype=dtype).view(-1, 1)) + if not torch.isfinite(values).all(): + raise ValueError(f"Represented weights overflow the materialization dtype: {key}") + result[key] = values + converted.append({"name": key, "shape": list(values.shape), "normalized": normalized, "source_compute_dtype": str(weight.dtype)}) + return result, { + "schema_version": 1, + "source_backend": "cuda-int8-linear", + "target": "floating_weights_for_deployment_calibration", + "dtype": str(dtype), + "converted_weights": converted, + "source_recipes": copy.deepcopy(recipes), + "dynamic_activation_quantization_preserved": False, + "training_resume_supported": False, + } diff --git a/tests/test_quantized_materialization.py b/tests/test_quantized_materialization.py new file mode 100644 index 0000000..bcebc82 --- /dev/null +++ b/tests/test_quantized_materialization.py @@ -0,0 +1,97 @@ +import copy +import json + +import pytest +import torch + +from mini_trainer.export import main as export_checkpoint +from mini_trainer.hierarchical.model import HierarchicalClassifier +from mini_trainer.modeling import Classifier, classification_module +from mini_trainer.modeling.quantized_training import materialize_quantized_training_state, prepare_quantized_training +from tests.test_integration_train import TinyMockModel + +pytest.importorskip("torchao") +pytest.importorskip("triton") + + +@pytest.mark.parametrize("normalized", [False, True]) +def test_materialization_preserves_represented_weights_and_source(normalized): + torch.manual_seed(9) + model = torch.nn.Linear(8, 3) + if normalized: + torch.nn.utils.parametrizations.weight_norm(model, dim=0) + prepare_quantized_training(model) + weight = model.parametrizations.weight.original1 if normalized else model.weight + with torch.no_grad(): + weight.scale[0].neg_() + weight.scale[1].zero_() + state = model.state_dict() + codes, scales = weight.int_data.clone(), weight.scale.clone() + source_recipe = copy.deepcopy(state["_quantized_training"]) + values, report = materialize_quantized_training_state(state) + restored = torch.nn.Linear(8, 3) + if normalized: + torch.nn.utils.parametrizations.weight_norm(restored, dim=0) + magnitude = state["parametrizations.weight.original0"].flatten() + expected = codes.float() * (scales.sign() * magnitude / codes.float().norm(dim=1)).view(-1, 1) + else: + expected = codes.float() * scales.view(-1, 1) + restored.load_state_dict(values, strict=True) + torch.testing.assert_close(restored.weight, expected, rtol=1e-6, atol=1e-7) + assert torch.isfinite(restored(torch.randn(2, 8))).all() + torch.testing.assert_close(weight.int_data, codes, rtol=0, atol=0) + torch.testing.assert_close(weight.scale, scales, rtol=0, atol=0) + assert state["_quantized_training"] == source_recipe + assert not report["dynamic_activation_quantization_preserved"] + assert not report["training_resume_supported"] + assert not any(getattr(v, "_is_quantized_training", False) for v in values.values()) + values["bias"].zero_() + assert not torch.equal(values["bias"], state["bias"]) + report["source_recipes"]["_quantized_training"]["quantized_modules"].append("changed") + assert state["_quantized_training"] == source_recipe + + +@pytest.mark.parametrize("failure", ["missing_recipe", "wrong_module", "orphan_weight", "bad_scale", "zero_direction"]) +def test_invalid_native_state_is_rejected(failure): + model = torch.nn.utils.parametrizations.weight_norm(torch.nn.Linear(8, 3), dim=0) + prepare_quantized_training(model) + state = model.state_dict() + weight = state["parametrizations.weight.original1"] + if failure == "missing_recipe": + state.pop("_quantized_training") + elif failure == "wrong_module": + state["_quantized_training"]["quantized_modules"] = ["unknown"] + elif failure == "orphan_weight": + state["extra.weight"] = weight + elif failure == "bad_scale": + weight.scale[0] = float("nan") + else: + weight.int_data[0].zero_() + with pytest.raises(ValueError): + materialize_quantized_training_state(state) + + +@pytest.mark.parametrize("head", [Classifier, HierarchicalClassifier]) +def test_masked_normalized_checkpoint_materializes_and_exports_on_cpu(tmp_path, head): + pytest.importorskip("onnx") + pytest.importorskip("onnxscript") + pytest.importorskip("onnxruntime") + kwargs = {"sparse_masks": [torch.tensor([0, 0, 1])]} if head is HierarchicalClassifier else {} + model, _ = head.build(model_type=TinyMockModel(), num_classes=3, hidden=True, normalized=True, **kwargs) + classification_module(model).set_active_features([0, 2]) + prepare_quantized_training(model) + path = tmp_path / "native.pt" + torch.save(model.state_dict(), path) + before = path.read_bytes() + output = export_checkpoint(str(path), str(tmp_path / "export"), input_shape=[3, 5, 5], materialize_int8_training=True) + assert path.read_bytes() == before + manifest = json.loads((output / "manifest.json").read_text()) + assert manifest["quantized_training_forward"] is False + conversion = manifest["source"]["quantized_training_materialization"] + assert len(conversion["converted_weights"]) == 2 + assert not conversion["dynamic_activation_quantization_preserved"] + assert manifest["outputs"][0]["example_shape"][-1] == 2 + assert len(manifest["outputs"]) == (2 if head is HierarchicalClassifier else 1) + assert manifest["verification"]["reference_device"] == "cpu" + with pytest.raises(ValueError, match="reference_device='cuda'"): + export_checkpoint(str(path), str(tmp_path / "native-export"), input_shape=[3, 5, 5]) From c810767a917481c209aefa2b9505bb909792a450 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 15:57:12 +0200 Subject: [PATCH 068/155] fix: preserve shared normalized parameters during INT8 materialization --- docs/onnx.md | 8 +++-- mini_trainer/modeling/quantized_training.py | 36 ++++++++++++++++++--- tests/test_quantized_materialization.py | 36 +++++++++++++++++++++ 3 files changed, 73 insertions(+), 7 deletions(-) diff --git a/docs/onnx.md b/docs/onnx.md index a3e6c40..7e09dae 100644 --- a/docs/onnx.md +++ b/docs/onnx.md @@ -187,10 +187,12 @@ mt_export --weights native-int8-last.pt --output materialized-onnx \ The conversion checks the recorded native recipes, copies the state, materializes INT8 weight representations, and removes the recipe that would restore native training tensor types. It preserves class metadata, active-class masks and other -buffers. Normalized directions use their integer codes with scale signs absorbed -into the floating magnitudes. This follows the native effective-weight formula +buffers. Normalized directions use signed integer codes, with magnitudes set to +zero for zero-scale rows. This follows the native effective-weight formula and handles zero scales without introducing an ordinary weight-normalization -divide-by-zero. Undefined zero-code directions and invalid/nonfinite states fail. +divide-by-zero. Compatible parameter ties are retained by normal model loading; +incompatible tied roles/views fail instead of silently loading different values +into one shared parameter. Undefined zero-code directions and invalid/nonfinite states fail. The source checkpoint is never rewritten, and neither optimizers nor training state are carried into this deployment artifact. diff --git a/mini_trainer/modeling/quantized_training.py b/mini_trainer/modeling/quantized_training.py index 14a8f46..93475e9 100644 --- a/mini_trainer/modeling/quantized_training.py +++ b/mini_trainer/modeling/quantized_training.py @@ -171,8 +171,8 @@ def materialize_quantized_training_state(state_dict: dict, *, dtype=torch.float3 This removes dynamic activation quantization and is not a training-resume conversion or a promise of native forward parity. Normalized directions use - integer codes as direction and absorb scale signs into their magnitudes, - including zero scales. The caller's tensors and recipes remain untouched. + signed integer codes; zero scales require zero magnitudes. Incompatible tied + parameter roles fail explicitly. Caller tensors and recipes remain untouched. """ import copy from collections import OrderedDict @@ -200,7 +200,7 @@ def materialize_quantized_training_state(state_dict: dict, *, dtype=torch.float3 result = OrderedDict((key, copy.deepcopy(value)) for key, value in state_dict.items() if key not in recipes and key not in weights) if hasattr(state_dict, "_metadata"): result._metadata = copy.deepcopy(state_dict._metadata) - converted = [] + converted, touched = [], set(weights) for key, weight in weights.items(): codes, scales = weight.int_data, weight.scale if ( @@ -220,7 +220,10 @@ def materialize_quantized_training_state(state_dict: dict, *, dtype=torch.float3 raise ValueError(f"Invalid normalization magnitude: {magnitude_key}") if (torch.linalg.vector_norm(values.float(), dim=1) == 0).any(): raise ValueError(f"Native normalization has an undefined zero-code direction: {key}") - result[magnitude_key] = magnitude.to(dtype=dtype) * scales.sign().to(dtype=dtype).view(-1, 1) + signs = scales.sign().to(dtype=dtype) + values.mul_(torch.where(signs == 0, 1, signs).view(-1, 1)) + result[magnitude_key] = magnitude.to(dtype=dtype) * (signs != 0).view(-1, 1) + touched.add(magnitude_key) if not torch.isfinite(result[magnitude_key]).all(): raise ValueError(f"Normalization magnitude overflows the materialization dtype: {key}") else: @@ -229,11 +232,36 @@ def materialize_quantized_training_state(state_dict: dict, *, dtype=torch.float3 raise ValueError(f"Represented weights overflow the materialization dtype: {key}") result[key] = values converted.append({"name": key, "shape": list(values.shape), "normalized": normalized, "source_compute_dtype": str(weight.dtype)}) + # State loading can re-tie parameters through the architecture constructor. + # Different converted values for the same source storage would silently make + # the last loaded alias override another operation's intended representation. + aliases = {} + for key, value in state_dict.items(): + if not isinstance(value, torch.Tensor) or not value.numel(): + continue + tensors = (value.int_data, value.scale) if key in weights else (value,) + storage = tuple((str(t.device), t.untyped_storage().data_ptr()) for t in tensors) + aliases.setdefault(storage, []).append(key) + for group in aliases.values(): + if len(group) < 2 or not touched.intersection(group): + continue + first = group[0] + for key in group[1:]: + a, b = state_dict[first], state_dict[key] + views = (a.int_data, b.int_data) if first in weights and key in weights else (a, b) + if ( + views[0].shape != views[1].shape + or views[0].stride() != views[1].stride() + or views[0].storage_offset() != views[1].storage_offset() + or not torch.equal(result[first], result[key]) + ): + raise ValueError(f"Materialization cannot preserve tied parameter roles/views: {group}") return result, { "schema_version": 1, "source_backend": "cuda-int8-linear", "target": "floating_weights_for_deployment_calibration", "dtype": str(dtype), + "normalization_parameterization": "signed codes; zero magnitude for zero scales", "converted_weights": converted, "source_recipes": copy.deepcopy(recipes), "dynamic_activation_quantization_preserved": False, diff --git a/tests/test_quantized_materialization.py b/tests/test_quantized_materialization.py index bcebc82..dc43afa 100644 --- a/tests/test_quantized_materialization.py +++ b/tests/test_quantized_materialization.py @@ -71,6 +71,42 @@ def test_invalid_native_state_is_rejected(failure): materialize_quantized_training_state(state) +def _shared_magnitude_model(): + model = torch.nn.ModuleDict({name: torch.nn.utils.parametrizations.weight_norm(torch.nn.Linear(8, 3), dim=0) for name in ("a", "b")}) + model["b"].parametrizations.weight.original0 = model["a"].parametrizations.weight.original0 + return model + + +def test_shared_magnitude_with_opposite_scale_signs_preserves_both_weights(): + model = _shared_magnitude_model() + prepare_quantized_training(model) + model["b"].parametrizations.weight.original1.scale.neg_() + expected = {name: model[name].weight.dequantize().detach().clone() for name in ("a", "b")} + state, _ = materialize_quantized_training_state(model.state_dict()) + restored = _shared_magnitude_model() + restored.load_state_dict(state, strict=True) + for name in expected: + torch.testing.assert_close(restored[name].weight, expected[name], rtol=1e-6, atol=1e-7) + + +def test_incompatible_shared_magnitude_zero_scale_fails_explicitly(): + model = _shared_magnitude_model() + prepare_quantized_training(model) + model["b"].parametrizations.weight.original1.scale[0].zero_() + with pytest.raises(ValueError, match="tied parameter"): + materialize_quantized_training_state(model.state_dict()) + + +def test_weight_shared_between_normalized_and_ordinary_roles_fails_explicitly(): + model = torch.nn.ModuleDict( + {"a": torch.nn.Linear(8, 3), "b": torch.nn.utils.parametrizations.weight_norm(torch.nn.Linear(8, 3), dim=0)} + ) + model["a"].weight = model["b"].parametrizations.weight.original1 + prepare_quantized_training(model) + with pytest.raises(ValueError, match="tied parameter"): + materialize_quantized_training_state(model.state_dict()) + + @pytest.mark.parametrize("head", [Classifier, HierarchicalClassifier]) def test_masked_normalized_checkpoint_materializes_and_exports_on_cpu(tmp_path, head): pytest.importorskip("onnx") From 138b4acf07131c37db2fe44fec77ee083ddd80c8 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 15:57:22 +0200 Subject: [PATCH 069/155] docs: report native INT8 checkpoint deployment results and remaining goals --- docs/benchmarks.md | 85 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 39 +++++++++++------ 2 files changed, 112 insertions(+), 12 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index e26b68a..29a15f2 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1845,6 +1845,91 @@ Ignored artifacts: `tmp-trt-probe/direct-{flat,hierarchical}-fp16/`, samples. This follow-up changes documentation only; all six benchmark processes and both full-validation cases completed with finite outputs. +### Native INT8 checkpoint to calibrated TensorRT deployment + +The explicit `mt_export --materialize-int8-training` path connects a trained +native INT8 checkpoint to floating export and subsequent static calibration. +It does not silently substitute the native export: the source checkpoint hash, +removed dynamic activation quantization and conversion recipe are recorded in +the manifest. See the [export contract](onnx.md#explicit-materialization-for-deployment-calibration). + +This study uses the actual native checkpoints behind the verified native ONNX +baselines: `tmp-normalization-blair/{flat,hierarchical}/training/weights/last.pt`. +Their hashes were checked against `tmp-native-int8-onnx/verified/{head}/manifest.json`, +and the source files remained unchanged. These are different training runs from +the initial floating-checkpoint TensorRT study above, so its engines and metrics +are not reused as baselines here. + +Both EfficientNetV2-S heads are normalized with symmetric hidden layers. The new +exports passed floating ONNX verification on CPU. The maintained preparation +command reproduced all retained NPZ hashes for the 128 training calibration +images and 912 held-out validation images. Percentile 99.9 calibration used the +same signed symmetric INT8 activation/per-channel weight, floating-bias recipe, +with asymmetric histograms and one CPU thread. Dynamic activation row quantization +from native training is absent from this new artifact. + +TensorRT 10.16.1.11 on the RTX 3080 Ti Laptop built both INT8 engines and matched +FP16 baselines from the same materialized models. Settings were profile 1–8, +128px inputs, builder optimization 1, workspace 1 GiB, and TF32 disabled. Detailed +inspection verified **170 INT8 convolutions and two INT8 head GEMMs in each INT8 +engine**, including INT8 inputs/weights or integer GEMM tactics. This establishes +integer GPU execution for the converted artifacts, not the native dynamic graph. + +All four stages were evaluated on all 912 validation images, using the maintained +collector and mini_metrics comparison with threshold zero and no tuning. Native +means the previously verified native integer ONNX graph on CPU; materialized +FP32 means ONNX CPU execution before recalibration. FP16 and INT8 mean direct +TensorRT execution. The metrics are: + +| Head / stage | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat / native | 0.769112 | 0.758984 | 0.796193 | 1.000000 | 0.810716 | +| Flat / materialized FP32 | 0.770778 | 0.762868 | 0.796316 | 1.000000 | 0.810397 | +| Flat / TensorRT FP16 | 0.772581 | 0.765220 | 0.797016 | 1.000000 | 0.811644 | +| Flat / TensorRT INT8 | 0.769551 | 0.762561 | 0.789897 | 1.000000 | 0.805218 | +| Hierarchical leaf / native | 0.730698 | 0.720275 | 0.799803 | 1.000000 | 0.787638 | +| Hierarchical leaf / materialized FP32 | 0.735466 | 0.725632 | 0.801459 | 1.000000 | 0.791476 | +| Hierarchical leaf / TensorRT FP16 | 0.732321 | 0.722396 | 0.800159 | 1.000000 | 0.789964 | +| Hierarchical leaf / TensorRT INT8 | 0.736413 | 0.730936 | 0.777745 | 1.000000 | 0.791877 | +| Hierarchical parent / native | 0.867051 | 0.844461 | 0.913180 | 1.000000 | 0.857871 | +| Hierarchical parent / materialized FP32 | 0.870255 | 0.847898 | 0.915910 | 1.000000 | 0.859540 | +| Hierarchical parent / TensorRT FP16 | 0.870844 | 0.848284 | 0.916666 | 1.000000 | 0.862149 | +| Hierarchical parent / TensorRT INT8 | 0.869788 | 0.849624 | 0.913274 | 1.000000 | 0.857367 | + +Materialization changes 5 flat, 9 hierarchical leaf and 4 parent predictions +against native. The final INT8 candidate changes 65, 64 and 21 respectively. +Relative to native, INT8 F1 changes are +0.044, +0.572 and +0.274 percentage points, +while hierarchical leaf precision drops 2.206 points. Relative to matched FP16, +the corresponding F1 changes are −0.303, +0.409 and −0.106 points, with a 2.241-point +hierarchical leaf precision drop. Small F1 gains in a single-checkpoint comparison +are not evidence that quantization generally improves model quality. + +INT8 engine sizes are 26,325,140 bytes (flat) and 26,492,548 bytes (hierarchical), +versus 46,455,028 and 46,154,172 for FP16. Reported context requirements are +5,586,944 versus 5,669,888 bytes. These are file/context observations, not total +runtime-memory savings. No new latency benchmark or target-hardware performance +claim is made; a worthwhile trade-off must still be verified against FP16 on +the intended workload and device. + +The conversion regressions cover ordinary/normalized weights, negative/zero scales, +source immutability, masked flat/hierarchical checkpoint exports and explicit native +default behavior. A further shared-magnitude regression exposed a sign-placement +issue; signs now belong to the directions, while zero-scale rows use zero +magnitudes. Compatible shared magnitudes work, and incompatible tied roles/views +fail explicitly. The two real checkpoints have no nonpositive normalized direction +scales, so their materialized parameter values are unchanged by this correction. +All 12 focused materialization tests passed. `bash dev/check.sh all` passed static +checks and 470 tests, with 152 skips and one known EMA expected failure. Those +CPU regression checks supplement the real TensorRT execution above; they do not +establish target GPU or ARM correctness. + +All artifacts are retained under `tmp-native-materialized/`: source-linked floating +exports, prepared inputs, calibrated QDQ graphs, detailed engine inspection, +per-batch scores, prediction CSVs and native/FP16 comparison reports. The experiment +provides a distinct, measured QT-checkpoint-to-INT8-GPU route. It does not establish +unchanged native quantization, distributed training, million-class capacity, +ARM execution or target-machine cost benefits. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 9c64f2d..39b9bbc 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the benchmark report](benchmarks.md#tensorrt-int8-versus-fp16-initial-paired-trade-off). +recorded in [the checkpoint conversion study](benchmarks.md#native-int8-checkpoint-to-calibrated-tensorrt-deployment). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -16,30 +16,41 @@ or integer operator count sufficient evidence of production readiness. | --- | --- | --- | | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | +| QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | | Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | | CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | Native training currently leaves convolutions, gradients and optimizer states -floating. Calibrated TensorRT inference is a separate recipe derived from floating -checkpoints; it does not prove that the native dynamic training quantizer exports -unchanged into integer GPU execution. DDP/FSDP remain unsupported for native QT. +floating. Calibrated TensorRT inference now also has an explicit route from trained +native INT8 checkpoints through floating materialization and static recalibration. +The initial float-checkpoint study and this new study are distinct. Neither proves +that the native dynamic training quantizer exports unchanged into integer GPU +execution. DDP/FSDP remain unsupported for native QT. EMA repair is explicitly deferred and is not a prerequisite for this goal. -The latest TensorRT comparison uses matched builder settings and three fresh +The earlier floating-checkpoint TensorRT timing comparison uses matched builder settings and three fresh paired processes per head. At batches 1 and 8, INT8 did not beat FP16 in local host latency. Engines are approximately 43% smaller, while reported execution context memory is only 1.46% lower; total runtime memory has not been measured reliably. CUDA-event durations conflicted with enclosing host timing and are excluded from device-only performance conclusions. -Against FP16, INT8 Macro-F1 changes are approximately −0.40 percentage points for +In that earlier study, against FP16, INT8 Macro-F1 changes are approximately −0.40 percentage points for the flat classifier, +0.09 for hierarchical leaves and −1.23 for parents. Recall and Theil's U also require consideration; a small leaf F1 improvement is not a general quality improvement. All five requested metrics are recorded in the benchmark report. These are single-checkpoint validation results, not a multi-seed production acceptance study. +The new native-checkpoint conversion study evaluates all 912 Blair validation +images at four stages: native integer ONNX, materialized floating ONNX, TensorRT +FP16 and TensorRT INT8. Against matched FP16, INT8 Macro-F1 changes are −0.303, ++0.409 and −0.106 percentage points for flat, hierarchical leaf and parent +outputs; hierarchical leaf precision drops 2.241 points. No new latency result +was measured for these checkpoints. Regression validation passed all static +checks and **470 tests**, with **152 skips** and **one known EMA expected failure**. + ## Remaining work, in practical order 1. **Make the latest experiments reproducible from a clean checkout.** Engine @@ -76,12 +87,16 @@ production acceptance study. separately. Optimize the dominant measured costs rather than assuming integer arithmetic is faster. Include full and frozen-backbone training/fine-tuning. -3. **Settle the trained-checkpoint-to-deployment contract.** Either provide a GPU - implementation preserving native dynamic quantization, or explicitly convert - and calibrate a native QT checkpoint into a distinct deployment artifact and - validate the resulting quality change. The existing float-checkpoint TensorRT - candidate does not close that loop. Keep model configurations generic and - verify normalized/masked/hierarchical output semantics and class ordering. +3. **Qualify the trained-checkpoint-to-deployment contract.** Explicit conversion + and calibration now connect the matched native QT checkpoints to distinct + TensorRT INT8 artifacts. Both real heads have full-data comparisons against + the native and materialized FP16 baselines; see the + [conversion study](benchmarks.md#native-int8-checkpoint-to-calibrated-tensorrt-deployment). + Extend qualification to the large-head/target profiles and verify worthwhile + efficiency. Keep normalized/masked/hierarchical semantics and class ordering + under regression coverage. Unsupported tied parameter roles/views must fail + explicitly. An exact native dynamic-quantizer GPU implementation remains a + different, unimplemented route, not a claim made by materialization. 4. **Close the training and target-hardware evidence gaps.** Run the paired training profiles on the intended A40/A100/B300-class GPU and AMD EPYC systems, From 6717dc20f665b11f0f3f6572e2faeec3d5d6e20c Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 16:08:11 +0200 Subject: [PATCH 070/155] feat: compose paired inference and metric evaluation in fresh processes --- dev/benchmarks/README.md | 42 ++++++ dev/benchmarks/inference_pair.py | 154 ++++++++++++++++++++++ docs/benchmarks.md | 30 +++++ docs/quantization-status.md | 9 +- tests/test_benchmark_dataset_inference.py | 39 ++++++ 5 files changed, 272 insertions(+), 2 deletions(-) create mode 100644 dev/benchmarks/inference_pair.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index fa15e71..2c86e8c 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1145,3 +1145,45 @@ collection manifests continue to support manually prepared multiple inputs. The default factory requires the repository's PyTorch/Torchvision environment. The resulting NPZs can be transferred to the lighter ONNX CPU collection environment without reconstructing preprocessing on the edge device. + +### Paired inference quality pipeline + +`dev.benchmarks.inference_pair` composes the maintained dataset collectors and +`mini_metrics` evaluator. Each stage runs in a fresh process, sequentially, so +baseline and candidate runtimes do not coexist in GPU memory. It accepts the +same prepared held-out manifest for both models and retains prediction CSVs, +source hashes, runtime reports, subprocess logs, five-metric comparisons and a +`summary.md` suitable for a CI job summary. It installs nothing. + +```bash +.venv/bin/python -m dev.benchmarks.inference_pair \ + --manifest prepared-val/manifest.json \ + --baseline float/model.onnx --candidate int8/model.onnx \ + --output paired-quality \ + --baseline-runtime '{"backend":"onnx","threads":1}' \ + --candidate-runtime '{"backend":"onnx","threads":1}' \ + --metrics-python /path/to/metrics-env/bin/python +``` + +Use `{"backend":"tensorrt","python":"/path/to/gpu-env/bin/python"}` for an +engine artifact, independently on either side. ONNX options include `provider`, +`provider_options`, `optimization` and `threads`; TensorRT accepts `device`. +`save_scores: true` retains batch score arrays when needed. Interpreter paths +must name actual executables; virtual-environment symlinks are preserved. All +input paths are resolved from the calling directory. Child commands run from +this checkout, inheriting the explicitly prepared environment, with +`PYTHONHASHSEED=0` for repeatable metric evaluation. No local sibling package is +required; install a compatible `mini_metrics` in the chosen metrics environment. + +An existing output directory is rejected. A failed stage stops the pipeline, +returns a nonzero exit status and retains its log and the top-level failure +report; later stages do not run. Use a new output directory for retries. Reports +record metric deltas in units of 0–1, not percentage points, and preserve undefined +metrics. `evaluated` means collection and metric calculation completed, not that +the candidate passed a production gate. + +This command composes **quality evaluation only**. Preparation, calibration, +engine builds, integer-placement inspection and paired performance measurement +remain separate maintained commands. In particular, collector process duration +is not a latency benchmark. Run the ONNX recipe on the actual ARM device and the +TensorRT recipe on the intended GPU before making target support claims. diff --git a/dev/benchmarks/inference_pair.py b/dev/benchmarks/inference_pair.py new file mode 100644 index 0000000..a3a62cf --- /dev/null +++ b/dev/benchmarks/inference_pair.py @@ -0,0 +1,154 @@ +"""Run paired held-out inference and mini_metrics evaluation in fresh processes.""" + +import json +import os +import subprocess +import sys +from argparse import ArgumentParser +from pathlib import Path + +from .onnx_inference import file_hash + + +def run_pair(manifest, baseline, candidate, output, baseline_runtime=None, candidate_runtime=None, metrics_python=sys.executable): + """Compose maintained collectors without loading either runtime in this process. + + Runtime dictionaries accept a Python executable and collector options. Paths + are resolved from the caller's directory; child commands run from the checkout. + No dependencies are installed and existing output directories are never reused. + """ + manifest = Path(manifest).resolve(strict=True) + models = [Path(path).resolve(strict=True) for path in (baseline, candidate)] + commands = [] + output = Path(output).resolve() + for name, model, runtime in zip(("baseline", "candidate"), models, (baseline_runtime, candidate_runtime), strict=True): + options = dict(runtime or {}) + python = str(Path(options.pop("python", sys.executable)).absolute()) + allowed = {"backend", "provider", "provider_options", "threads", "optimization", "device", "save_scores"} + if options.keys() - allowed: + raise ValueError(f"Unknown {name} runtime options: {sorted(options.keys() - allowed)}") + command = [ + python, + "-m", + "dev.benchmarks.dataset_inference", + "--model", + str(model), + "--manifest", + str(manifest), + "--output", + str(output / name), + ] + for key, value in options.items(): + if key == "save_scores": + if not isinstance(value, bool): + raise ValueError("save_scores must be a boolean") + if value: + command.append("--save-scores") + else: + command.extend(["--" + key.replace("_", "-"), json.dumps(value) if key == "provider_options" else str(value)]) + if name == "candidate": + command.extend(["--baseline-bundle", str(output / "baseline/evaluation.json")]) + commands.append((name, command, "inferred")) + commands.append( + ( + "quality", + [ + str(Path(metrics_python).absolute()), + "-m", + "dev.benchmarks.quality_compare", + "--manifest", + str(output / "candidate/comparison.json"), + "--output", + str(output / "quality"), + ], + "evaluated", + ) + ) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "manifest": {"path": str(manifest), "sha256": file_hash(manifest)}, + "python_hash_seed": "0", + "stages": [], + "scope": "Paired held-out quality only; not timing, integer placement or production acceptance.", + } + report_path = output / "report.json" + + def save(): + report_path.write_text(json.dumps(report, indent=2) + "\n") + + try: + for name, command, expected in commands: + stage = {"name": name, "command": command, "status": "running", "log": f"{name}.log"} + report["stages"].append(stage) + save() + with (output / stage["log"]).open("w") as log: + result = subprocess.run( + command, + cwd=Path(__file__).resolve().parents[2], + env={**os.environ, "PYTHONHASHSEED": "0"}, + stdout=log, + stderr=subprocess.STDOUT, + check=False, + ) + stage["returncode"] = result.returncode + if result.returncode: + stage["status"] = "failed" + raise RuntimeError(f"{name} failed with exit code {result.returncode}; see {output / stage['log']}") + child_path = output / name / "report.json" + child = json.loads(child_path.read_text()) + if child["status"] != expected: + raise RuntimeError(f"Unexpected {name} report status: {child['status']}") + stage.update(status=expected, report_sha256=file_hash(child_path)) + save() + report.update(status="evaluated", levels=child["levels"], models=child["models"], undefined_metrics=child["undefined_metrics"]) + lines = [ + "# Paired inference quality", + "", + report["scope"], + "", + "| Level | Samples | Changed predictions | F1 delta | Recall delta | Precision delta | Coverage delta | Theil U delta |", + "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |", + ] + for level in child["levels"]: + values = [level["candidate_minus_baseline"][metric] for metric in ("f1", "recall", "precision", "coverage", "theilU")] + label = str(level["name"]).replace("|", "\\|").replace("\n", " ").replace("\r", " ") + lines.append( + "| " + + " | ".join( + [ + label, + str(level["samples"]), + str(level["prediction_changes"]), + *["undefined" if v is None else f"{v:+.6f}" for v in values], + ] + ) + + " |" + ) + lines.extend(["", "Deltas are candidate minus baseline in metric units. Undefined values are not passes.", ""]) + (output / "summary.md").write_text("\n".join(lines)) + except BaseException as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + for stage in report["stages"]: + if stage["status"] == "running": + stage["status"] = "failed" + raise + finally: + save() + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + for name in ("manifest", "baseline", "candidate", "output"): + parser.add_argument("--" + name, type=Path, required=True) + for name in ("baseline", "candidate"): + parser.add_argument(f"--{name}-runtime", type=json.loads, help="JSON collector options and optional Python executable path") + parser.add_argument("--metrics-python", default=sys.executable, help="Explicit evaluation environment Python; installs nothing") + run_pair(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 29a15f2..6efee1d 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1930,6 +1930,36 @@ provides a distinct, measured QT-checkpoint-to-INT8-GPU route. It does not estab unchanged native quantization, distributed training, million-class capacity, ARM execution or target-machine cost benefits. +### Paired inference quality orchestration + +`dev.benchmarks.inference_pair` now composes baseline collection, candidate +collection and the five-metric comparison in sequential fresh processes. Separate +Python interpreters can supply the inference and metric dependencies. It retains +commands, logs, child report hashes, failure status and a Markdown summary, and +rejects existing output directories. See the +[command documentation](../dev/benchmarks/README.md#paired-inference-quality-pipeline). + +Four full 912-image Blair comparisons replayed the retained artifacts above: +native integer ONNX versus materialized ONNX on CPU, and TensorRT FP16 versus +INT8, each for flat and hierarchical heads. Both CPU pairs and the flat TensorRT +pair reproduced prediction CSVs byte-for-byte. The hierarchical TensorRT baseline +had one confidence value differ by 4.28e-12; all other CSV fields and the candidate +CSV were unchanged. All four pairs reproduced every discrete prediction, metric +value and metric delta exactly. This is orchestration reproducibility evidence, +not a new quality improvement, timing result or target-hardware qualification. + +Ignored outputs are `tmp-inference-pair-{flat,hierarchical}/` and +`tmp-inference-pair-trt-{flat,hierarchical}/`. The replays reuse existing models, +engines and prepared inputs and do not duplicate them or retain full score arrays. +The small hierarchical oracle also exercises the real subprocess pipeline; +failure and output-reuse regressions verify that incomplete runs cannot report +successful evaluation. Preparation, calibration, engine building, placement and +performance still need orchestration before this is a complete deployment pipeline. +Validation: `bash dev/check.sh all` passed static checks and 473 tests, with +152 skips and one known EMA expected failure. The focused collector/pipeline +suite passed 12 tests with one optional CUDA skip. The real TensorRT replays +were separate intentional GPU runs; ordinary CPU checks do not establish GPU support. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 39b9bbc..67ea2d6 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -70,8 +70,13 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur replays for both heads. Maintained [image preparation](../dev/benchmarks/README.md#maintained-image-input-preparation) now reproduces every retained calibration and validation NPZ hash for both - heads from source images and export metadata. Automate the composed pipeline - and package its explicit optional environments. The default preparation factory + heads from source images and export metadata. A + [paired quality pipeline](../dev/benchmarks/README.md#paired-inference-quality-pipeline) + now runs both collectors and the evaluator in fresh processes, retaining logs, + failure reports and a Markdown summary, with separately selectable runtime and + metric interpreters. Compose preparation, calibration, builds, placement and + performance checks with it, and package the explicit optional environments. + The default preparation factory uses current architecture-loader transforms; custom preprocessing still needs an explicit reviewed factory and input verification. Preserve calibration records, class/preprocessing contracts, hashes, failures diff --git a/tests/test_benchmark_dataset_inference.py b/tests/test_benchmark_dataset_inference.py index e8575a3..90b7ef4 100644 --- a/tests/test_benchmark_dataset_inference.py +++ b/tests/test_benchmark_dataset_inference.py @@ -7,6 +7,7 @@ import pytest from dev.benchmarks.dataset_inference import collect, inference_manifest, pair_bundle, predictions +from dev.benchmarks.inference_pair import run_pair from dev.benchmarks.quality_compare import read_manifest, read_predictions @@ -74,6 +75,44 @@ def test_score_semantics_and_shape_are_explicit(): predictions(values, probability, 1) +def test_pair_pipeline_runs_real_children_and_preserves_evidence(example, tmp_path): + pytest.importorskip("onnxruntime") + pytest.importorskip("mini_metrics") + model, manifest, _ = example + output = tmp_path / "paired" + report = run_pair(manifest, model, model, output, candidate_runtime={"save_scores": True}) + assert report["status"] == "evaluated" + assert [stage["status"] for stage in report["stages"]] == ["inferred", "inferred", "evaluated"] + assert report["stages"][0]["command"][0] == sys.executable + assert all(level["prediction_changes"] == 0 for level in report["levels"]) + assert all(value == 0 for level in report["levels"] for value in level["candidate_minus_baseline"].values()) + assert (output / "candidate/scores-00001.npz").is_file() + assert "Theil U delta" in (output / "summary.md").read_text() + with pytest.raises(FileExistsError): + run_pair(manifest, model, model, output) + + +def test_pair_pipeline_retains_failure_and_stops_before_candidate(example, tmp_path): + model, manifest, _ = example + output = tmp_path / "failed-pair" + with pytest.raises(RuntimeError, match="baseline failed"): + run_pair(manifest, model, model, output, baseline_runtime={"backend": "invalid"}) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" + assert len(report["stages"]) == 1 + assert report["stages"][0]["returncode"] != 0 + assert (output / "baseline.log").stat().st_size > 0 + assert not (output / "candidate").exists() + + +def test_pair_pipeline_rejects_unknown_options_before_creating_output(example, tmp_path): + model, manifest, _ = example + output = tmp_path / "invalid-pair" + with pytest.raises(ValueError, match="Unknown candidate"): + run_pair(manifest, model, model, output, candidate_runtime={"typo": True}) + assert not output.exists() + + def test_cpu_collection_handles_multiple_inputs_levels_and_partial_batch(example, tmp_path): pytest.importorskip("onnxruntime") model, manifest, metadata = example From 58db1979397d7964b31805c1eae1dd1c99ad9b6c Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 16:22:19 +0200 Subject: [PATCH 071/155] feat: measure isolated ONNX CPU memory and inference trade-offs --- dev/benchmarks/README.md | 49 ++++++++++ dev/benchmarks/onnx_cpu_memory.py | 147 ++++++++++++++++++++++++++++++ docs/benchmarks.md | 75 +++++++++++++++ docs/quantization-status.md | 8 +- tests/test_benchmark_onnx.py | 63 +++++++++++++ 5 files changed, 341 insertions(+), 1 deletion(-) create mode 100644 dev/benchmarks/onnx_cpu_memory.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 2c86e8c..2d3dca3 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1187,3 +1187,52 @@ engine builds, integer-placement inspection and paired performance measurement remain separate maintained commands. In particular, collector process duration is not a latency benchmark. Run the ONNX recipe on the actual ARM device and the TensorRT recipe on the intended GPU before making target support claims. + +### Single-process ONNX CPU memory probe + +Use `dev.benchmarks.onnx_cpu_memory` on Linux, including the intended ARM device, +with **one model in each fresh interpreter process**. It measures a CPU-only, +unprofiled session; it does not import PyTorch or `mini_trainer`. NumPy, ONNX and +ONNX Runtime must already be installed for that architecture. + +```bash +.venv/bin/python -m dev.benchmarks.onnx_cpu_memory \ + --model float/model.onnx --inputs batch-one.npz --output cpu-float-trial-1 \ + --threads 1 --warmup 3 --repeats 31 +.venv/bin/python -m dev.benchmarks.onnx_cpu_memory \ + --model int8/model.onnx --inputs batch-one.npz --output cpu-int8-trial-1 \ + --threads 1 --warmup 3 --repeats 31 +``` + +Repeat in new processes and new directories at least three times, reversing the +baseline/candidate order on alternate trials. Keep input hashes, thread counts, +CPU affinity and runtime settings matched. Run timing trials without competing +benchmarks. A fresh process does not imply cold filesystem caches; the command +neither evicts caches nor changes the machine's CPU governor, affinity or cooling. +For sustained edge claims, also run longer trials and record target power/cooling +conditions and throttling. The current command does not collect thermal telemetry. + +Reports retain all warm timing samples, session-construction and first-inference +times, effective providers, CPU affinity, runtime versions/build, source/input +hashes and named output arrays. Memory snapshots cover runtime import, input +loading, session creation, first inference, warmup and the final measurement. +`resident_bytes` and `proportional_resident_bytes` use Linux `smaps_rollup` RSS/PSS; +`peak_resident_bytes` uses the approximate `status` VmHWM high-water mark since +exec. Shared mapped pages contribute fully to RSS and proportionally to PSS. +See the [kernel proc documentation](https://docs.kernel.org/filesystems/proc.html) +for these accounting distinctions and the accuracy limits of status counters. + +The peak includes interpreter/import/input/output and validation allocations. It +is not a model-only allocation count, and subtracting two lifetime peaks does not +isolate model memory. `ru_maxrss` is deliberately not used: a local regression +showed that it retained a parent's pre-exec high-water mark in a fresh child, +while VmHWM described the child's new address space. Snapshot reads are outside +the inference timing intervals. ONNX provenance inspection and output serialization +happen after the last memory snapshot to avoid inflating the reported peak with +large inline graph tensors or archive-writing buffers. + +An existing output directory is rejected; execution failures retain partial +reports. `measured` is not deployment acceptance. Pair results with the maintained +quality evaluator and separate operator-placement probe. Timing excludes image IO +and preprocessing; CPU/x86 measurements do not establish ARM kernels, memory, +quality or speed, and this probe does not measure GPU memory. diff --git a/dev/benchmarks/onnx_cpu_memory.py b/dev/benchmarks/onnx_cpu_memory.py new file mode 100644 index 0000000..1c5cc65 --- /dev/null +++ b/dev/benchmarks/onnx_cpu_memory.py @@ -0,0 +1,147 @@ +"""Measure one ONNX CPU model per fresh Linux process, without profiling.""" + +import json +import os +import platform +import statistics +import time +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np + +from .onnx_inference import file_hash, model_files + + +def resident_memory(): + """Linux mapping snapshot and approximate post-exec high-water mark, in bytes.""" + if platform.system() != "Linux": + raise RuntimeError("This memory probe requires Linux /proc") + fields = {} + for path, names in (("status", ("VmRSS", "VmHWM")), ("smaps_rollup", ("Rss", "Pss", "Swap"))): + for line in Path(f"/proc/self/{path}").read_text().splitlines(): + name, _, value = line.partition(":") + if name in names: + amount, unit = value.split() + if unit != "kB": + raise RuntimeError(f"Unexpected /proc memory unit: {unit}") + fields[name] = int(amount) * 1024 + return { + "resident_bytes": fields["Rss"], + "proportional_resident_bytes": fields["Pss"], + "swap_bytes": fields["Swap"], + "approximate_status_resident_bytes": fields["VmRSS"], + "peak_resident_bytes": fields["VmHWM"], + } + + +def measure(model, inputs, output, threads=1, warmup=3, repeats=31, optimization="all"): + """Use the CLI in a fresh process: in-process calls inherit earlier RSS peaks.""" + if min(threads, warmup, repeats) < 1 or optimization not in ("all", "disable"): + raise ValueError("Positive threads/warmup/repeats and all/disable optimization are required") + model, inputs = Path(model).resolve(strict=True), Path(inputs).resolve(strict=True) + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "environment": {"platform": platform.platform(), "machine": platform.machine(), "python": platform.python_version()}, + "settings": {"threads": threads, "inter_op_threads": 1, "warmup": warmup, "repeats": repeats, "optimization": optimization}, + "memory": {}, + "memory_sources": { + "resident_bytes": "/proc/self/smaps_rollup Rss", + "proportional_resident_bytes": "/proc/self/smaps_rollup Pss", + "swap_bytes": "/proc/self/smaps_rollup Swap", + "peak_resident_bytes": "/proc/self/status VmHWM (approximate)", + }, + "scope": ( + "Single CPU session, preprocessed NumPy IO, no profiling. RSS includes interpreter, imports, inputs and outputs. " + "Peak is the approximate high-water mark since exec through each snapshot, not model-only memory. " + "Memory includes finite-input/output validation. Snapshot reads occur outside inference timings. " + "Run in a fresh process for each model/trial. Snapshots precede ONNX artifact inspection and output serialization. " + "Session load is not guaranteed disk-cold; inference excludes image decoding and preprocessing." + ), + } + try: + report["memory"]["before_runtime_import"] = resident_memory() + import onnxruntime as ort + + report["versions"] = {"onnxruntime": ort.__version__, "numpy": np.__version__} + report["runtime_build"] = ort.get_build_info() + report["environment"]["cpu_affinity"] = sorted(os.sched_getaffinity(0)) + report["memory"]["after_runtime_import"] = resident_memory() + with np.load(inputs, allow_pickle=False) as archive: + feeds = {name: np.ascontiguousarray(archive[name]) for name in archive.files} + if not feeds or any(not np.isfinite(value).all() for value in feeds.values()): + raise ValueError("Inputs must contain finite named arrays") + report["input_arrays"] = {name: {"shape": list(a.shape), "dtype": str(a.dtype)} for name, a in feeds.items()} + report["memory"]["before_session"] = resident_memory() + options = ort.SessionOptions() + options.intra_op_num_threads = threads + options.inter_op_num_threads = 1 + options.graph_optimization_level = getattr( + ort.GraphOptimizationLevel, "ORT_ENABLE_ALL" if optimization == "all" else "ORT_DISABLE_ALL" + ) + started = time.perf_counter() + session = ort.InferenceSession(str(model), sess_options=options, providers=["CPUExecutionProvider"]) + report["session_load_seconds"] = time.perf_counter() - started + session.disable_fallback() + report["session_providers"] = session.get_providers() + report["memory"]["after_session"] = resident_memory() + if report["session_providers"] != ["CPUExecutionProvider"]: + raise RuntimeError("Expected an exclusively CPU session") + if set(feeds) != {node.name for node in session.get_inputs()}: + raise ValueError("Input names do not match model") + started = time.perf_counter() + predictions = session.run(None, feeds) + report["first_run_seconds"] = time.perf_counter() - started + report["memory"]["after_first_run"] = resident_memory() + del predictions + for _ in range(warmup): + session.run(None, feeds) + report["memory"]["after_warmup"] = resident_memory() + report["seconds"] = [] + for _ in range(repeats): + started = time.perf_counter() + predictions = session.run(None, feeds) + report["seconds"].append(time.perf_counter() - started) + # Do not retain the previous outputs while allocating the next batch. + if any(not np.isfinite(value).all() for value in predictions): + raise ValueError("Nonfinite model outputs") + del predictions + report["memory"]["after_measurement"] = resident_memory() + report["median_seconds"] = statistics.median(report["seconds"]) + # Provenance inspection can deserialize large inline tensors. Do it only + # after capturing the measurement peak, so it cannot inflate that result. + import onnx + + report["versions"]["onnx"] = onnx.__version__ + report["model_files"] = model_files(model, onnx) + report["inputs"] = {"path": str(inputs), "sha256": file_hash(inputs)} + predictions = session.run(None, feeds) + names = [node.name for node in session.get_outputs()] + np.savez(output / "outputs.npz", **dict(zip(names, predictions, strict=True))) + report["outputs"] = {"names": names, "sha256": file_hash(output / "outputs.npz")} + report["status"] = "measured" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + for name in ("model", "inputs", "output"): + parser.add_argument("--" + name, type=Path, required=True) + parser.add_argument("--threads", type=int, default=1) + parser.add_argument("--warmup", type=int, default=3) + parser.add_argument("--repeats", type=int, default=31) + parser.add_argument("--optimization", choices=["all", "disable"], default="all") + measure(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 6efee1d..a0ec994 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1960,6 +1960,81 @@ Validation: `bash dev/check.sh all` passed static checks and 473 tests, with suite passed 12 tests with one optional CUDA skip. The real TensorRT replays were separate intentional GPU runs; ordinary CPU checks do not establish GPU support. +### Isolated Linux CPU memory and inference study + +`dev.benchmarks.onnx_cpu_memory` measures one CPU model per fresh Linux interpreter, +with no profiling session or other model loaded. It records session load, first +inference, all warm timings, and resident-memory snapshots before/after runtime +import, input loading, session creation, first inference, warmup and measurement. +The [command guide](../dev/benchmarks/README.md#single-process-onnx-cpu-memory-probe) +defines the memory accounting and target handoff procedure. + +A regression exposed inherited `ru_maxrss`: after a parent allocated 150 MiB, +its fresh child reported 167,700 KiB through `getrusage`, while `/proc/self/status` +reported a 13,404 KiB VmHWM for the child's new address space. The probe therefore +uses `smaps_rollup` RSS/PSS snapshots and approximate post-exec VmHWM, rather than +attributing a parent's earlier peak to model inference. It records memory before +ONNX graph provenance inspection and output serialization. Interpreter, runtime, +inputs/outputs and finite-value checks remain part of the measured process. + +The paired quality runner separately evaluated the materialized floating ONNX +and signed calibrated QDQ models from the native-checkpoint study on all 912 +Blair validation images, using ONNX Runtime CPU. These are CPU runtime results, +not reused TensorRT predictions. The five mini_metrics results are: + +| Head / stage | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat / float | 0.770778 | 0.762868 | 0.796316 | 1.000000 | 0.810397 | +| Flat / INT8 | 0.764407 | 0.756174 | 0.786411 | 1.000000 | 0.802939 | +| Hierarchical leaf / float | 0.735466 | 0.725632 | 0.801459 | 1.000000 | 0.791476 | +| Hierarchical leaf / INT8 | 0.738374 | 0.732758 | 0.774078 | 1.000000 | 0.795192 | +| Hierarchical parent / float | 0.870255 | 0.847898 | 0.915910 | 1.000000 | 0.859540 | +| Hierarchical parent / INT8 | 0.869179 | 0.848718 | 0.914401 | 1.000000 | 0.858741 | + +Flat F1 decreases 0.637 percentage points; hierarchical leaf F1 increases 0.291 +points while precision decreases 2.738 points. Parent F1 decreases 0.108 points. +The candidate changes 67 flat, 55 leaf and 16 parent predictions. Calibration +remains training-only; validation thresholds are fixed at zero without tuning. +Local quality artifacts are `tmp-edge-quality-{flat,hierarchical}/`. + +After all regression and quality processes finished, three fresh-process trials +per model used batch one, 128px inputs, one intra-op/inter-op thread, three warmup +runs and 31 measured runs. Float/INT8 order was reversed in the middle trial. +The machine was the local i7-12800H x86 WSL environment, ONNX Runtime 1.29; CPU +affinity remained 0–19, rather than a pinned core. Session construction and first +inference were recorded separately. No filesystem-cache eviction, thermal +control or ARM emulation was performed. + +| Head / model | Warm median ms, trials 1/2/3 | Final RSS MiB, trials 1/2/3 | Approximate peak RSS MiB, trials 1/2/3 | +| --- | --- | --- | --- | +| Flat / float | 29.277 / 26.685 / 26.726 | 192.270 / 196.605 / 192.512 | 193.691 / 198.086 / 193.980 | +| Flat / INT8 | 42.963 / 42.599 / 44.455 | 104.992 / 106.230 / 105.867 | 104.812 / 106.117 / 105.668 | +| Hierarchical / float | 27.914 / 38.539 / 27.565 | 193.785 / 190.727 / 192.156 | 195.164 / 192.207 / 193.547 | +| Hierarchical / INT8 | 45.976 / 40.411 / 45.793 | 105.406 / 105.633 / 105.980 | 105.359 / 105.395 / 105.926 | + +INT8 final resident memory was 44.6–46.0% lower across these pairs. Warm median +latency was higher in all pairs: candidate/baseline ratios 1.468/1.596/1.663 for +flat and 1.647/1.049/1.661 for hierarchical. These are separate-process median +ratios, not adjacent per-inference paired ratios. Reported swap was zero. The +approximate status high-water counter can be slightly below the more precise +smaps snapshot; retain both sources rather than treating them as identical +accounting. This demonstrates local process-memory savings, not a speed gain or +an ARM production result. + +Separate post-measurement profiling found **63 QLinearConv and two QGemm** +operations, with **107 floating Conv** operations, on CPU for each candidate. +The same signed QDQ/floating-bias recipe executed 170 integer convolutions under +TensorRT; its CPU execution is hybrid. A CPU-specific calibration/fusion study +is therefore the next useful step before target ARM qualification, rather than +assuming that the TensorRT recipe is also the best edge recipe. + +`tmp-edge-process-probe/` retains the trial driver, all raw timings/memory reports, +output arrays, batch-one input provenance and separate placement profiles. No +models or source datasets were duplicated. All 13 focused ONNX benchmark tests +passed; `bash dev/check.sh all` passed static checks and 476 tests, with 152 skips +and one known EMA expected failure. Hardware-specific validation, sustained-load +thermal behavior and end-to-end preprocessing costs remain unverified. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 67ea2d6..bfd2ced 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -18,7 +18,7 @@ or integer operator count sufficient evidence of production readiness. | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | | QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | | Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | -| CPU/edge inference | Calibrated ONNX CPU execution and quality measurements on x86; portable provider/timing runner | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | +| CPU/edge inference | Calibrated ONNX CPU execution and full Blair metrics on x86; isolated process-memory/timing probe; about 45% lower resident memory but slower batch-one inference for the tested candidate | CPU-specific recipe/fusion tuning; Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | Native training currently leaves convolutions, gradients and optimizer states @@ -118,6 +118,12 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur sustained throughput and process memory with explicit threads and preprocessing, and reevaluate all five metrics. Verify the actual runtime kernels and installed dependency set; neither x86 CPU results nor CUDA results establish ARM support. + The [isolated Linux CPU study](benchmarks.md#isolated-linux-cpu-memory-and-inference-study) + now supplies a maintained RSS/PSS/peak and load/first/warm inference probe. + Its local candidate saves resident memory but retains 107 floating convolutions + and is slower; investigate CPU-specific calibration/fusion before selecting the + edge recipe. Repeat on real ARM hardware with preprocessing and sustained-load + conditions, rather than extrapolating the local memory percentage. 6. **Turn the accepted trade-offs into continuous release evidence.** Select concrete quality and benefit thresholds for each supported deployment profile diff --git a/tests/test_benchmark_onnx.py b/tests/test_benchmark_onnx.py index 1a649b5..2f90f2b 100644 --- a/tests/test_benchmark_onnx.py +++ b/tests/test_benchmark_onnx.py @@ -1,4 +1,6 @@ import json +import subprocess +import sys import numpy as np import pytest @@ -110,3 +112,64 @@ def test_advertised_provider_with_only_cpu_execution_fails(model_and_inputs, tmp assert report["status"] == "failed" assert all(e["provider"] == "CPUExecutionProvider" for e in report["models"][0]["execution"]) assert report["models"][0]["seconds"] == [] + + +@pytest.mark.skipif(sys.platform != "linux", reason="Linux resident-memory probe") +@pytest.mark.parametrize("invalid", [False, True]) +def test_isolated_cpu_memory_probe_records_measurement_or_failure(model_and_inputs, tmp_path, invalid): + model, inputs = model_and_inputs + if invalid: + np.savez(inputs, wrong=np.ones((2, 2), dtype=np.float32)) + output = tmp_path / "memory" + command = [ + sys.executable, + "-m", + "dev.benchmarks.onnx_cpu_memory", + "--model", + str(model), + "--inputs", + str(inputs), + "--output", + str(output), + "--warmup", + "1", + "--repeats", + "2", + ] + result = subprocess.run(command, capture_output=True, text=True, check=False) + report = json.loads((output / "report.json").read_text()) + if invalid: + assert result.returncode != 0 and report["status"] == "failed" + assert "Input names" in report["error"] + assert "median_seconds" not in report and not (output / "outputs.npz").exists() + return + assert result.returncode == 0, result.stderr + assert report["status"] == "measured" + assert report["session_providers"] == ["CPUExecutionProvider"] + assert report["session_load_seconds"] > 0 and report["first_run_seconds"] > 0 + assert len(report["seconds"]) == 2 and report["median_seconds"] > 0 + assert len(report["model_files"]) == 2 + snapshots = list(report["memory"].values()) + assert len(snapshots) == 7 + assert all(s["resident_bytes"] > 0 and s["peak_resident_bytes"] > 0 and s["swap_bytes"] >= 0 for s in snapshots) + peaks = [s["peak_resident_bytes"] for s in snapshots] + assert peaks == sorted(peaks) + assert report["environment"]["cpu_affinity"] + with np.load(output / "outputs.npz") as actual, np.load(inputs) as expected: + np.testing.assert_array_equal(actual["scores"], expected["images"]) + assert not list(output.glob("*profile*")) + original_report = (output / "report.json").read_bytes() + assert subprocess.run(command, capture_output=True, check=False).returncode != 0 + assert (output / "report.json").read_bytes() == original_report + + +@pytest.mark.skipif(sys.platform != "linux", reason="Linux resident-memory probe") +def test_memory_peak_excludes_parent_pre_exec_allocations(): + child = ( + "import json, resource; from dev.benchmarks.onnx_cpu_memory import resident_memory; " + "print(json.dumps({'memory': resident_memory(), 'rusage': resource.getrusage(resource.RUSAGE_SELF).ru_maxrss * 1024}))" + ) + parent = f"import subprocess, sys; retained = bytearray(128 * 1024**2); subprocess.run([sys.executable, '-c', {child!r}], check=True)" + result = subprocess.run([sys.executable, "-c", parent], capture_output=True, text=True, check=True) + report = json.loads(result.stdout) + assert report["memory"]["peak_resident_bytes"] < report["rusage"] - 64 * 1024**2 From 1a26a5110a8b5e2a2df4e7b75a57e69c8fbf27ed Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 16:31:19 +0200 Subject: [PATCH 072/155] docs: validate faster CPU quantization with unsigned activations --- dev/benchmarks/README.md | 36 +++++++++++++++++ docs/benchmarks.md | 79 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 12 +++--- 3 files changed, 122 insertions(+), 5 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 2d3dca3..d94e11e 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1236,3 +1236,39 @@ reports. `measured` is not deployment acceptance. Pair results with the maintain quality evaluator and separate operator-placement probe. Timing excludes image IO and preprocessing; CPU/x86 measurements do not establish ARM kernels, memory, quality or speed, and this probe does not measure GPU memory. + +### CPU-specific QDQ candidate for EfficientNetV2 + +The native-checkpoint CPU follow-up found that the TensorRT-oriented signed, +floating-bias recipe left 107 convolutions floating under the tested x86 ONNX +Runtime. Quantizing biases alone did not change that coverage. The maintained +calibrator's unsigned-activation, INT32-bias defaults instead produced 170 +`QLinearConv` and two `QGemm` executions for both normalized symmetric flat and +hierarchical EfficientNetV2-S heads in that environment. + +```bash +.venv/bin/python -m dev.benchmarks.onnx_calibration \ + --model materialized/model.onnx --manifest prepared-train/manifest.json \ + --output cpu-qdq --activation-type uint8 \ + --method percentile --percentile 99.9 --threads 1 + +.venv/bin/python -m dev.benchmarks.onnx_inference \ + --model cpu-qdq/model.onnx --inputs batch-one.npz --output cpu-placement \ + --provider CPUExecutionProvider --threads 1 \ + --require-provider-op QLinearConv --require-provider-op QGemm +``` + +This recipe uses asymmetric activation ranges, symmetric per-channel INT8 weights +and quantized INT32 biases; omit `--symmetric-activations` and `--float-bias`. +It is a separate CPU deployment candidate, not a replacement for the signed +TensorRT recipe. Required operator names in the example reflect these tested +models; inspect the actual optimized execution for other architectures. The +presence of required integer types does not by itself prove that all weighted +operations use them: also inspect the counts and remaining floating operators. + +Use the paired quality pipeline for the complete held-out split and the isolated +CPU memory probe in separate fresh processes for resource measurements. Keep the +same input files, explicit threads and floating baseline. This x86 result does +not establish ARM fusion, kernel behavior or performance; run the full procedure +on the intended edge device before accepting that profile. See the +[measured comparison](../../docs/benchmarks.md#cpu-specific-activation-and-bias-calibration). diff --git a/docs/benchmarks.md b/docs/benchmarks.md index a0ec994..f9161b3 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2035,6 +2035,85 @@ passed; `bash dev/check.sh all` passed static checks and 476 tests, with 152 ski and one known EMA expected failure. Hardware-specific validation, sustained-load thermal behavior and end-to-end preprocessing costs remain unverified. +### CPU-specific activation and bias calibration + +This follow-up isolates the previous CPU recipe's incomplete integer coverage. +It uses the same native-trained checkpoints, materialized floating exports, +128 training calibration images, percentile 99.9 histograms and 912 validation +images. Both new candidates reproduce the previous **parsed calibration ranges +exactly**; JSON file hashes differ because of key ordering. No validation data +or metric thresholds were used to select ranges. + +1. Signed symmetric INT8 activations with INT32 biases changed only the bias + option from the previous TensorRT-oriented recipe. It still executed 63 + QLinearConv, 107 floating Conv and two QGemm operations on CPU. Bias quantization + alone did not resolve the coverage gap. Its full-data quality results are + retained rather than discarded. +2. Unsigned asymmetric UINT8 activations with symmetric per-channel INT8 weights + and INT32 biases used the maintained calibrator's CPU defaults. Both models + executed **170 QLinearConv and two QGemm operations, with no floating Conv**, + under the local CPU provider. This is a separate recipe from the TensorRT + candidate, not a change to the package's training/export defaults. + +The paired quality runner evaluated both candidates independently against the +same materialized float baseline. Candidate metric values are below; the float +values are in the preceding CPU study. + +| Candidate / output | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Signed, INT32 bias / flat | 0.764407 | 0.756174 | 0.786411 | 1.000000 | 0.802939 | +| Signed, INT32 bias / hierarchical leaf | 0.735748 | 0.730662 | 0.772329 | 1.000000 | 0.792383 | +| Signed, INT32 bias / hierarchical parent | 0.867726 | 0.845839 | 0.912249 | 1.000000 | 0.856400 | +| Unsigned, INT32 bias / flat | 0.764523 | 0.761055 | 0.783247 | 1.000000 | 0.805098 | +| Unsigned, INT32 bias / hierarchical leaf | 0.733602 | 0.727803 | 0.771896 | 1.000000 | 0.793045 | +| Unsigned, INT32 bias / hierarchical parent | 0.868766 | 0.848444 | 0.913857 | 1.000000 | 0.857750 | + +For the unsigned candidate, Macro-F1 changes against float are −0.626, −0.186 and +−0.149 percentage points for flat, leaf and parent outputs. Hierarchical leaf +precision decreases 2.956 points, despite a small recall increase. Coverage remains +one. The unsigned candidate changes 66, 60 and 18 predictions respectively; +the signed/INT32-bias candidate changes 67, 59 and 17. These remain single-trained- +checkpoint comparisons, not a general quality-improvement claim. + +After calibration, profiling and quality processes finished, three fresh-process +trials per model measured the unsigned candidate against newly run float +baselines. Settings match the preceding CPU study: batch one, 128px, one intra-op +and inter-op thread, three warmups and 31 timings, reversing recipe order in the +middle trial. Input hashes, settings and reported environments match within +every pair. The i7-12800H WSL x86 CPU affinity remained 0–19; no claim of controlled +thermals, disk-cold startup or ARM emulation is made. + +| Head / model | Warm median ms, trials 1/2/3 | Final RSS MiB, trials 1/2/3 | Approximate peak RSS MiB, trials 1/2/3 | +| --- | --- | --- | --- | +| Flat / float | 23.484 / 25.838 / 24.330 | 196.812 / 191.426 / 191.430 | 198.230 / 192.855 / 192.848 | +| Flat / unsigned INT8 | 11.022 / 12.236 / 11.983 | 103.586 / 103.852 / 103.137 | 102.738 / 103.059 / 102.273 | +| Hierarchical / float | 22.611 / 27.651 / 22.634 | 198.129 / 191.621 / 191.625 | 199.609 / 193.039 / 194.797 | +| Hierarchical / unsigned INT8 | 12.703 / 12.605 / 12.049 | 103.414 / 103.242 / 104.441 | 102.551 / 102.453 / 103.645 | + +Candidate/baseline median latency ratios are 0.469/0.474/0.493 for flat and +0.562/0.456/0.532 for hierarchical: **44–54% lower warm latency**, with **45–48% +lower final resident memory** across the six pairs. These are separate-process +median ratios, not adjacent timing pairs. Session load was usually slower for +INT8 (139–181 ms versus 109–150 ms for float), so this supports repeated inference +after startup rather than a universal end-to-end speed claim. Memory sources and +their different accuracy are as documented in the preceding study. + +This establishes a useful local x86 quality/resource trade-off and a concrete +candidate for target-device validation. It does not establish ARM execution, +sustained thermal behavior, image/preprocessing throughput, large-class heads, +GPU inference speed or HPC training benefits. Do not extrapolate these percentages +to Raspberry Pi, Spark/RTX or A40/A100/B300 systems. + +Reproduction uses the maintained [CPU recipe](../dev/benchmarks/README.md#cpu-specific-qdq-candidate-for-efficientnetv2), +paired quality runner and isolated memory probe. Retained ignored artifacts are +`tmp-cpu-bias-{flat,hierarchical}/`, `tmp-cpu-u8-{flat,hierarchical}/` and +`tmp-cpu-u8-trials/`, including calibration reports/ranges, placement profiles, +all prediction CSVs and five metrics, raw timing/memory reports and the trial +driver. Batch-one inputs and source provenance are reused from +`tmp-edge-process-probe/`. All four calibration and quality cases, four placement +checks, and twelve timing processes completed successfully. Static checks passed; +this increment changes documentation only, so the full runtime suite was not rerun. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index bfd2ced..cf5d992 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the checkpoint conversion study](benchmarks.md#native-int8-checkpoint-to-calibrated-tensorrt-deployment). +recorded in [the CPU recipe study](benchmarks.md#cpu-specific-activation-and-bias-calibration). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -18,7 +18,7 @@ or integer operator count sufficient evidence of production readiness. | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | | QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | | Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | -| CPU/edge inference | Calibrated ONNX CPU execution and full Blair metrics on x86; isolated process-memory/timing probe; about 45% lower resident memory but slower batch-one inference for the tested candidate | CPU-specific recipe/fusion tuning; Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging | +| CPU/edge inference | Full Blair metrics and isolated process-memory/timing on x86; unsigned CPU recipe executes 170 integer convolutions and two head GEMMs, with 44–54% lower warm batch-one latency and 45–48% lower resident memory in three trials per head | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging; larger-class qualification | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | Native training currently leaves convolutions, gradients and optimizer states @@ -120,9 +120,11 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur dependency set; neither x86 CPU results nor CUDA results establish ARM support. The [isolated Linux CPU study](benchmarks.md#isolated-linux-cpu-memory-and-inference-study) now supplies a maintained RSS/PSS/peak and load/first/warm inference probe. - Its local candidate saves resident memory but retains 107 floating convolutions - and is slower; investigate CPU-specific calibration/fusion before selecting the - edge recipe. Repeat on real ARM hardware with preprocessing and sustained-load + The initial signed candidate saved memory but retained 107 floating convolutions + and was slower. A [CPU-specific unsigned recipe](benchmarks.md#cpu-specific-activation-and-bias-calibration) + now executes all 170 convolutions as integer operations locally, with substantial + latency/memory savings and a largest observed quality loss of 2.956 percentage + points in hierarchical leaf precision. Repeat on real ARM hardware with preprocessing and sustained-load conditions, rather than extrapolating the local memory percentage. 6. **Turn the accepted trade-offs into continuous release evidence.** Select From 4606650992379511d48e3991e8031bbf8a3632de Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 16:43:28 +0200 Subject: [PATCH 073/155] feat: compose CPU deployment quality and resource validation --- dev/benchmarks/README.md | 50 +++++++ dev/benchmarks/cpu_deployment.py | 175 ++++++++++++++++++++++ docs/benchmarks.md | 40 +++++ docs/quantization-status.md | 7 +- tests/test_benchmark_dataset_inference.py | 59 ++++++++ 5 files changed, 329 insertions(+), 2 deletions(-) create mode 100644 dev/benchmarks/cpu_deployment.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index d94e11e..621fe5b 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1272,3 +1272,53 @@ same input files, explicit threads and floating baseline. This x86 result does not establish ARM fusion, kernel behavior or performance; run the full procedure on the intended edge device before accepting that profile. See the [measured comparison](../../docs/benchmarks.md#cpu-specific-activation-and-bias-calibration). + +### Composed CPU deployment comparison + +`dev.benchmarks.cpu_deployment` connects full held-out quality evaluation, +operator/provider inspection and repeated isolated memory/latency trials for +already exported baseline and candidate ONNX bundles. Run from a checkout in a +prepared Linux CPU environment; dependencies are never installed implicitly. + +```bash +.venv/bin/python -m dev.benchmarks.cpu_deployment \ + --baseline materialized/model.onnx --candidate cpu-qdq/model.onnx \ + --manifest prepared-val/manifest.json --inputs batch-one.npz \ + --output cpu-deployment-run-1 --threads 1 --trials 3 --warmup 3 --repeats 31 \ + --require-provider-op QLinearConv --require-provider-op QGemm \ + --metrics-python /path/to/metrics-env/bin/python +``` + +The manifest supplies held-out identities, class ordering and score semantics. +The separate NPZ selects the fixed batch shape for resource measurement; carry +its preprocessing and sample-selection provenance alongside it. Baseline and +candidate use that same NPZ. Preparation and calibration remain separate steps. +The main interpreter supplies ONNX CPU dependencies, while `--metrics-python` +can select an independently prepared mini_metrics environment. + +Execution is sequential: paired quality, baseline/candidate placement, then +resource trials. Every profiling and resource command gets its own process; +profiling allocations cannot inflate the memory trial's high-water mark. The +first trial runs baseline then candidate, the next reverses the order, and so +on. Each resource pair must have matching settings and reported environments. +The graph/external-weight hashes must match those observed during quality +collection, and the resource input hash must remain unchanged through placement +and timing. Drift or a failed child stops the run and retains the failed phase, +logs and completed evidence. An existing output directory is rejected. + +The output contains the child reports and logs, a top-level `report.json`, and +`summary.md` combining all five quality deltas with per-trial latency and memory +ratios. These ratios compare separate-process warm medians; they are not adjacent +per-inference pairs. All raw timings, startup observations and memory accounting +remain in the child reports. Requested candidate operation types must occur on +CPU, but this does not assert that every weighted operation is integer: inspect +`execution` counts and remaining floating operators. The operation requirements +are explicit and optional; no model-specific operator list is hardcoded. + +Run the command without competing tests or benchmarks when using timing results. +It does not evict disk caches or control temperature, affinity or power states. +`evaluated` means the procedure completed, not that quality/resource trade-offs +passed production acceptance. For visible CI reporting, retain the complete +output directory even on failure and append its `summary.md` to the job summary +when present. Durable cross-run hosting, recipe acceptance gates, preprocessing +costs and actual target-device verification remain separate requirements. diff --git a/dev/benchmarks/cpu_deployment.py b/dev/benchmarks/cpu_deployment.py new file mode 100644 index 0000000..3eeab97 --- /dev/null +++ b/dev/benchmarks/cpu_deployment.py @@ -0,0 +1,175 @@ +"""Compose CPU deployment quality, placement and isolated resource measurements.""" + +import json +import os +import subprocess +import sys +from argparse import ArgumentParser +from pathlib import Path + +from .inference_pair import run_pair +from .onnx_inference import file_hash + + +def evaluate( + baseline, candidate, manifest, inputs, output, threads=1, trials=3, warmup=3, repeats=31, required_ops=(), metrics_python=sys.executable +): + if min(threads, trials, warmup, repeats) < 1: + raise ValueError("Threads, trials, warmup and repeats must be positive") + paths = {name: Path(path).resolve(strict=True) for name, path in (("baseline", baseline), ("candidate", candidate))} + manifest, inputs, output = Path(manifest).resolve(strict=True), Path(inputs).resolve(strict=True), Path(output).resolve() + inputs_hash = file_hash(inputs) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "required_candidate_ops": list(required_ops), + "stages": [], + "pairs": [], + "scope": ( + "CPU quality, observed operation placement and separate-process resource trials; no automatic acceptance gate. " + "Latency excludes image decoding/preprocessing. Peak RSS is approximate and includes process overhead. " + "Target claims require execution on that target; no thermal control or disk-cache eviction." + ), + } + + def save(): + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + + def child(name, module, arguments, expected): + command = [sys.executable, "-m", f"dev.benchmarks.{module}", *map(str, arguments), "--output", str(output / name)] + stage = {"name": name, "command": command, "status": "running", "log": f"{name}.log"} + report["stages"].append(stage) + save() + with (output / stage["log"]).open("w") as log: + process = subprocess.run( + command, + cwd=Path(__file__).resolve().parents[2], + env={**os.environ, "PYTHONHASHSEED": "0"}, + stdout=log, + stderr=subprocess.STDOUT, + check=False, + ) + stage["returncode"] = process.returncode + if process.returncode: + raise RuntimeError(f"{name} failed; see {output / stage['log']}") + result_path = output / name / "report.json" + result = json.loads(result_path.read_text()) + if result["status"] != expected: + raise RuntimeError(f"Unexpected {name} status: {result['status']}") + stage.update(status=expected, report_sha256=file_hash(result_path)) + return result + + try: + report["phase"] = "quality" + save() + runtime = {"backend": "onnx", "threads": threads} + quality = run_pair(manifest, paths["baseline"], paths["candidate"], output / "quality", runtime, runtime, metrics_python) + report["quality"] = { + "report_sha256": file_hash(output / "quality/report.json"), + "levels": quality["levels"], + "models": quality["models"], + "undefined_metrics": quality["undefined_metrics"], + } + expected_files = {role: json.loads((output / f"quality/{role}/report.json").read_text())["model_files"] for role in paths} + report["phase"] = "placement" + report["execution"] = {} + for role, model in paths.items(): + args = [ + "--model", + model, + "--inputs", + inputs, + "--provider", + "CPUExecutionProvider", + "--threads", + threads, + "--warmup", + 1, + "--repeats", + 1, + ] + if role == "candidate": + for operation in required_ops: + args.extend(["--require-provider-op", operation]) + placement = child(f"placement-{role}", "onnx_inference", args, "passed") + if placement["models"][0]["files"] != expected_files[role] or placement["inputs"]["sha256"] != inputs_hash: + raise ValueError("Model or timing inputs changed between quality and placement") + report["execution"][role] = placement["models"][0]["execution"] + report["phase"] = "resources" + for trial in range(trials): + pair = {} + order = ("baseline", "candidate") if trial % 2 == 0 else ("candidate", "baseline") + for role in order: + result = child( + f"trial-{trial}-{role}", + "onnx_cpu_memory", + ["--model", paths[role], "--inputs", inputs, "--threads", threads, "--warmup", warmup, "--repeats", repeats], + "measured", + ) + if result["model_files"] != expected_files[role] or result["inputs"]["sha256"] != inputs_hash: + raise ValueError("Model or timing inputs changed between quality and resource measurement") + pair[role] = result + a, b = pair["baseline"], pair["candidate"] + if any(a[key] != b[key] for key in ("settings", "versions", "environment", "runtime_build", "memory_sources")): + raise ValueError("Paired resource environments or settings differ") + ratios = {"warm_latency": b["median_seconds"] / a["median_seconds"]} + for key in ("resident_bytes", "peak_resident_bytes"): + ratios[key] = b["memory"]["after_measurement"][key] / a["memory"]["after_measurement"][key] + report["pairs"].append({"trial": trial, "order": list(order), "candidate_over_baseline": ratios}) + save() + lines = [ + "# CPU deployment comparison", + "", + report["scope"], + "", + "| Trial | Warm latency ratio | Resident memory ratio | Approximate peak ratio |", + "| --- | ---: | ---: | ---: |", + ] + for pair in report["pairs"]: + ratios = pair["candidate_over_baseline"] + lines.append( + f"| {pair['trial']} | {ratios['warm_latency']:.4f} | {ratios['resident_bytes']:.4f} | {ratios['peak_resident_bytes']:.4f} |" + ) + lines.extend( + [ + "", + "Ratios are candidate / baseline; below one is lower. " + "These compare separate-process medians, not adjacent inference pairs.", + "", + (output / "quality/summary.md").read_text(), + ] + ) + (output / "summary.md").write_text("\n".join(lines)) + report.update(status="evaluated", phase="complete") + except BaseException as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + for stage in report["stages"]: + if stage["status"] == "running": + stage["status"] = "failed" + raise + finally: + save() + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + for name in ("baseline", "candidate", "manifest", "inputs", "output"): + parser.add_argument("--" + name, type=Path, required=True) + for name, default in (("threads", 1), ("trials", 3), ("warmup", 3), ("repeats", 31)): + parser.add_argument("--" + name, type=int, default=default) + parser.add_argument( + "--require-provider-op", + dest="required_ops", + action="append", + default=[], + help="Required candidate CPU operation type; inspect counts for coverage", + ) + parser.add_argument("--metrics-python", default=sys.executable) + evaluate(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index f9161b3..f6b6305 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2114,6 +2114,46 @@ driver. Batch-one inputs and source provenance are reused from checks, and twelve timing processes completed successfully. Static checks passed; this increment changes documentation only, so the full runtime suite was not rerun. +### Composed CPU deployment validation + +`dev.benchmarks.cpu_deployment` now runs the full held-out quality comparison, +baseline/candidate operation inspection and repeated fresh-process resource +trials in one command. It links graph/external-weight hashes across phases, +rejects changed timing inputs, checks paired environments/settings and retains +logs, child reports and a combined Markdown summary. Candidate operator +requirements are explicit, not hardcoded by architecture. See the +[command guide](../dev/benchmarks/README.md#composed-cpu-deployment-comparison). + +Both real heads completed the command with the unsigned CPU candidates and the +same materialized float baselines, all 912 held-out Blair images, batch-one +resource inputs, one thread, three trial pairs, three warmups and 31 repetitions. +The full regression suite finished before these runs; the two complete model +comparisons ran sequentially. Predictions and all five metric values reproduced +the preceding CPU recipe study exactly for both baseline and candidate CSVs. +Each candidate again executed 170 QLinearConv and two QGemm operations with no +floating Conv. Every phase's artifact identity checks passed. + +| Head | Candidate/baseline warm latency ratios, trials 1/2/3 | Final RSS ratios, trials 1/2/3 | +| --- | --- | --- | +| Flat | 0.489 / 0.474 / 0.588 | 0.540 / 0.518 / 0.536 | +| Hierarchical | 0.454 / 0.453 / 0.491 | 0.520 / 0.534 / 0.545 | + +These fresh x86 runs replicate the local latency/memory benefit. They are +separate-process median comparisons, not adjacent inference-pair measurements, +and do not establish sustained thermal behavior or target ARM performance. +Raw warm timings, startup observations, approximate memory peaks, output arrays, +placement profiles and quality reports are retained in +`tmp-cpu-deployment-{flat,hierarchical}/`. Existing model and input bundles are +reused; they are not copied into these report directories. + +Regression coverage exercises the real subprocess pipeline with a small +two-level oracle, alternating execution order, rejection of existing outputs, +missing required operations and changed timing inputs. `bash dev/check.sh all` +passed static checks and 479 tests, with 152 skips and one known EMA expected +failure. This composes evaluation of existing CPU deployment artifacts; +preparation/calibration, GPU orchestration, durable CI hosting and profile-specific +acceptance gates remain unfinished. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index cf5d992..d43765d 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -74,8 +74,11 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur [paired quality pipeline](../dev/benchmarks/README.md#paired-inference-quality-pipeline) now runs both collectors and the evaluator in fresh processes, retaining logs, failure reports and a Markdown summary, with separately selectable runtime and - metric interpreters. Compose preparation, calibration, builds, placement and - performance checks with it, and package the explicit optional environments. + metric interpreters. A [composed CPU deployment check](../dev/benchmarks/README.md#composed-cpu-deployment-comparison) + now connects full quality, candidate operation requirements and repeated + fresh-process memory/latency trials, checks artifact hashes across phases and + produces a combined Markdown summary. Preparation, calibration, GPU builds and + GPU resource checks still need composition; package the explicit optional environments. The default preparation factory uses current architecture-loader transforms; custom preprocessing still needs an explicit reviewed factory and input verification. diff --git a/tests/test_benchmark_dataset_inference.py b/tests/test_benchmark_dataset_inference.py index 90b7ef4..3a7f079 100644 --- a/tests/test_benchmark_dataset_inference.py +++ b/tests/test_benchmark_dataset_inference.py @@ -6,6 +6,7 @@ import numpy as np import pytest +from dev.benchmarks.cpu_deployment import evaluate as evaluate_cpu from dev.benchmarks.dataset_inference import collect, inference_manifest, pair_bundle, predictions from dev.benchmarks.inference_pair import run_pair from dev.benchmarks.quality_compare import read_manifest, read_predictions @@ -113,6 +114,64 @@ def test_pair_pipeline_rejects_unknown_options_before_creating_output(example, t assert not output.exists() +@pytest.mark.skipif(sys.platform != "linux", reason="Linux process memory measurements") +def test_cpu_deployment_composes_quality_placement_and_alternating_trials(example, tmp_path): + pytest.importorskip("onnxruntime") + pytest.importorskip("mini_metrics") + model, manifest, _ = example + output = tmp_path / "deployment" + report = evaluate_cpu(model, model, manifest, tmp_path / "batch-0.npz", output, trials=2, warmup=1, repeats=2, required_ops=["Add"]) + assert report["status"] == "evaluated" + assert [p["order"] for p in report["pairs"]] == [["baseline", "candidate"], ["candidate", "baseline"]] + assert len(report["stages"]) == 6 + assert all(value > 0 for p in report["pairs"] for value in p["candidate_over_baseline"].values()) + assert all(level["prediction_changes"] == 0 for level in report["quality"]["levels"]) + assert any(op["op"] == "Add" for op in report["execution"]["candidate"]) + summary = (output / "summary.md").read_text() + assert "Warm latency ratio" in summary and "Theil U delta" in summary + with pytest.raises(FileExistsError): + evaluate_cpu(model, model, manifest, tmp_path / "batch-0.npz", output) + + +def test_cpu_deployment_missing_required_operation_stops_before_resource_trials(example, tmp_path): + pytest.importorskip("onnxruntime") + pytest.importorskip("mini_metrics") + model, manifest, _ = example + output = tmp_path / "failed-deployment" + with pytest.raises(RuntimeError, match="placement-candidate failed"): + evaluate_cpu(model, model, manifest, tmp_path / "batch-0.npz", output, required_ops=["QLinearConv"]) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" and report["phase"] == "placement" + assert report["pairs"] == [] and not list(output.glob("trial-*")) + assert report["stages"][-1]["status"] == "failed" + assert (output / "quality/summary.md").exists() + + +@pytest.mark.skipif(sys.platform != "linux", reason="Linux process memory measurements") +def test_cpu_deployment_rejects_changed_timing_input_after_placement(example, tmp_path, monkeypatch): + pytest.importorskip("onnxruntime") + pytest.importorskip("mini_metrics") + model, manifest, _ = example + inputs, output = tmp_path / "batch-0.npz", tmp_path / "changed-input" + original = subprocess.run + + def change_before_measurement(command, **kwargs): + if "dev.benchmarks.onnx_cpu_memory" in command: + with np.load(inputs) as archive: + arrays = {k: v.copy() for k, v in archive.items()} + arrays["x"] += 0.5 + np.savez(inputs, **arrays) + return original(command, **kwargs) + + monkeypatch.setattr(subprocess, "run", change_before_measurement) + with pytest.raises(ValueError, match="changed between quality and resource"): + evaluate_cpu(model, model, manifest, inputs, output, trials=1, warmup=1, repeats=1) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" and report["phase"] == "resources" + assert report["pairs"] == [] + assert not (output / "summary.md").exists() + + def test_cpu_collection_handles_multiple_inputs_levels_and_partial_batch(example, tmp_path): pytest.importorskip("onnxruntime") model, manifest, metadata = example From ad11e6d9d38fd4ba0b0b251057ba247077c85bbc Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 17:13:26 +0200 Subject: [PATCH 074/155] feat: benchmark large-head full and frozen INT8 training --- dev/benchmarks/README.md | 49 +++++++ dev/benchmarks/large_head_training.py | 202 ++++++++++++++++++++++++++ docs/benchmarks.md | 67 +++++++++ docs/quantization-status.md | 6 +- tests/test_benchmark_large_head.py | 27 ++++ 5 files changed, 350 insertions(+), 1 deletion(-) create mode 100644 dev/benchmarks/large_head_training.py create mode 100644 tests/test_benchmark_large_head.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 621fe5b..c8f4f7b 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1322,3 +1322,52 @@ passed production acceptance. For visible CI reporting, retain the complete output directory even on failure and append its `summary.md` to the job summary when present. Durable cross-run hosting, recipe acceptance gates, preprocessing costs and actual target-device verification remain separate requirements. + +### Large-head training capacity command + +`dev.benchmarks.large_head_training` makes the earlier large-class probe reusable +on the intended training machines. It builds the actual backbone with a symmetric, +normalized flat or hierarchical head, then performs eager MuonAuxAdamW updates +through the existing builder, scaler and optimizer-step helper. It uses fixed +synthetic uint8 images and labels, not a convergence or dataset-quality benchmark. + +```bash +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ + .venv/bin/python -m dev.benchmarks.large_head_training \ + --classes 10000 --backbone efficientnet_v2_s --batch-size 32 --image-size 128 \ + --dtype float16 --seed 42 --warmup 3 --steps 5 --output capacity-float + +# Same settings, separate fresh process, quantized weights/saved Linear inputs: +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 TORCHINDUCTOR_COMPILE_THREADS=1 \ + .venv/bin/python -m dev.benchmarks.large_head_training \ + --classes 10000 --backbone efficientnet_v2_s --batch-size 32 --image-size 128 \ + --dtype float16 --seed 42 --warmup 3 --steps 5 --quantized --output capacity-int8 +``` + +Add `--hierarchical` to both commands for a synthetic parent taxonomy grouping +consecutive leaves in groups of 100. Add `--frozen` to both for head-only training: +backbone parameters are frozen and its modules use evaluation mode, while the head +trains. Parameters intentionally frozen by the classifier remain frozen. Backbone +forward computation still runs on every update; embeddings are not cached. +`--dtype bfloat16` selects a separate AMP profile; `float32` disables autocast. +Use `--device cpu --dtype float32` only for offline floating diagnostics. Native +INT8 training requires an accessible CUDA device. + +Use new output directories, matching settings, alternating float/INT8 order and +paired seeds. A separate CPU generator fixes inputs independently of quantization +setup RNG consumption; reports retain their hash. The report includes warmup and +measured losses, applied-update flags, trainable/gradient parameter counts, +unused trainable parameter names, quantization coverage and storage, warm median +update time and CUDA allocated/reserved peaks. Valid normalized heads can retain +unused BatchNorm parameters; these are reported rather than silently reclassified +or treated as missing gradients in an active branch. Measured skipped/nonfinite +updates or a head with no gradients fail, retaining the partial report. + +Setup/warmup allocation and measured-phase allocation are recorded separately. +Kernel-cache state is not controlled; there is no cold-compilation speed claim. +The probe does not exercise the full trainer's loading, scheduler, checkpoint, +resume, augmentation, distributed or convergence behavior. Use the integrated +real-data profiles for those comparisons. Run target GPU generations separately; +a laptop result cannot certify A40/A100/B300 or the intended desktop. Keep full +models at 100k classes or below on the laptop, and increase capacity only on a +machine provisioned for it. No weights or datasets are saved by this probe. diff --git a/dev/benchmarks/large_head_training.py b/dev/benchmarks/large_head_training.py new file mode 100644 index 0000000..fbcfbaa --- /dev/null +++ b/dev/benchmarks/large_head_training.py @@ -0,0 +1,202 @@ +"""Synthetic full-model or frozen-backbone training capacity probe; not convergence.""" + +import hashlib +import json +import platform +import statistics +import time +from argparse import ArgumentParser +from pathlib import Path + +import torch + +from mini_trainer.builders import BaseBuilder +from mini_trainer.hierarchical.model import HierarchicalClassifier +from mini_trainer.modeling import Classifier +from mini_trainer.modeling.classifier import classification_module +from mini_trainer.modeling.quantized_training import prepare_quantized_training +from mini_trainer.trainer import _optimizer_step +from mini_trainer.training import MuonAuxAdamW + + +def run( + output, + classes=10000, + hierarchical=False, + frozen=False, + quantized=False, + batch_size=32, + image_size=128, + seed=42, + warmup=3, + steps=5, + device="cuda:0", + dtype="float16", + backbone="efficientnet_v2_s", +): + device = torch.device(device) + if min(classes, batch_size, image_size, warmup, steps) < 1 or classes < 2 or batch_size < 2: + raise ValueError("Require at least two classes/samples and positive image size, warmup and steps") + if device.type not in ("cpu", "cuda") or dtype not in ("float32", "float16", "bfloat16"): + raise ValueError("Unsupported device or dtype") + if device.type == "cpu" and (quantized or dtype != "float32"): + raise ValueError("CPU is a float32 diagnostic only; native INT8 training requires CUDA") + if device.type == "cuda" and not torch.cuda.is_available(): + raise RuntimeError("An accessible CUDA GPU is required") + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + torch.set_num_threads(1) + torch.manual_seed(seed) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": hashlib.sha256(Path(__file__).read_bytes()).hexdigest(), + "settings": { + "classes": classes, + "hierarchical": hierarchical, + "frozen_backbone": frozen, + "quantized": quantized, + "batch_size": batch_size, + "image_size": image_size, + "seed": seed, + "warmup": warmup, + "steps": steps, + "device": str(device), + "dtype": dtype, + "backbone": backbone, + "hidden": "symmetric", + "normalized": True, + }, + "environment": {"platform": platform.platform(), "torch": torch.__version__}, + "warmup": [], + "steps": [], + "scope": ( + "Repeated fixed synthetic uint8 images and labels, eager MuonAuxAdamW and AMP. " + "Full-model or frozen-backbone updates; no cached embeddings. " + "No convergence, loader, checkpoint, distributed or target-hardware performance claims." + ), + } + + def synchronize(): + if device.type == "cuda": + torch.cuda.synchronize(device) + + try: + if device.type == "cuda": + torch.cuda.set_device(device) + torch.cuda.reset_peak_memory_stats(device) + report["environment"]["gpu"] = torch.cuda.get_device_name(device) + # Independent CPU generator prevents quantization/setup RNG consumption + # from changing the synthetic workload between float and INT8 runs. + generator = torch.Generator().manual_seed(seed + 1) + images = torch.randint(0, 256, (batch_size, 3, image_size, image_size), generator=generator, dtype=torch.uint8) + labels = torch.randint(classes, (batch_size,), generator=generator) + report["input_sha256"] = hashlib.sha256(images.numpy().tobytes() + labels.numpy().tobytes()).hexdigest() + images, labels = images.to(device), labels.to(device) + cls = HierarchicalClassifier if hierarchical else Classifier + kwargs = {"sparse_masks": [torch.arange(classes, device=device) // 100]} if hierarchical else {} + model, preprocess = cls.build( + model_type=backbone, + num_classes=classes, + hidden=True, + normalized=True, + model_args={"pretrained": False}, + device=device, + resize_size=image_size, + **kwargs, + ) + head = classification_module(model) + report["head_trainable_parameters"] = sum(p.numel() for p in head.parameters() if p.requires_grad) + head_ids = {id(p) for p in head.parameters()} + if frozen: + for parameter in model.parameters(): + if id(parameter) not in head_ids: + parameter.requires_grad_(False) + model.eval() + head.train() + else: + model.train() + report["recipe"] = prepare_quantized_training(model) if quantized else None + if sum(p.numel() for p in head.parameters() if p.requires_grad) != report["head_trainable_parameters"]: + raise RuntimeError("Freezing/preparation changed the head's trainable parameter contract") + optimizer = BaseBuilder.build_optimizer(model, MuonAuxAdamW, lr=0.01, weight_decay=0.0) + scaler = BaseBuilder.build_scaler(device.type, enabled=device.type == "cuda" and dtype == "float16") + report["parameters"] = { + "total": sum(p.numel() for p in model.parameters()), + "trainable": sum(p.numel() for p in model.parameters() if p.requires_grad), + } + report["parameter_bytes"] = sum( + p.int_data.numel() + p.scale.numel() * p.scale.element_size() + if getattr(p, "_is_quantized_training", False) + else p.numel() * p.element_size() + for p in model.parameters() + ) + torch.manual_seed(seed + 2) + + def step(): + with torch.autocast(device.type, dtype=getattr(torch, dtype), enabled=dtype != "float32"): + scores = model(preprocess(images)) + values = scores if isinstance(scores, (tuple, list)) else [scores] + loss = torch.nn.functional.cross_entropy(values[0], labels) + if hierarchical: + loss = loss + torch.nn.functional.cross_entropy(values[1], labels // 100) + optimizer.zero_grad(set_to_none=True) + scaler.scale(loss).backward() + scaler.unscale_(optimizer) + torch.nn.utils.clip_grad_norm_(model.parameters(), 5) + updated = _optimizer_step(optimizer, scaler) + return loss.detach(), updated + + for _ in range(warmup): + loss, updated = step() + report["warmup"].append({"loss": float(loss), "updated": updated}) + synchronize() + if device.type == "cuda": + report["setup_peak_allocated_bytes"] = torch.cuda.max_memory_allocated(device) + torch.cuda.reset_peak_memory_stats(device) + for _ in range(steps): + synchronize() + started = time.perf_counter() + loss, updated = step() + synchronize() + elapsed = time.perf_counter() - started + report["steps"].append({"seconds": elapsed, "loss": float(loss), "updated": updated}) + if device.type == "cuda": + report["measured_peak_allocated_bytes"] = torch.cuda.max_memory_allocated(device) + report["measured_peak_reserved_bytes"] = torch.cuda.max_memory_reserved(device) + if any(not s["updated"] or not torch.isfinite(torch.tensor(s["loss"])) for s in report["steps"]): + raise RuntimeError("Measured updates were skipped or nonfinite; timings are not a successful training comparison") + # Some valid heads retain trainable parameters for inactive branches + # (for example BatchNorm when normalization selects unit embeddings). + # Record those explicitly, while requiring an active head gradient. + report["unused_trainable_parameters"] = [name for name, p in model.named_parameters() if p.requires_grad and p.grad is None] + report["parameters"]["with_gradient"] = sum(p.numel() for p in model.parameters() if p.grad is not None) + if not any(p.requires_grad and p.grad is not None for p in head.parameters()): + raise RuntimeError("The classification head did not receive gradients") + if frozen and any(p.grad is not None for p in model.parameters() if not p.requires_grad): + raise RuntimeError("A frozen parameter received a gradient") + report["median_seconds_per_update"] = statistics.median(s["seconds"] for s in report["steps"]) + report["status"] = "measured" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, required=True) + for name, default in (("classes", 10000), ("batch-size", 32), ("image-size", 128), ("seed", 42), ("warmup", 3), ("steps", 5)): + parser.add_argument("--" + name, type=int, default=default) + for flag in ("hierarchical", "frozen", "quantized"): + parser.add_argument("--" + flag, action="store_true") + parser.add_argument("--device", default="cuda:0") + parser.add_argument("--dtype", choices=["float32", "float16", "bfloat16"], default="float16") + parser.add_argument("--backbone", default="efficientnet_v2_s") + run(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index f6b6305..3548056 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2154,6 +2154,73 @@ failure. This composes evaluation of existing CPU deployment artifacts; preparation/calibration, GPU orchestration, durable CI hosting and profile-specific acceptance gates remain unfinished. +### Maintained 100k-class BF16 training comparison + +`dev.benchmarks.large_head_training` turns the earlier capacity probe into a +maintained command with independent synthetic-input RNG, applied-update checks, +failure reports and separate full/frozen-backbone modes. It preserves the head's +intentionally frozen parameters. Valid unused normalized-head BatchNorm parameters +are reported explicitly; they do not imply a missing gradient in an active branch. +See the [command guide](../dev/benchmarks/README.md#large-head-training-capacity-command). + +The local study ran 24 fresh GPU processes: normalized symmetric EfficientNetV2-S +flat/hierarchical heads, 100,000 leaf classes, full/frozen backbone, float/native +INT8 storage, and seeds 42/43/44. Both paths used BF16 autocast, FP32 gradients and +eager MuonAuxAdamW (learning rate 0.01, zero weight decay, norm clipping at 5). +The hierarchical taxonomy groups leaves in consecutive groups of 100. Batch size +was 32 at 128px, with three warmup and five measured updates on each fixed batch. +Floating parameters remain FP32; native INT8 quantizes the two head Linear weights +and saved Linear inputs. Convolutions, gradients and optimizer states remain floating. + +The backbone was initialized without pretrained weights in every case. Frozen +mode is therefore a synthetic capacity diagnostic, not realistic pretrained +fine-tuning: it evaluates the backbone on every batch and trains the head, without +caching embeddings. Loss is leaf cross-entropy plus parent cross-entropy for the +hierarchical case. These are new inputs and BF16 profiles; earlier FP16 results +are not reused as baselines. Float/INT8 order reverses for seed 43. + +All input hashes, settings (apart from quantization), parameter counts and reported +environments matched within each pair. All 72 warmup and 120 measured optimizer +updates were applied. The only unused trainable parameters in every run were the +normalized head's inactive BatchNorm weight and bias. No frozen parameters received +gradients. Measurements began after the full regression suite and GPU smoke checks +had exited, and all GPU cases ran sequentially on the RTX 3080 Ti Laptop with +PyTorch 2.12.0+cu130 and TorchAO 0.17.0. + +| Head / mode | Float median update ms, seeds 42/43/44 | INT8 median update ms, seeds 42/43/44 | Float measured peak GiB | INT8 measured peak GiB | +| --- | --- | --- | ---: | ---: | +| Flat / full | 137.349 / 110.825 / 148.278 | 167.257 / 143.157 / 151.248 | 3.819 | 3.154 | +| Hierarchical / full | 132.841 / 145.857 / 152.000 | 132.494 / 120.835 / 134.396 | 3.820 | 3.154 | +| Flat / frozen | 68.440 / 78.138 / 103.929 | 88.448 / 63.521 / 80.553 | 2.800 | 3.093 | +| Hierarchical / frozen | 90.369 / 73.076 / 91.959 | 72.296 / 92.531 / 92.770 | 2.801 | 3.094 | + +Measured-phase CUDA allocated peaks were identical across the three seeds for +each configuration. Native INT8 reduced the full-model peak by **17.4%**, but +**increased the frozen-backbone peak by 10.5%**. Parameter storage decreased from +about 0.559 to 0.197 GiB in both modes; that storage reduction is not proof of a +runtime-memory reduction. The frozen-mode peak needs allocation profiling before +choosing an optimization; this comparison alone does not identify its cause. + +INT8/float update-time ratios were 1.218/1.292/1.020 for flat/full, +0.997/0.828/0.884 for hierarchical/full, 1.292/0.813/0.775 for flat/frozen and +0.800/1.266/1.009 for hierarchical/frozen. These short local trials do not establish +a general speedup or stable target throughput. No device clock/thermal controls +or cold compiler-cache controls were imposed. Timings use synchronized host +boundaries around preprocessing, forward, backward, clipping and optimizer work; +scalar loss reporting follows the timed interval. Setup/warmup allocated peaks +and measured allocated/reserved peaks are retained separately. These are PyTorch +allocator measurements, not total device/process memory. + +`tmp-large-head-bf16/` retains the paired driver, logs, reports and comparison JSON. +The probe saves no checkpoints or datasets. CPU float32 full/frozen diagnostics +cover both head types, while separate FP16 and 100k-class hierarchical BF16 native +INT8 CUDA smoke cases checked execution. `bash dev/check.sh all` passed static +checks and 482 tests, with 152 skips and one known EMA expected failure. +Actual HPC/desktop hardware, realistic pretrained fine-tuning, longer training, +convergence, loading, checkpoint/resume, scheduler and distributed behavior still +require the corresponding integrated profiles; this capacity probe does not +certify those parts of the goal. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index d43765d..66bd2e8 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -14,7 +14,7 @@ or integer operator count sufficient evidence of production readiness. | Workstream | Verified locally | What remains unproven | | --- | --- | --- | -| Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions, full EfficientNetV2-S updates at 100k classes; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | +| Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions; maintained 100k-class full/frozen BF16 capacity comparison: full-model peak 17.4% lower, frozen peak 10.5% higher, mixed timing; bounded preparation and initialization | Frozen-mode peak allocation profiling; benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | | QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | | Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | @@ -94,6 +94,10 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur setup, steady-state throughput/latency, transfers, loading and runtime memory separately. Optimize the dominant measured costs rather than assuming integer arithmetic is faster. Include full and frozen-backbone training/fine-tuning. + The [100k-class BF16 capacity study](benchmarks.md#maintained-100k-class-bf16-training-comparison) + now covers both modes and heads across three seeds. Profile the reproducible + frozen-mode INT8 peak regression before treating that mode as a memory benefit; + repeat with realistic pretrained features and longer integrated training. 3. **Qualify the trained-checkpoint-to-deployment contract.** Explicit conversion and calibration now connect the matched native QT checkpoints to distinct diff --git a/tests/test_benchmark_large_head.py b/tests/test_benchmark_large_head.py new file mode 100644 index 0000000..bb96187 --- /dev/null +++ b/tests/test_benchmark_large_head.py @@ -0,0 +1,27 @@ +import pytest + +from dev.benchmarks.large_head_training import run + + +@pytest.mark.parametrize("hierarchical", [False, True]) +def test_large_head_cpu_diagnostic_checks_full_and_frozen_updates(tmp_path, hierarchical): + common = dict(classes=101, hierarchical=hierarchical, batch_size=2, image_size=32, warmup=1, steps=1, device="cpu", dtype="float32") + full = run(tmp_path / "full", **common) + frozen = run(tmp_path / "frozen", frozen=True, **common) + assert full["status"] == frozen["status"] == "measured" + assert full["input_sha256"] == frozen["input_sha256"] + assert full["parameters"]["total"] == frozen["parameters"]["total"] + assert frozen["parameters"]["trainable"] < full["parameters"]["trainable"] + assert frozen["parameters"]["trainable"] == frozen["head_trainable_parameters"] == full["head_trainable_parameters"] + for report in (full, frozen): + assert report["steps"][0]["updated"] + assert report["median_seconds_per_update"] > 0 + assert "measured_peak_allocated_bytes" not in report + with pytest.raises(FileExistsError): + run(tmp_path / "full", **common) + + +def test_native_quantized_training_is_not_reported_as_cpu_training(tmp_path): + with pytest.raises(ValueError, match="requires CUDA"): + run(tmp_path / "invalid", device="cpu", dtype="float32", quantized=True) + assert not (tmp_path / "invalid").exists() From 7ecff2f3ecb499f1a0abf494df9b207a513eba63 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 17:34:08 +0200 Subject: [PATCH 075/155] perf: fuse mixed-dtype INT8 weight updates with matching rounding --- mini_trainer/modeling/_quantized_training.py | 7 +++- mini_trainer/modeling/_quantized_update.py | 4 ++- tests/test_quantized_training_model.py | 38 ++++++++++++++++++++ 3 files changed, 47 insertions(+), 2 deletions(-) diff --git a/mini_trainer/modeling/_quantized_training.py b/mini_trainer/modeling/_quantized_training.py index 12cbaa2..8e5c132 100644 --- a/mini_trainer/modeling/_quantized_training.py +++ b/mini_trainer/modeling/_quantized_training.py @@ -306,7 +306,12 @@ def _apply_weight_update(original, update, alpha, denominator=None): and 0 < original.shape[1] <= 16384 and isinstance(update, torch.Tensor) and update.shape == original.shape - and update.dtype == original.dtype + and ( + update.dtype == original.dtype + or denominator is None + and original.dtype == torch.float32 + and update.dtype in (torch.float16, torch.bfloat16) + ) and update.device == original.device and ( denominator is None diff --git a/mini_trainer/modeling/_quantized_update.py b/mini_trainer/modeling/_quantized_update.py index 75b2c80..fb973b8 100644 --- a/mini_trainer/modeling/_quantized_update.py +++ b/mini_trainer/modeling/_quantized_update.py @@ -40,7 +40,9 @@ def _update_rows( if DIVIDE: divisor = tl.load(denominator + row * denominator_row_stride + column * denominator_column_stride, valid, 1).to(tl.float32) change = (change / divisor).to(dtype).to(tl.float32) - change = (change * alpha).to(dtype).to(tl.float32) + # Muon produces BF16 updates for FP32 parameters. Match the eager + # update * alpha rounding before promotion in the weight addition. + change = (change * alpha).to(update.dtype.element_ty).to(tl.float32) values = (represented + change).to(dtype).to(tl.float32) maximum = tl.max(tl.where(valid, tl.abs(values), 0), 0) next_scale = (maximum / 127).to(dtype) diff --git a/tests/test_quantized_training_model.py b/tests/test_quantized_training_model.py index b4de067..8779a55 100644 --- a/tests/test_quantized_training_model.py +++ b/tests/test_quantized_training_model.py @@ -389,6 +389,44 @@ def apply(target): assert weight.scale.shape == (33,) +@pytest.mark.parametrize("dtype", [torch.float16, torch.bfloat16]) +@pytest.mark.parametrize("transposed", [False, True]) +def test_cuda_mixed_update_matches_materialized_trajectory(dtype, transposed, monkeypatch): + from mini_trainer.modeling._quantized_training import TrainingWeight + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify mixed-dtype storage updates") + assert torch.cuda.is_available() + torch.manual_seed(29) + weight = nn.Parameter(TrainingWeight.from_float(torch.randn(33, 67, device="cuda"))) + reference = nn.Parameter(weight.detach().clone()) + update = torch.randn((67, 33) if transposed else (33, 67), device="cuda", dtype=dtype) + if transposed: + update = update.T + with torch.no_grad(): + # Warm both kernels before comparing RNG consumption and trajectories. + warm = nn.Parameter(weight.detach().clone()) + warm.add_(update, alpha=-0.03) + warm.copy_(reference.dequantize()) + for alpha in (-0.03, 0.001, -0.1): + rng = torch.cuda.get_rng_state() + reference.copy_(reference.dequantize() + update * alpha) + expected_rng = torch.cuda.get_rng_state() + torch.cuda.set_rng_state(rng) + version = weight._version + + def forbidden_dequantize(self): + raise AssertionError("Mixed update materialized a floating weight matrix") + + with monkeypatch.context() as patch: + patch.setattr(TrainingWeight, "dequantize", forbidden_dequantize) + assert weight.add_(update, alpha=alpha) is weight + assert weight._version > version + assert torch.equal(torch.cuda.get_rng_state(), expected_rng) + assert torch.equal(weight.int_data, reference.int_data) + assert torch.equal(weight.scale, reference.scale) + + def test_cuda_storage_update_invalidates_saved_weight(): from mini_trainer.modeling._quantized_training import TrainingWeight From d675ff3875ec47f0b8e5db13ecf0bdfadae2488a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 17:34:23 +0200 Subject: [PATCH 076/155] fix: release replaced float weights in frozen training benchmarks --- dev/benchmarks/large_head_training.py | 3 ++ docs/benchmarks.md | 39 ++++++++++++++++++++++- docs/quantization-status.md | 10 +++--- tests/test_benchmark_large_head.py | 45 +++++++++++++++++++++++++++ 4 files changed, 92 insertions(+), 5 deletions(-) diff --git a/dev/benchmarks/large_head_training.py b/dev/benchmarks/large_head_training.py index fbcfbaa..9b9887c 100644 --- a/dev/benchmarks/large_head_training.py +++ b/dev/benchmarks/large_head_training.py @@ -112,6 +112,9 @@ def synchronize(): for parameter in model.parameters(): if id(parameter) not in head_ids: parameter.requires_grad_(False) + # Do not retain the final floating parameter after INT8 preparation + # replaces it; a large classification head can dominate this probe. + del parameter model.eval() head.train() else: diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 3548056..9984d11 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2196,7 +2196,9 @@ PyTorch 2.12.0+cu130 and TorchAO 0.17.0. Measured-phase CUDA allocated peaks were identical across the three seeds for each configuration. Native INT8 reduced the full-model peak by **17.4%**, but -**increased the frozen-backbone peak by 10.5%**. Parameter storage decreased from +**increased the frozen-backbone peak by 10.5%** in the original probe. This frozen +result was subsequently traced to a retained floating parameter in the probe; +see the correction below. Parameter storage decreased from about 0.559 to 0.197 GiB in both modes; that storage reduction is not proof of a runtime-memory reduction. The frozen-mode peak needs allocation profiling before choosing an optimization; this comparison alone does not identify its cause. @@ -2221,6 +2223,41 @@ convergence, loading, checkpoint/resume, scheduler and distributed behavior stil require the corresponding integrated profiles; this capacity probe does not certify those parts of the goal. +### Correcting the frozen-head allocation comparison + +Allocation tracing of the 100k-class BF16 probe placed the frozen-mode peak in +AdamW, which updates the final classification layer. A CUDA allocator snapshot +also identified a live 512,000,000-byte allocation from the original floating +head constructor after INT8 preparation. The probe's backbone-freezing loop +retained its last `parameter` local after preparation replaced that parameter. +The corrected probe releases that reference before preparation. This was a +measurement artifact, not a required floating master weight for native training. +GPU lifetime regressions check that replaced parameters have been released before +optimizer construction for both flat and hierarchical heads. + +The corrected flat frozen run measured 2.616 GiB allocated peak versus 2.800 GiB +for float, a 6.6% reduction, instead of the original 3.093 GiB INT8 peak. The +hierarchical frozen pair measured 2.617 GiB INT8 versus 2.801 GiB float, also 6.6% +lower. Each uses +the same seed 42, 100k classes, batch 32, image size 128, three warmup and five +measured updates, normalized symmetric EfficientNetV2-S and BF16 settings. The +full-model comparison remained 3.819 GiB float versus 3.154 GiB INT8. Reports and +diagnostic allocation snapshots are retained under ignored `tmp-mixed-update/`. +CPU regression checks overlapped the corrected diagnostic runs, so these runs +support allocation conclusions only, not new timing claims or target certification. + +A separate update-path improvement handles FP16/BF16 optimizer updates to FP32 +INT8 weights without materializing a floating weight matrix. It preserves the +update dtype's multiplication rounding before FP32 addition. CUDA regressions +compare exact codes, scales and RNG consumption across repeated updates and +transposed inputs, and forbid dequantization during the fused update. This closes +a Muon fallback but did not change the measured full-model or frozen peak by +itself; it must not be credited with the probe correction's memory reduction. + +Static checks passed. The CPU-default regression run passed 482 tests, with 156 +skips and the known EMA expected failure. Separate intentional CUDA runs passed +the 11 focused update/version checks and both new parameter-lifetime checks. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 66bd2e8..9007102 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -14,7 +14,7 @@ or integer operator count sufficient evidence of production readiness. | Workstream | Verified locally | What remains unproven | | --- | --- | --- | -| Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions; maintained 100k-class full/frozen BF16 capacity comparison: full-model peak 17.4% lower, frozen peak 10.5% higher, mixed timing; bounded preparation and initialization | Frozen-mode peak allocation profiling; benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | +| Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions; maintained 100k-class BF16 capacity comparison: full-model peak 17.4% lower; corrected frozen probe peak 6.6% lower after releasing an obsolete floating parameter; mixed timing; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | | QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | | Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | @@ -95,9 +95,11 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur separately. Optimize the dominant measured costs rather than assuming integer arithmetic is faster. Include full and frozen-backbone training/fine-tuning. The [100k-class BF16 capacity study](benchmarks.md#maintained-100k-class-bf16-training-comparison) - now covers both modes and heads across three seeds. Profile the reproducible - frozen-mode INT8 peak regression before treating that mode as a memory benefit; - repeat with realistic pretrained features and longer integrated training. + covers both modes and heads across three seeds. Its apparent frozen-mode + regression was a retained floating parameter in the probe; the + [corrected allocation comparison](benchmarks.md#correcting-the-frozen-head-allocation-comparison) + shows a 6.6% local peak reduction for both heads. Repeat the corrected probe + across seeds and with realistic pretrained features and longer integrated training. 3. **Qualify the trained-checkpoint-to-deployment contract.** Explicit conversion and calibration now connect the matched native QT checkpoints to distinct diff --git a/tests/test_benchmark_large_head.py b/tests/test_benchmark_large_head.py index bb96187..b5af5ea 100644 --- a/tests/test_benchmark_large_head.py +++ b/tests/test_benchmark_large_head.py @@ -25,3 +25,48 @@ def test_native_quantized_training_is_not_reported_as_cpu_training(tmp_path): with pytest.raises(ValueError, match="requires CUDA"): run(tmp_path / "invalid", device="cpu", dtype="float32", quantized=True) assert not (tmp_path / "invalid").exists() + + +@pytest.mark.parametrize("hierarchical", [False, True]) +def test_cuda_frozen_probe_releases_replaced_float_parameters(tmp_path, monkeypatch, hierarchical): + import os + import weakref + + import torch + + from dev.benchmarks import large_head_training as probe + + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 to verify frozen INT8 parameter lifetime") + assert torch.cuda.is_available() + prepare = probe.prepare_quantized_training + build_optimizer = probe.BaseBuilder.build_optimizer + replaced = [] + + def capture_replacements(model): + before = {id(p): weakref.ref(p) for p in model.parameters()} + recipe = prepare(model) + after = {id(p) for p in model.parameters()} + replaced.extend(ref for key, ref in before.items() if key not in after) + return recipe + + def verify_released(*args, **kwargs): + assert replaced + assert all(ref() is None for ref in replaced), "Probe retains replaced floating parameters" + return build_optimizer(*args, **kwargs) + + monkeypatch.setattr(probe, "prepare_quantized_training", capture_replacements) + monkeypatch.setattr(probe.BaseBuilder, "build_optimizer", verify_released) + result = run( + tmp_path / "quantized", + classes=101, + hierarchical=hierarchical, + batch_size=2, + image_size=32, + warmup=1, + steps=1, + frozen=True, + quantized=True, + dtype="bfloat16", + ) + assert result["status"] == "measured" From 8a15a75eb1212b31fad3f31c2359b9ede98da884 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 17:37:57 +0200 Subject: [PATCH 077/155] test: isolate fresh INT8 tuning from compiled kernel caches --- docs/benchmarks.md | 5 +++++ tests/test_quantized_training.py | 16 ++++++++++++++++ 2 files changed, 21 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 9984d11..e14cb7c 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2257,6 +2257,11 @@ itself; it must not be credited with the probe correction's memory reduction. Static checks passed. The CPU-default regression run passed 482 tests, with 156 skips and the known EMA expected failure. Separate intentional CUDA runs passed the 11 focused update/version checks and both new parameter-lifetime checks. +The broader affected CUDA suite passed all 95 tests after isolating its existing +fresh-tuning regression in a subprocess with compiled caches disabled at startup. +Initially that test failed because no tuning callback ran; clearing only the +tuner cache or disabling graph caches late in a shared process was insufficient. +The isolated test retains its tuning, finite-gradient and optimizer assertions. ### Maintained image preparation reproduction diff --git a/tests/test_quantized_training.py b/tests/test_quantized_training.py index eaaf244..7a90b70 100644 --- a/tests/test_quantized_training.py +++ b/tests/test_quantized_training.py @@ -229,6 +229,22 @@ def test_dense_backward_graph_with_fresh_kernel_tuning(monkeypatch): pytest.importorskip("torchao") if os.environ.get("RUN_CUDA_TESTS") != "1": pytest.skip("Set RUN_CUDA_TESTS=1 for fresh INT8 kernel tuning in CUDA graphs") + # Earlier compiled tests can retain generated kernels in process even when + # graph caches are disabled later. Start this fresh-tuning contract with + # caches disabled before importing the compiler; keep all assertions below. + if os.environ.get("MINI_TRAINER_FRESH_TUNING_CHILD") != "1": + import subprocess + import sys + + environment = dict(os.environ, MINI_TRAINER_FRESH_TUNING_CHILD="1", TORCHINDUCTOR_FORCE_DISABLE_CACHES="1") + result = subprocess.run( + [sys.executable, "-m", "pytest", f"{__file__}::test_dense_backward_graph_with_fresh_kernel_tuning", "-q"], + env=environment, + capture_output=True, + text=True, + ) + assert result.returncode == 0, result.stdout + result.stderr + return from dev.benchmarks.models import DenseImageMLP from mini_trainer.modeling import Classifier, EmbeddingContext from mini_trainer.modeling import _quantized_matmul as matmul From ee21642a1c6ffaf436a0d93d101ecc3fdd21f108 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 17:56:32 +0200 Subject: [PATCH 078/155] feat: expose parameter-frozen dataset fine-tuning benchmarks --- dev/benchmarks/README.md | 12 +++++++++++ dev/benchmarks/run.py | 11 +++++++++++ dev/benchmarks/summarize.py | 9 +++++---- tests/test_benchmark_datasets.py | 33 ++++++++++++++++++++++++++++--- tests/test_benchmark_synthetic.py | 2 ++ 5 files changed, 60 insertions(+), 7 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index c8f4f7b..fdc5e56 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -328,6 +328,18 @@ The CPU integration test exercises both real EfficientNetV2 heads with synthetic image files and a deliberately reordered taxonomy; it checks training, reload, predictions, class order and identical split manifests without a download. +Add `--pretrained --fine-tune` to both precision runs for the existing builder's +parameter-frozen fine-tuning regime. The benchmark keeps floating backbone +parameters in FP32 and uses the requested AMP dtype for compute; INT8 preparation +still follows the ordinary recipe and reports its actual operator coverage. +Backbone parameters receive no optimizer updates, but BatchNorm running statistics +and dropout retain normal training behavior. This is distinct from the +`large_head_training --frozen` capacity probe, which evaluates the backbone. +Reports record `fine_tune`, `backbone_floating_dtype` and `backbone_training_mode`; +the option is also retained in failure reports. Apply identical settings and seeds +to both precision runs. The flag alone does not establish a speed, memory or +quality benefit, and random frozen features are only an execution diagnostic. + ### Validation when target hardware is unavailable Use the local GPU to vary batch size, resolution, cache mode and worker count diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 0054d88..1de4c2b 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -62,6 +62,7 @@ def run( normalized: bool | None = None, image_size: int | None = None, pretrained: bool = False, + fine_tune: bool = False, batch_size: int = 32, cache_workers: int | None = None, model_profile: str = "default", @@ -184,6 +185,8 @@ def run( optimizer_cudagraphs=optimizer_cudagraphs, model_builder_kwargs={ "model_type": model_type, + "fine_tune": fine_tune, + "fine_tune_dtype": torch.float32, "hidden": hidden if hidden else False, "normalized": normalized, **({"model_args": {"pretrained": pretrained}} if backbone else {}), @@ -283,6 +286,9 @@ def run( "normalized": normalized, "image_size": size, "pretrained": pretrained, + "fine_tune": fine_tune, + "backbone_floating_dtype": "float32", + "backbone_training_mode": "train", "hidden_width": classification_module(model).preclassification_size, "optimizer": optimizer, "learning_rate": learning_rate, @@ -384,6 +390,9 @@ def main(): parser.add_argument("--normalized", action=BooleanOptionalAction, default=None) parser.add_argument("--image-size", type=int) parser.add_argument("--pretrained", action="store_true", help="Allow downloading pretrained backbone weights.") + parser.add_argument( + "--fine-tune", action="store_true", help="Freeze backbone parameters; retain normal training modes and FP32 storage." + ) parser.add_argument("--batch-size", type=int, default=32) parser.add_argument( "--allow-nondeterministic", @@ -422,6 +431,7 @@ def main(): normalized=args.normalized, image_size=args.image_size, pretrained=args.pretrained, + fine_tune=args.fine_tune, batch_size=args.batch_size, cache_workers=args.cache_workers, model_profile=args.model_profile, @@ -451,6 +461,7 @@ def main(): "normalized": args.normalized, "image_size": args.image_size, "pretrained": args.pretrained, + "fine_tune": args.fine_tune, "batch_size": args.batch_size, "cache_workers": args.cache_workers, "model_profile": args.model_profile, diff --git a/dev/benchmarks/summarize.py b/dev/benchmarks/summarize.py index 5886d0f..fb96cf9 100644 --- a/dev/benchmarks/summarize.py +++ b/dev/benchmarks/summarize.py @@ -11,8 +11,8 @@ def summarize(directory: Path) -> str: "# Dataset benchmark results", "", "| Run | Status | Device / precision | QT coverage | Accuracy by level | Parameter bytes | Peak CUDA MiB | " - "Median train epoch 3+ | Training wall time |", - "| --- | --- | --- | --- | --- | --- | --- | --- | --- |", + "Median train epoch 3+ | Training wall time | Backbone parameters |", + "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", ] reports = sorted(directory.rglob("report.json")) for path in reports: @@ -47,10 +47,10 @@ def summarize(directory: Path) -> str: ) lines.append( f"| {name} | {report['status']} | {device} | {quantization} | {accuracy} | {parameter_bytes} | " - f"{peak_memory} | {later_duration} | {duration} |" + f"{peak_memory} | {later_duration} | {duration} | {'frozen' if report.get('fine_tune') else 'trainable'} |" ) if not reports: - lines.append("| No reports produced | incomplete | — | — | — | — | — | — | — |") + lines.append("| No reports produced | incomplete | — | — | — | — | — | — | — | — |") lines.extend( [ "", @@ -61,6 +61,7 @@ def summarize(directory: Path) -> str: "only with matching hardware, dataset/configuration and timing scope. See JSON reports", "for provenance, errors and explicit coverage flags. CPU results do not validate GPU behavior.", "QT coverage counts quantized Linear modules; other operations may remain floating point.", + "Frozen backbone parameters do not imply evaluation mode: fine-tuning retains normal BatchNorm/dropout behavior.", "Parameter bytes describe stored parameters. CUDA peaks cover training, excluding final held-out inference.", "Older CUDA readings without a scope marker are unverified because logger resets could hide earlier peaks.", "Later-epoch medians use timed training phases from epoch 3 onward, including loading, preprocessing and batch logging.", diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index eaedfec..b329c7b 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -60,13 +60,17 @@ def test_blair_requires_explicit_covering_taxonomy(tmp_path): prepare_real(root, tmp_path, name="blair", seed=42, class_spec=path) -def test_summary_preserves_failures_and_unmeasured_fields(tmp_path): +@pytest.mark.parametrize("fine_tune", [False, True]) +def test_summary_preserves_failures_and_unmeasured_fields(tmp_path, fine_tune): path = tmp_path / "failed" path.mkdir() - (path / "report.json").write_text(json.dumps({"status": "failed", "device": "cuda:0", "dtype": "float16", "quantized_training": True})) + (path / "report.json").write_text( + json.dumps({"status": "failed", "device": "cuda:0", "dtype": "float16", "quantized_training": True, "fine_tune": fine_tune}) + ) summary = summarize(tmp_path) assert "| failed | failed | cuda:0 / float16 | requested | — | — | — | — |" in summary assert "CPU results do not validate GPU" in summary + assert (" | frozen |" if fine_tune else " | trainable |") in summary assert "No reports produced" in summarize(Path(tmp_path / "missing")) @@ -193,11 +197,28 @@ def test_summary_later_epoch_median_requires_timing_scope(tmp_path): assert "| — | 123.00s |" in summarize(tmp_path) -def test_efficientnet_flat_and_hierarchical_share_blair_splits(tmp_path): +@pytest.mark.parametrize("fine_tune", [False, True]) +def test_efficientnet_flat_and_hierarchical_share_blair_splits(tmp_path, monkeypatch, fine_tune): import numpy as np import torch from dev.benchmarks.run import run + from mini_trainer.builders import BaseBuilder + from mini_trainer.modeling import classification_module + + captured = {} + build_optimizer = BaseBuilder.build_optimizer + + def inspect_optimizer(model, *args, **kwargs): + head_ids = {id(p) for p in classification_module(model).parameters()} + backbone = [p for p in model.parameters() if id(p) not in head_ids] + assert backbone and all(p.requires_grad != fine_tune for p in backbone) + if fine_tune: + captured["backbone"] = [(p, p.detach().clone()) for p in backbone] + captured["head"] = [(p, p.detach().clone()) for p in classification_module(model).parameters() if p.requires_grad] + return build_optimizer(model, *args, **kwargs) + + monkeypatch.setattr(BaseBuilder, "build_optimizer", inspect_optimizer) root = tmp_path / "images" make_dataset(root) @@ -227,8 +248,13 @@ def test_efficientnet_flat_and_hierarchical_share_blair_splits(tmp_path): image_size=32, batch_size=4, cache_workers=0, + fine_tune=fine_tune, ) ) + if fine_tune: + assert all(torch.equal(p, initial) and p.grad is None for p, initial in captured["backbone"]) + assert any(not torch.equal(p, initial) for p, initial in captured["head"]) + captured.clear() finally: torch.set_num_threads(previous_threads) flat, hierarchical = reports @@ -237,6 +263,7 @@ def test_efficientnet_flat_and_hierarchical_share_blair_splits(tmp_path): assert hierarchical["class_mapping"]["0"] == flat["class_mapping"] assert flat["hidden_width"] == hierarchical["hidden_width"] == 1280 assert all(report["normalized"] and not report["pretrained"] for report in reports) + assert all(report["fine_tune"] == fine_tune and report["backbone_training_mode"] == "train" for report in reports) with np.load(tmp_path / "flat/predictions.npz") as a, np.load(tmp_path / "hierarchical/predictions.npz") as b: np.testing.assert_array_equal(a["paths"], b["paths"]) np.testing.assert_array_equal(a["labels"], b["labels"]) diff --git a/tests/test_benchmark_synthetic.py b/tests/test_benchmark_synthetic.py index f5e5f5f..0b636fc 100644 --- a/tests/test_benchmark_synthetic.py +++ b/tests/test_benchmark_synthetic.py @@ -81,6 +81,7 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): "reduce-overhead", "--compile-optimizer", "--optimizer-cudagraphs", + "--fine-tune", ], ) monkeypatch.setattr(torch.cuda, "is_available", lambda: False) @@ -93,6 +94,7 @@ def test_cli_retains_failure_report(tmp_path, monkeypatch): assert report["device"] == "cuda:0" assert report["compile_mode"] == "reduce-overhead" assert report["optimizer_cudagraphs"] is True + assert report["fine_tune"] is True assert report["error"]["type"] == "RuntimeError" assert "test_accuracy" not in report From cd9ad68e80b5afd6be65b51ee690acba7f398baf Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 17:57:05 +0200 Subject: [PATCH 079/155] docs: report pretrained Blair fine-tuning quantization results --- docs/benchmarks.md | 63 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 12 +++++++ 2 files changed, 75 insertions(+) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index e14cb7c..cedc5b3 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2263,6 +2263,69 @@ Initially that test failed because no tuning callback ran; clearing only the tuner cache or disabling graph caches late in a shared process was insufficient. The isolated test retains its tuning, finite-gradient and optimizer assertions. +### Pretrained parameter-frozen Blair fine-tuning + +The dataset runner now exposes the existing builder's `--fine-tune` option and +records it in successful/failed reports and rendered summaries. Its floating +backbone storage remains FP32, with the selected AMP compute dtype. Backbone +parameters are frozen, while BatchNorm statistics and dropout retain normal +training behavior. This differs from the evaluation-mode backbone in the +synthetic capacity probe. CPU integration checks verify unchanged backbone +parameters, head updates, checkpoint reload and matched splits for both real +EfficientNetV2-S head configurations. + +The first local comparison uses cached pretrained EfficientNetV2-S, symmetric +normalized flat/hierarchical heads, BF16 autocast, MuonAuxAdamW, learning rate +0.01, five epochs, batch 32, image size 128 and seed 42. Data remain 3,704 training, +912 validation and 1,161 held-out test images, with 25 leaves and 15 parents. +The common dataset manifest SHA256 is +`1cef6c7d9133d889b6c8eff0ffef29b1db5d6653ab4adf0c005739af8c2921c6`. +The two native INT8 recipes quantize `classifier.hidden` and `classifier.linear`; +convolutions, gradients and optimizer states remain floating. Each trained +checkpoint was reloaded before collecting finite held-out predictions. + +`mini_metrics` evaluates fixed argmax predictions at threshold zero, with no +abstention or threshold tuning, independently at each level. The following are +unit-interval values, not percentages: + +| Head / level | Precision | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | --- | ---: | ---: | ---: | ---: | ---: | +| Flat / leaf | Float | 0.479750 | 0.470175 | 0.530647 | 1.0 | 0.573066 | +| Flat / leaf | INT8 | 0.482557 | 0.470687 | 0.533579 | 1.0 | 0.572262 | +| Hierarchical / leaf | Float | 0.479853 | 0.468765 | 0.532174 | 1.0 | 0.566776 | +| Hierarchical / leaf | INT8 | 0.480858 | 0.475068 | 0.530723 | 1.0 | 0.576768 | +| Hierarchical / parent | Float | 0.638768 | 0.620933 | 0.683742 | 1.0 | 0.637285 | +| Hierarchical / parent | INT8 | 0.643042 | 0.620931 | 0.699244 | 1.0 | 0.639065 | + +Macro-F1 differences are +0.281, +0.100 and +0.427 percentage points respectively. +Other metrics move in both directions. Predictions change on 333, 327 and 214 +of the 1,161 images. These are separately trained stochastic models, not a score +parity check or evidence that quantization generally improves quality. Matching +the initial seed does not synchronize dropout with stochastic quantized updates. +The absolute leaf Macro-F1 near 0.48 also does not establish useful production +quality or parity with full-backbone training. + +Training allocated peaks were approximately 206.90 MiB float and 170.03 MiB INT8 +for both heads (17.8% lower). These are local PyTorch allocation measurements, +not total process/device memory. CPU regression checks overlapped the GPU runs; +their recorded durations are excluded from throughput or time-to-quality claims. +Repeat isolated, alternating multi-seed comparisons and full-backbone baselines +before accepting an efficiency/quality trade-off, then verify the target hardware. + +Ignored `tmp-fine-tune-blair/` retains the paired command driver, checkpoints, +training reports, NPZ predictions, dataset manifests and logs. Its +`prepare-quality.py` validates identities/labels/mappings and creates hashed CSV +and held-out manifests for the maintained `dev.benchmarks.quality_compare` +command, executed in the existing separate `mini_metrics` environment. Quality +reports include source hashes for that package and all five metrics. Making the +NPZ-to-quality adapter a generic maintained command and publishing durable results +remain follow-up work; local artifacts alone are not a continuous pipeline. + +Validation passed static checks and 483 CPU-default tests, with 158 skips and +the known EMA expected failure. Five additional focused summary/CLI checks passed +after adding the visible backbone-policy column. All four real GPU training and +reload runs and both five-metric evaluations completed successfully. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 9007102..faf0296 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -29,6 +29,15 @@ that the native dynamic training quantizer exports unchanged into integer GPU execution. DDP/FSDP remain unsupported for native QT. EMA repair is explicitly deferred and is not a prerequisite for this goal. +The dataset runner now exposes parameter-frozen fine-tuning through the existing +builder. A [five-epoch pretrained Blair comparison](benchmarks.md#pretrained-parameter-frozen-blair-fine-tuning) +completed both normalized symmetric heads with BF16 and native INT8 and evaluated +all five requested metrics on 1,161 held-out images. Macro-F1 differences were +small, with mixed changes in other metrics; this single seed is not evidence of +a general quality improvement. Local allocated peaks were 17.8% lower, but CPU +checks overlapped the runs, so no timing benefit is claimed. This regime retains +training-mode BatchNorm/dropout, unlike the synthetic evaluation-mode backbone. + The earlier floating-checkpoint TensorRT timing comparison uses matched builder settings and three fresh paired processes per head. At batches 1 and 8, INT8 did not beat FP16 in local host latency. Engines are approximately 43% smaller, while reported execution @@ -100,6 +109,9 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur [corrected allocation comparison](benchmarks.md#correcting-the-frozen-head-allocation-comparison) shows a 6.6% local peak reduction for both heads. Repeat the corrected probe across seeds and with realistic pretrained features and longer integrated training. + The initial parameter-frozen pretrained Blair study now provides execution and + quality evidence; repeat it in isolated processes across seeds alongside matched + full-backbone baselines to establish time to useful quality. 3. **Qualify the trained-checkpoint-to-deployment contract.** Explicit conversion and calibration now connect the matched native QT checkpoints to distinct From 393a4fae8307af6d8f0d24cfa3d033729930b7c2 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 18:13:16 +0200 Subject: [PATCH 080/155] feat: connect saved training predictions to paired quality evaluation --- dev/benchmarks/README.md | 45 +++++ dev/benchmarks/training_predictions.py | 167 +++++++++++++++++++ docs/benchmarks.md | 40 ++++- docs/quantization-status.md | 5 + tests/test_benchmark_training_predictions.py | 131 +++++++++++++++ 5 files changed, 385 insertions(+), 3 deletions(-) create mode 100644 dev/benchmarks/training_predictions.py create mode 100644 tests/test_benchmark_training_predictions.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index fdc5e56..0ef501c 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1383,3 +1383,48 @@ real-data profiles for those comparisons. Run target GPU generations separately; a laptop result cannot certify A40/A100/B300 or the intended desktop. Keep full models at 100k classes or below on the laptop, and increase capacity only on a machine provisioned for it. No weights or datasets are saved by this probe. + +## Training predictions to paired quality evaluation + +Use saved `dev.benchmarks.run` output directories to prepare the same held-out +contract consumed by `quality_compare`, independently of the training device: + +```bash +.venv/bin/python -m dev.benchmarks.training_predictions \ + --baseline results/float --candidate results/int8 --output results/quality-inputs +# Use an explicitly prepared environment containing mini_metrics: +/path/to/metrics-python -m dev.benchmarks.quality_compare \ + --manifest results/quality-inputs/manifest.json --output results/quality +``` + +Run from the repository root. This installs nothing and does not import PyTorch, +load checkpoints or rerun inference. The adapter accepts the runner's synthetic +and real-dataset layouts, flat heads and any number of reported hierarchy levels. +Transfer `report.json`, `predictions.npz` and the dataset manifest from the training +machine; checkpoints and source images are not needed for this evaluation step. +The training report's checkpoint identifier is retained, not independently +verified against a checkpoint file. + +Both runs must have completed training, checkpoint reload and inference, and +must declare that test images were not used for training or selection. A synthetic +run that completed inference but missed its quality gate can still be evaluated; +its original failure status remains in the manifest. Execution failures without +completed inference are rejected. The adapter checks each dataset-manifest hash, +held-out paths/labels, ordered class mappings, image hashes, archive levels, +leaf aliases and finite floating scores before creating output. Different +manifest order or training metadata is allowed when the held-out identities and +image hashes agree; different test labels, image hashes or class order is rejected. + +The new directory contains canonical `baseline.csv`, `candidate.csv` and +`manifest.json`, with source hashes and both original training reports. Fixed +argmax predictions use row-wise softmax confidence at threshold zero, without +abstention or threshold optimization. Confidence conversion does not allocate a +second dense float64 score matrix, although NumPy still loads an archive's score +array into memory. The confidence values are not calibration evidence. The +existing evaluator computes Macro-F1, Macro-Recall, Macro-Precision, Coverage and +Theil's U independently per level, and records its mini_metrics source hashes. + +This is a saved-prediction quality comparison. It does not establish runtime +performance, checkpoint-to-score parity, source-image integrity beyond the +recorded hashes, or production acceptance. Preserve the evaluator's report and +CSV/manifest bundle alongside the original training reports for continuous runs. diff --git a/dev/benchmarks/training_predictions.py b/dev/benchmarks/training_predictions.py new file mode 100644 index 0000000..3c9cb0e --- /dev/null +++ b/dev/benchmarks/training_predictions.py @@ -0,0 +1,167 @@ +"""Prepare paired training-run predictions for the five-metric quality evaluator.""" + +import csv +import json +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np + +from .onnx_inference import file_hash +from .quality_compare import COLUMNS, read_manifest, read_predictions, validate_dataset + + +def _is_sha256(value): + return isinstance(value, str) and len(value) == 64 and all(c in "0123456789abcdef" for c in value) + + +def _read_run(directory): + directory = Path(directory).resolve() + report_path = directory / "report.json" + report = json.loads(report_path.read_bytes()) + if ( + report.get("schema_version") != 1 + or report.get("status") not in ("passed", "completed", "failed") + or not all(report.get("coverage", {}).get(key) is True for key in ("training", "checkpoint_reload", "inference")) + or report.get("test_used_for_training_or_selection") is not False + or report.get("score_semantics") != "model_eval_forward" + or not _is_sha256(report.get("checkpoint_sha256")) + ): + raise ValueError("Require a completed training/reload/inference report with unused held-out test predictions") + mapping = report["class_mapping"] + if report.get("head") == "flat": + mappings = [mapping] + elif report.get("head") == "hierarchical" and set(mapping) == {str(i) for i in range(len(mapping))}: + mappings = [mapping[str(i)] for i in range(len(mapping))] + else: + raise ValueError("Require flat or contiguous hierarchical class mappings") + classes = [] + for level in mappings: + if not level or any(type(i) is not int for i in level.values()) or set(level.values()) != set(range(len(level))): + raise ValueError("Class indices must be unique contiguous integers starting at zero") + classes.append([name for name, _ in sorted(level.items(), key=lambda item: item[1])]) + inventory_path = directory / ("data/manifest.json" if report.get("dataset") == "synthetic" else "dataset_manifest.json") + if file_hash(inventory_path) != report.get("dataset_manifest_sha256"): + raise ValueError("Dataset manifest hash does not match the training report") + inventory = json.loads(inventory_path.read_bytes()) + records = [record for record in inventory["records"] if record["split"] == "test"] + paths = [record["path"] for record in records] + if not paths or len(set(paths)) != len(paths): + raise ValueError("Require nonempty, unique held-out image paths") + samples = [] + image_hashes = {} + for identifier, record in enumerate(sorted(records, key=lambda record: record["path"])): + targets = [record["label"]] if len(classes) == 1 else record.get("targets", []) + if len(targets) != len(classes) or any(type(i) is not int or not 0 <= i < len(c) for i, c in zip(targets, classes, strict=True)): + raise ValueError("Held-out targets do not match the class mappings") + digest = record.get("sha256", "") + if not _is_sha256(digest): + raise ValueError("Require held-out image SHA256 identifiers") + image_hashes[record["path"]] = digest + samples.append( + {"instance_id": identifier, "filename": record["path"], "labels": [c[i] for c, i in zip(classes, targets, strict=True)]} + ) + metadata = { + "schema_version": 1, + "split": "test", + "levels": [{"name": "leaf" if i == 0 else f"level_{i}", "classes": c} for i, c in enumerate(classes)], + "samples": samples, + "provenance": {"dataset_manifest_sha256": file_hash(inventory_path), "image_sha256": image_hashes}, + } + validate_dataset(metadata) + predictions_path = directory / "predictions.npz" + tables = [] + with np.load(predictions_path, allow_pickle=False) as data: + if not np.array_equal(data["paths"], paths): + raise ValueError("Prediction paths differ from the held-out manifest order") + expected_keys = {f"{key}_{level}" for level in range(len(classes)) for key in ("scores", "labels")} + if set(data.files) != expected_keys | {"scores", "labels", "paths"}: + raise ValueError("Prediction archive has missing or unexpected levels/arrays") + for key in ("scores", "labels"): + if not np.array_equal(data[key], data[f"{key}_0"]): + raise ValueError("Leaf prediction aliases disagree") + indices = {path: i for i, path in enumerate(paths)} + for level, names in enumerate(classes): + scores, labels = data[f"scores_{level}"], data[f"labels_{level}"] + targets = [r["label"] if len(classes) == 1 else r["targets"][level] for r in records] + if labels.dtype.kind not in "iu" or not np.array_equal(labels, targets): + raise ValueError("Prediction labels differ from the held-out manifest") + if scores.dtype.kind != "f" or scores.shape != (len(paths), len(names)): + raise ValueError("Require finite floating scores with one row per image and column per class") + # This is a fixed argmax comparison, not a confidence calibration + # claim. Softmax supplies a bounded confidence for threshold zero. + for sample in samples: + row = indices[sample["filename"]] + values = scores[row].astype(np.float64) + if not np.isfinite(values).all(): + raise ValueError("Require finite floating scores") + prediction = int(values.argmax()) + confidence = float(1 / np.exp(values - values[prediction]).sum()) + tables.append( + ( + sample["instance_id"], + sample["filename"], + level, + names[labels[row]], + names[prediction], + confidence, + 0.0, + ) + ) + provenance = { + "report_path": str(report_path), + "report_sha256": file_hash(report_path), + "predictions_sha256": file_hash(predictions_path), + "training_report": report, + "checkpoint_verification": "Checkpoint identifier is taken from the training report; no checkpoint is loaded.", + } + return metadata, tables, provenance + + +def prepare(baseline, candidate, output): + """Validate both inputs before creating a portable quality-evaluation bundle.""" + output = Path(output) + if output.exists(): + raise FileExistsError(output) + runs = {mode: _read_run(path) for mode, path in (("baseline", baseline), ("candidate", candidate))} + metadata = runs["baseline"][0] + other = runs["candidate"][0] + if metadata["levels"] != other["levels"] or metadata["samples"] != other["samples"]: + raise ValueError("Paired runs must have identical held-out identities, labels and ordered class mappings") + if metadata["provenance"]["image_sha256"] != other["provenance"]["image_sha256"]: + raise ValueError("Paired held-out image hashes differ") + metadata["provenance"].update( + adapter_sha256=file_hash(__file__), + policy="Fixed argmax, softmax confidence, threshold zero; no threshold tuning or abstention.", + scope="Saved test predictions only; does not rerun inference, verify source images or establish performance/acceptance.", + ) + output.mkdir(parents=True) + for mode, (_, rows, provenance) in runs.items(): + path = output / f"{mode}.csv" + with path.open("w", newline="") as stream: + writer = csv.writer(stream) + writer.writerow(COLUMNS) + writer.writerows(rows) + metadata[mode] = { + "path": path.name, + "sha256": file_hash(path), + "classes": [s["classes"] for s in metadata["levels"]], + "provenance": provenance, + } + read_predictions(path, metadata[mode], metadata) + manifest = output / "manifest.json" + manifest.write_text(json.dumps(metadata, indent=2, allow_nan=False) + "\n") + read_manifest(manifest) + return manifest + + +def main(): + parser = ArgumentParser(description=__doc__) + parser.add_argument("--baseline", type=Path, required=True, help="Completed dataset benchmark directory") + parser.add_argument("--candidate", type=Path, required=True, help="Paired dataset benchmark directory") + parser.add_argument("--output", type=Path, required=True, help="New directory for CSVs and quality manifest") + print(prepare(**vars(parser.parse_args()))) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index cedc5b3..e9a91cf 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2317,15 +2317,49 @@ training reports, NPZ predictions, dataset manifests and logs. Its `prepare-quality.py` validates identities/labels/mappings and creates hashed CSV and held-out manifests for the maintained `dev.benchmarks.quality_compare` command, executed in the existing separate `mini_metrics` environment. Quality -reports include source hashes for that package and all five metrics. Making the -NPZ-to-quality adapter a generic maintained command and publishing durable results -remain follow-up work; local artifacts alone are not a continuous pipeline. +reports include source hashes for that package and all five metrics. The maintained +adapter below replaces the local conversion script. Publishing durable results +remains follow-up work; local artifacts alone are not a continuous pipeline. Validation passed static checks and 483 CPU-default tests, with 158 skips and the known EMA expected failure. Five additional focused summary/CLI checks passed after adding the visible backbone-policy column. All four real GPU training and reload runs and both five-metric evaluations completed successfully. +### Maintained training prediction quality inputs + +`dev.benchmarks.training_predictions` now connects saved dataset benchmark runs +to `quality_compare` without model loading or a dependency on the training device. +It supports the synthetic/real manifest layouts and flat or arbitrary-depth +hierarchical outputs. The [command](../dev/benchmarks/README.md#training-predictions-to-paired-quality-evaluation) +produces canonical paired CSVs and a portable held-out manifest, retaining both +source reports, prediction hashes, reported checkpoint identifiers and image hashes. +It validates the actual dataset-manifest bytes against each report and checks +archive identities, labels, class order, finite scores and leaf aliases before +creating output. Reordered records are canonicalized; mismatched held-out evidence +is rejected. It does not independently reload checkpoints or read source images. + +The retained pretrained Blair fine-tuning pairs were converted and reevaluated +with the existing separate mini_metrics environment. All five metrics reproduced +the earlier values exactly for flat leaves and hierarchical leaves/parents. +This replaces a local data-conversion step, not the models, predictions, metric +policy or performance evidence. Conversion uses row-wise softmax confidence at +threshold zero, without an additional dense float64 probability matrix; the +source score array is still loaded by NumPy. Transported reports, NPZ predictions +and dataset manifests are sufficient, without large checkpoint/image transfers. + +Focused tests cover flat, three-level hierarchical and synthetic quality-failure +reports; exact identity canonicalization; literal labels such as `001`; unchanged +inputs; and rejection of inconsistent hashes, mappings, labels, paths and arrays. +Execution failures without completed inference are rejected. Evaluating a +completed synthetic run that missed its oracle gate does not mark that gate as +passed: its original status is retained in the bundle. Durable publication and +profile-specific acceptance gates remain unfinished. + +Static checks passed. The full CPU-default suite passed 497 tests, with 158 skips +and the known EMA expected failure; all 17 final focused adapter checks also +passed. Final-code replays again reproduced both Blair metric reports exactly. + ### Maintained image preparation reproduction `dev.benchmarks.prepare_inputs` now generates both calibration and held-out NPZ diff --git a/docs/quantization-status.md b/docs/quantization-status.md index faf0296..8516a21 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -95,6 +95,11 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur and raw timing samples. Resolve or exclude inconsistent timing sources. The current detailed probes and engines are retained locally under ignored `tmp-*` directories; documentation alone does not make them a continuous pipeline. + The [training-prediction adapter](../dev/benchmarks/README.md#training-predictions-to-paired-quality-evaluation) + now also feeds saved flat/hierarchical training predictions into the shared + five-metric evaluator without checkpoint or image transfers. It reproduced the + pretrained fine-tuning metrics exactly; continuous orchestration and durable + result publication remain to be completed. 2. **Find and verify the beneficial workload regimes.** Pair INT8 with practical FP16/BF16 baselines while varying batch size, resolution and head size in a diff --git a/tests/test_benchmark_training_predictions.py b/tests/test_benchmark_training_predictions.py new file mode 100644 index 0000000..422ffd9 --- /dev/null +++ b/tests/test_benchmark_training_predictions.py @@ -0,0 +1,131 @@ +import json + +import numpy as np +import pytest + +from dev.benchmarks.onnx_inference import file_hash +from dev.benchmarks.quality_compare import read_manifest, read_predictions +from dev.benchmarks.training_predictions import prepare + + +def make_run(path, *, hierarchical=False, synthetic=False, reverse=False): + path.mkdir() + levels = 3 if hierarchical else 1 + records = [ + {"path": "b.jpg", "split": "test", "label": 0, "targets": [0] * levels, "sha256": "a" * 64}, + {"path": "a.jpg", "split": "test", "label": 1, "targets": [1] * levels, "sha256": "b" * 64}, + ] + if reverse: + records.reverse() + inventory_path = path / ("data/manifest.json" if synthetic else "dataset_manifest.json") + inventory_path.parent.mkdir(exist_ok=True) + inventory_path.write_text(json.dumps({"records": records})) + mapping = {"001": 0, "1": 1} + report = { + "schema_version": 1, + "status": "failed" if synthetic else "completed", + "coverage": {"training": True, "checkpoint_reload": True, "inference": True}, + "test_used_for_training_or_selection": False, + "score_semantics": "model_eval_forward", + "head": "hierarchical" if hierarchical else "flat", + "dataset": "synthetic" if synthetic else "blair", + "class_mapping": {str(i): mapping for i in range(levels)} if hierarchical else mapping, + "dataset_manifest_sha256": file_hash(inventory_path), + "checkpoint_sha256": "c" * 64, + } + (path / "report.json").write_text(json.dumps(report)) + labels = np.array([r["label"] for r in records], dtype=np.int64) + scores = np.eye(2, dtype=np.float32)[labels] * 1000 - 500 + arrays = {"paths": np.array([r["path"] for r in records]), "scores": scores, "labels": labels} + for i in range(levels): + arrays[f"scores_{i}"] = scores + arrays[f"labels_{i}"] = labels + np.savez(path / "predictions.npz", **arrays) + return path + + +@pytest.mark.parametrize("hierarchical,synthetic", [(False, False), (True, False), (False, True)]) +def test_adapter_canonicalizes_identity_and_preserves_literal_labels(tmp_path, hierarchical, synthetic): + a = make_run(tmp_path / "a", hierarchical=hierarchical, synthetic=synthetic) + b = make_run(tmp_path / "b", hierarchical=hierarchical, synthetic=synthetic, reverse=True) + before = {p: file_hash(p) for folder in (a, b) for p in folder.rglob("*") if p.is_file()} + manifest = prepare(a, b, tmp_path / "out") + metadata, _ = read_manifest(manifest) + assert [s["filename"] for s in metadata["samples"]] == ["a.jpg", "b.jpg"] + assert len(metadata["levels"]) == (3 if hierarchical else 1) + tables = [read_predictions(manifest.parent / metadata[m]["path"], metadata[m], metadata)[0] for m in ("baseline", "candidate")] + assert tables[0] == tables[1] + assert set(tables[0]["prediction"]) == {"001", "1"} + assert tables[0]["confidence"] == [1.0] * len(tables[0]["confidence"]) + assert {p: file_hash(p) for p in before} == before + # Results bundles can be transported without checkpoints or source images. + assert not (a / "training").exists() + with pytest.raises(FileExistsError): + prepare(a, b, tmp_path / "out") + + +@pytest.mark.parametrize( + "corruption", + [ + "labels", + "nan", + "aliases", + "paths", + "extra_level", + "class_indices", + "class_order", + "checkpoint", + "manifest_hash", + "incomplete", + "image_hash", + "targets", + "identity", + "target_identity", + ], +) +def test_adapter_rejects_inconsistent_evidence_before_writing(tmp_path, corruption): + a, b = make_run(tmp_path / "a"), make_run(tmp_path / "b") + report_path = b / "report.json" + report = json.loads(report_path.read_text()) + with np.load(b / "predictions.npz") as archive: + arrays = dict(archive) + if corruption == "labels": + arrays["labels"] = arrays["labels_0"] = np.array([1, 0]) + elif corruption == "nan": + arrays["scores"] = arrays["scores_0"] = np.full((2, 2), np.nan) + elif corruption == "aliases": + arrays["scores"] = -arrays["scores_0"] + elif corruption == "paths": + arrays["paths"] = arrays["paths"][::-1] + elif corruption == "extra_level": + arrays["scores_1"] = arrays["scores"] + elif corruption == "class_indices": + report["class_mapping"] = {"001": 1, "1": 1} + elif corruption == "class_order": + report["class_mapping"] = {"001": 1, "1": 0} + elif corruption == "checkpoint": + del report["checkpoint_sha256"] + elif corruption == "manifest_hash": + report["dataset_manifest_sha256"] = "0" * 64 + elif corruption == "incomplete": + report["coverage"]["inference"] = False + else: + source = b / "dataset_manifest.json" + inventory = json.loads(source.read_text()) + if corruption == "image_hash": + inventory["records"][0]["sha256"] = "d" * 64 + elif corruption == "identity": + inventory["records"][0]["path"] = "z.jpg" + arrays["paths"] = np.array(["z.jpg", "a.jpg"]) + elif corruption == "target_identity": + inventory["records"][0]["label"] = 1 + arrays["labels"] = arrays["labels_0"] = np.array([1, 1]) + else: + inventory["records"][0]["label"] = 2 + source.write_text(json.dumps(inventory)) + report["dataset_manifest_sha256"] = file_hash(source) + np.savez(b / "predictions.npz", **arrays) + report_path.write_text(json.dumps(report)) + with pytest.raises(ValueError): + prepare(a, b, tmp_path / "out") + assert not (tmp_path / "out").exists() From 917cd19535c52e4e27e0cbc33efb5e4461349448 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 18:39:47 +0200 Subject: [PATCH 081/155] docs: report isolated three-seed INT8 training trade-offs --- docs/benchmarks.md | 78 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 18 +++++++-- 2 files changed, 93 insertions(+), 3 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index e9a91cf..925c75b 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2326,6 +2326,84 @@ the known EMA expected failure. Five additional focused summary/CLI checks passe after adding the visible backbone-policy column. All four real GPU training and reload runs and both five-metric evaluations completed successfully. +### Isolated three-seed flat-head training comparison + +At revision `393a4fa`, twelve fresh-process Blair runs compared full and +parameter-frozen pretrained EfficientNetV2-S training, each in BF16 and native +INT8, for seeds 42/43/44. All use the flat symmetric normalized head, five epochs, +batch 32, image size 128, MuonAuxAdamW with learning rate 0.01, CPU image caching, +zero loader/cache workers and one PyTorch thread. The parameter-frozen mode +retains normal BatchNorm/dropout training behavior. There were no overlapping +benchmark or test jobs. Seeds 42/44 run float before INT8 and full before frozen; +seed 43 reverses both orders. GPU clocks, thermals and background OS activity were +not controlled, so this is an isolated-job local study, not target certification. + +All twelve runs completed training and checkpoint-reloaded inference. Final +checkpoints record epoch 4, scheduler position 575 and Adam state step 575 in +every run. All six quality pairs passed the maintained adapter's checks, including +matched source/configuration within each pair. The ordered class mappings and +1,161 held-out image identities, labels and recorded hashes also agree across +all seeds and both training modes. Evaluation uses all five mini_metrics metrics, +fixed argmax predictions and threshold zero; no held-out threshold or seed selection. + +Median training phases below use epochs 3–5 and include loading, preprocessing, +compute and batch logging. Whole training calls also include setup, validation, +figures and checkpoints; they exclude final held-out prediction collection. + +| Mode | Seed | Float train epoch (s) | INT8 train epoch (s) | INT8 / float | Float training call (s) | INT8 training call (s) | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Full | 42 | 14.179 | 14.181 | 1.000 | 85.953 | 85.788 | +| Full | 43 | 16.622 | 17.388 | 1.046 | 97.105 | 105.722 | +| Full | 44 | 18.841 | 23.201 | 1.231 | 107.971 | 132.295 | +| Parameter frozen | 42 | 6.074 | 6.339 | 1.043 | 42.132 | 45.838 | +| Parameter frozen | 43 | 6.904 | 6.354 | 0.920 | 48.911 | 44.307 | +| Parameter frozen | 44 | 9.321 | 8.304 | 0.891 | 65.718 | 64.139 | + +Full-training INT8 was tied with or slower than BF16 in every pair. Frozen-mode +timings were mixed. Substantial duration changes between seeds/runs remain even +without overlapping tests; these measurements do not establish a stable speedup. +The median paired train-phase ratios were 1.046 for full training and 0.920 for +parameter-frozen training; whole-call ratios were 1.089 and 0.976 respectively. + +Allocated peaks were identical across the three seeds within each configuration: +full training used 1,230.655 MiB float versus 1,189.362 MiB INT8 (**3.4% lower**); +parameter-frozen training used 206.898 versus 170.032 MiB (**17.8% lower**). +These are the runner's scoped PyTorch allocator peaks, not total process/device +memory. Coverage remains the hidden and final Linear layers; convolution training, +gradients and optimizer states remain floating. + +The table records the observed range of INT8-minus-float metric differences, +multiplied by 100. Ranges span three paired seeds; they are not confidence intervals. +Coverage was 1.0 for every model. + +| Mode | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Full | −0.438 to +1.848 | −0.774 to +2.999 | −0.950 to +1.168 | 0 | −0.298 to +2.386 | +| Parameter frozen | −1.461 to +0.281 | −1.525 to +0.051 | −2.082 to +0.293 | 0 | −1.382 to −0.080 | + +There is no uniform quality improvement: the frozen INT8 models lose Macro-F1 +in two seeds and Theil's U in all three. Absolute Macro-F1 is approximately +0.715–0.748 float / 0.734–0.759 INT8 for full training, versus 0.476–0.480 / +0.465–0.483 for frozen training. The faster frozen regime therefore does not +provide the same quality at this fixed epoch budget and must not be presented +as an equivalent replacement for full training or a time-to-useful-quality gain. + +The within-mode quality losses are small enough to remain candidate trade-offs +under the user's stated tolerance, but full-training memory savings are modest +and a reliable speed benefit is unproven. Next prioritize measured execution costs +and larger-head regimes, complete the corresponding hierarchical study, and +verify the target training hardware. This does not justify making native INT8 a +default for small-head convolutional training. + +Ignored `tmp-isolated-flat-training/` retains the exact command protocol, driver, +logs, final/resume checkpoints, predictions, six portable quality bundles, all +mini_metrics reports, `comparison.json` and `checkpoint-checks.json`. Completed +runs' epoch-zero checkpoints were removed to limit temporary disk use. No library +code changed in this study; validation consists of the twelve real training runs, +six metric evaluations, cross-run identity/configuration checks and checkpoint +step checks described above. Hierarchical, larger-class and target-hardware +conclusions require their own corresponding measurements. + ### Maintained training prediction quality inputs `dev.benchmarks.training_predictions` now connects saved dataset benchmark runs diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 8516a21..158034e 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -38,6 +38,16 @@ a general quality improvement. Local allocated peaks were 17.8% lower, but CPU checks overlapped the runs, so no timing benefit is claimed. This regime retains training-mode BatchNorm/dropout, unlike the synthetic evaluation-mode backbone. +An [isolated three-seed flat-head study](benchmarks.md#isolated-three-seed-flat-head-training-comparison) +now compares full and parameter-frozen pretrained training in twelve fresh +processes. Full-training INT8 was tied with or slower than BF16 and saved 3.4% +allocated peak memory. Parameter-frozen INT8 saved 17.8%, with mixed timings and +a largest observed quality loss of 2.082 points in Macro-Precision. All five +metrics and checkpoint step counts were checked. The absolute frozen-model +quality remained well below full training at five epochs, and timing drift +prevents claiming a stable speedup. Hierarchical and target-hardware replication +and time-to-useful-quality evidence remain outstanding. + The earlier floating-checkpoint TensorRT timing comparison uses matched builder settings and three fresh paired processes per head. At batches 1 and 8, INT8 did not beat FP16 in local host latency. Engines are approximately 43% smaller, while reported execution @@ -114,9 +124,11 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur [corrected allocation comparison](benchmarks.md#correcting-the-frozen-head-allocation-comparison) shows a 6.6% local peak reduction for both heads. Repeat the corrected probe across seeds and with realistic pretrained features and longer integrated training. - The initial parameter-frozen pretrained Blair study now provides execution and - quality evidence; repeat it in isolated processes across seeds alongside matched - full-backbone baselines to establish time to useful quality. + The flat pretrained Blair profile now has isolated three-seed comparisons + against full-backbone baselines. Extend that comparison to hierarchical models + and longer matched-quality budgets; the current five-epoch frozen models do + not reach the full-training quality level. Investigate measured execution costs + and larger-head regimes rather than inferring a general speedup from these runs. 3. **Qualify the trained-checkpoint-to-deployment contract.** Explicit conversion and calibration now connect the matched native QT checkpoints to distinct From aff080722515e2dff2cbabb06cb43692743ff866 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 18:58:00 +0200 Subject: [PATCH 082/155] feat: benchmark compiled optimizers with large classifier heads --- dev/benchmarks/README.md | 15 ++++++++++++++- dev/benchmarks/large_head_training.py | 15 +++++++++++++-- tests/test_benchmark_large_head.py | 26 ++++++++++++++++++++++++-- 3 files changed, 51 insertions(+), 5 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 0ef501c..e25d19a 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1339,7 +1339,7 @@ costs and actual target-device verification remain separate requirements. `dev.benchmarks.large_head_training` makes the earlier large-class probe reusable on the intended training machines. It builds the actual backbone with a symmetric, -normalized flat or hierarchical head, then performs eager MuonAuxAdamW updates +normalized flat or hierarchical head, then performs MuonAuxAdamW updates (eager by default) through the existing builder, scaler and optimizer-step helper. It uses fixed synthetic uint8 images and labels, not a convergence or dataset-quality benchmark. @@ -1365,6 +1365,19 @@ forward computation still runs on every update; embeddings are not cached. Use `--device cpu --dtype float32` only for offline floating diagnostics. Native INT8 training requires an accessible CUDA device. +Add `--compile-optimizer` to use the existing optimizer-compilation path, and +optionally `--optimizer-cudagraphs` to request optimizer CUDA graphs. The model +forward/backward remains eager. Apply the same execution settings to both +precisions; `--warmup 8 --steps 5` provides a bounded initial comparison that +includes the helper's real eager initialization update before compilation. +Graph settings alone do not establish actual replay or a performance benefit. +Reports retain both flags and the composite optimizer's applied step count. +Compare setup peaks, measured allocated peaks and reserved memory separately: +lower steady allocated memory does not establish a lower startup requirement, +and graph pools may increase reserved memory. The probe has no scheduler or +checkpoint/resume workload; those contracts remain covered by the integrated +runner and optimizer tests. + Use new output directories, matching settings, alternating float/INT8 order and paired seeds. A separate CPU generator fixes inputs independently of quantization setup RNG consumption; reports retain their hash. The report includes warmup and diff --git a/dev/benchmarks/large_head_training.py b/dev/benchmarks/large_head_training.py index 9b9887c..7fc0286 100644 --- a/dev/benchmarks/large_head_training.py +++ b/dev/benchmarks/large_head_training.py @@ -17,6 +17,8 @@ from mini_trainer.modeling.quantized_training import prepare_quantized_training from mini_trainer.trainer import _optimizer_step from mini_trainer.training import MuonAuxAdamW +from mini_trainer.training.compilation import compile_optimizer as prepare_compiled_optimizer +from mini_trainer.training.compilation import validate_optimizer_compilation def run( @@ -33,8 +35,11 @@ def run( device="cuda:0", dtype="float16", backbone="efficientnet_v2_s", + compile_optimizer=False, + optimizer_cudagraphs=False, ): device = torch.device(device) + validate_optimizer_compilation(compile_optimizer, optimizer_cudagraphs, device) if min(classes, batch_size, image_size, warmup, steps) < 1 or classes < 2 or batch_size < 2: raise ValueError("Require at least two classes/samples and positive image size, warmup and steps") if device.type not in ("cpu", "cuda") or dtype not in ("float32", "float16", "bfloat16"): @@ -66,12 +71,15 @@ def run( "backbone": backbone, "hidden": "symmetric", "normalized": True, + "compile_optimizer": compile_optimizer, + "optimizer_cudagraphs": optimizer_cudagraphs, }, "environment": {"platform": platform.platform(), "torch": torch.__version__}, "warmup": [], "steps": [], "scope": ( - "Repeated fixed synthetic uint8 images and labels, eager MuonAuxAdamW and AMP. " + "Repeated fixed synthetic uint8 images and labels, MuonAuxAdamW and AMP. " + "Optimizer compilation/graph flags describe requested configuration, not verified graph replay. " "Full-model or frozen-backbone updates; no cached embeddings. " "No convergence, loader, checkpoint, distributed or target-hardware performance claims." ), @@ -123,6 +131,8 @@ def synchronize(): if sum(p.numel() for p in head.parameters() if p.requires_grad) != report["head_trainable_parameters"]: raise RuntimeError("Freezing/preparation changed the head's trainable parameter contract") optimizer = BaseBuilder.build_optimizer(model, MuonAuxAdamW, lr=0.01, weight_decay=0.0) + if compile_optimizer: + prepare_compiled_optimizer(optimizer, cudagraphs=optimizer_cudagraphs) scaler = BaseBuilder.build_scaler(device.type, enabled=device.type == "cuda" and dtype == "float16") report["parameters"] = { "total": sum(p.numel() for p in model.parameters()), @@ -179,6 +189,7 @@ def step(): if frozen and any(p.grad is not None for p in model.parameters() if not p.requires_grad): raise RuntimeError("A frozen parameter received a gradient") report["median_seconds_per_update"] = statistics.median(s["seconds"] for s in report["steps"]) + report["optimizer_steps"] = optimizer._step_count report["status"] = "measured" except Exception as error: report.update(status="failed", error=f"{type(error).__name__}: {error}") @@ -193,7 +204,7 @@ def main(): parser.add_argument("--output", type=Path, required=True) for name, default in (("classes", 10000), ("batch-size", 32), ("image-size", 128), ("seed", 42), ("warmup", 3), ("steps", 5)): parser.add_argument("--" + name, type=int, default=default) - for flag in ("hierarchical", "frozen", "quantized"): + for flag in ("hierarchical", "frozen", "quantized", "compile-optimizer", "optimizer-cudagraphs"): parser.add_argument("--" + flag, action="store_true") parser.add_argument("--device", default="cuda:0") parser.add_argument("--dtype", choices=["float32", "float16", "bfloat16"], default="float16") diff --git a/tests/test_benchmark_large_head.py b/tests/test_benchmark_large_head.py index b5af5ea..2392a91 100644 --- a/tests/test_benchmark_large_head.py +++ b/tests/test_benchmark_large_head.py @@ -16,6 +16,8 @@ def test_large_head_cpu_diagnostic_checks_full_and_frozen_updates(tmp_path, hier for report in (full, frozen): assert report["steps"][0]["updated"] assert report["median_seconds_per_update"] > 0 + assert report["optimizer_steps"] == 2 + assert report["settings"]["compile_optimizer"] is False assert "measured_peak_allocated_bytes" not in report with pytest.raises(FileExistsError): run(tmp_path / "full", **common) @@ -27,8 +29,16 @@ def test_native_quantized_training_is_not_reported_as_cpu_training(tmp_path): assert not (tmp_path / "invalid").exists() +@pytest.mark.parametrize("compiled,device", [(False, "cuda:0"), (True, "cpu")]) +def test_invalid_optimizer_graph_requests_fail_before_output(tmp_path, compiled, device): + with pytest.raises(ValueError, match="requires compile_optimizer|require CUDA"): + run(tmp_path / "invalid", device=device, compile_optimizer=compiled, optimizer_cudagraphs=True) + assert not (tmp_path / "invalid").exists() + + @pytest.mark.parametrize("hierarchical", [False, True]) -def test_cuda_frozen_probe_releases_replaced_float_parameters(tmp_path, monkeypatch, hierarchical): +@pytest.mark.parametrize("compiled", [False, True]) +def test_cuda_frozen_probe_releases_replaced_float_parameters(tmp_path, monkeypatch, hierarchical, compiled): import os import weakref @@ -41,7 +51,15 @@ def test_cuda_frozen_probe_releases_replaced_float_parameters(tmp_path, monkeypa assert torch.cuda.is_available() prepare = probe.prepare_quantized_training build_optimizer = probe.BaseBuilder.build_optimizer + compile_optimizer = probe.prepare_compiled_optimizer replaced = [] + compiled_calls = [] + + def inspect_compilation(optimizer, **kwargs): + result = compile_optimizer(optimizer, **kwargs) + assert all(getattr(optimizer, name)._mini_trainer_compiled for name in optimizer.optimizers) + compiled_calls.append(True) + return result def capture_replacements(model): before = {id(p): weakref.ref(p) for p in model.parameters()} @@ -57,16 +75,20 @@ def verify_released(*args, **kwargs): monkeypatch.setattr(probe, "prepare_quantized_training", capture_replacements) monkeypatch.setattr(probe.BaseBuilder, "build_optimizer", verify_released) + monkeypatch.setattr(probe, "prepare_compiled_optimizer", inspect_compilation) result = run( tmp_path / "quantized", classes=101, hierarchical=hierarchical, batch_size=2, image_size=32, - warmup=1, + warmup=3, steps=1, frozen=True, quantized=True, dtype="bfloat16", + compile_optimizer=compiled, ) assert result["status"] == "measured" + assert result["optimizer_steps"] == 4 + assert compiled_calls == ([True] if compiled else []) From 4c0ef1bf21db350b182f93dd1e010fcef5c75f38 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 19:08:00 +0200 Subject: [PATCH 083/155] docs: report large-head optimizer timing and memory trade-offs --- docs/benchmarks.md | 77 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 10 ++++- 2 files changed, 86 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 925c75b..07a7043 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2404,6 +2404,83 @@ six metric evaluations, cross-run identity/configuration checks and checkpoint step checks described above. Hierarchical, larger-class and target-hardware conclusions require their own corresponding measurements. + +### Isolated 100k-class optimizer compilation comparison + +Revision `aff0807` exposes the existing optimizer compilation and graph options +in the maintained large-head probe. Eighteen fresh processes compare eager, +compiled and compiled-with-graphs MuonAuxAdamW, each with float and native INT8 +weights, across three timing trials. The actual EfficientNetV2-S backbone is +frozen in evaluation mode, with a symmetric normalized flat head, 100,000 classes, +batch 32, 128px inputs and BF16 AMP. Model forward/backward remains eager. +There are 129,642,240 trainable head parameters. This tests repeated synthetic +inputs and labels; it does not measure dataset quality or real-image loading. + +Every run uses seed 42, eight warmup updates and twenty measured updates. All +18 processes exited successfully with finite losses and 28 applied optimizer +steps. Runner/input hashes, base settings and parameter counts match across all +runs. Execution order is eager/compiled/graphs, graphs/compiled/eager, then +compiled/eager/graphs; precision order reverses in trial two. No test or benchmark +jobs overlapped timing. This is three timing repetitions, not three quality seeds. +The local environment is PyTorch 2.12.0+cu130 on the RTX 3080 Ti Laptop GPU under +WSL2. Clocks, thermals and background OS activity were not controlled. + +The table gives each run's median synchronized host time per complete update: + +| Optimizer execution | Precision | Trial 1 (ms) | Trial 2 (ms) | Trial 3 (ms) | +| --- | --- | ---: | ---: | ---: | +| Eager | Float | 59.757 | 61.961 | 63.978 | +| Eager | INT8 | 59.852 | 57.971 | 63.903 | +| Compiled | Float | 50.383 | 52.349 | 55.588 | +| Compiled | INT8 | 48.948 | 52.282 | 56.059 | +| Compiled with graphs | Float | 51.907 | 54.809 | 52.127 | +| Compiled with graphs | INT8 | 55.695 | 62.919 | 55.946 | + +Compilation improves both precisions relative to their eager baseline in every +trial. INT8 versus float is approximately tied or mixed without graphs; graph +INT8 is 7.3–14.8% slower than graph float. These results support testing ordinary +optimizer compilation on the target machines, but do not demonstrate a distinct +INT8 speed advantage or justify changing the default execution policy. + +Allocator measurements were identical across trials within each configuration: + +| Optimizer execution | Precision | Setup peak allocated (GiB) | Measured peak allocated (GiB) | Measured peak reserved (GiB) | +| --- | --- | ---: | ---: | ---: | +| Eager | Float | 2.7999 | 2.7999 | 2.9492 | +| Eager | INT8 | 2.6160 | 2.6160 | 2.9492 | +| Compiled | Float | 2.7999 | 2.7999 | 2.9492 | +| Compiled | INT8 | 2.6160 | 2.2588 | 2.9492 | +| Compiled with graphs | Float | 2.7999 | 2.7989 | 3.4492 | +| Compiled with graphs | INT8 | 2.6160 | 2.1263 | 3.5898 | + +Ordinary compilation lowers INT8's steady allocated peak, but its setup peak is +unchanged: the existing compiler wrapper deliberately initializes optimizer state +with a real eager update. Graphs lower steady allocation further while increasing +reserved memory. Neither steady allocation improvement establishes a reduction +in startup VRAM requirements or total process memory. A separate 100k-class +hierarchical INT8 graph smoke run passed; an optimizer-only profile after eight +warmups observed two `cudaGraphLaunch` events during an applied update. This +confirms graph launches in that diagnostic, not capture of every operation or +hierarchical timing parity. + +Reproduce each configuration with the existing CUDA environment, adding +`--quantized`, `--compile-optimizer`, and then `--optimizer-cudagraphs` as applicable: + +```bash +CUDA_VISIBLE_DEVICES=0 .venv/bin/python -m dev.benchmarks.large_head_training \ + --classes 100000 --dtype bfloat16 --frozen --warmup 8 --steps 20 \ + --seed 42 --output fresh-output +``` + +The exact driver, command protocol, raw samples and reports remain in ignored +`tmp-large-head-optimizer/`; timing reports are under `timings/`, and separate +execution/memory diagnostics and graph profiling are excluded from timing claims. +Harness validation passed static checks, six focused CUDA/configuration checks, +and the full CPU-default suite (503 passed, 160 skipped, one known EMA expected +failure). No kernel or optimizer semantics changed in this milestone. Real +hierarchical training, longer quality-matched budgets, larger target-machine +workloads and sustained memory/throughput verification remain necessary. + ### Maintained training prediction quality inputs `dev.benchmarks.training_predictions` now connects saved dataset benchmark runs diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 158034e..c27780f 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the CPU recipe study](benchmarks.md#cpu-specific-activation-and-bias-calibration). +recorded in [the 100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -48,6 +48,14 @@ quality remained well below full training at five epochs, and timing drift prevents claiming a stable speedup. Hierarchical and target-hardware replication and time-to-useful-quality evidence remain outstanding. +The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) +now separates eager, compiled and graph execution in eighteen isolated processes. +Ordinary optimizer compilation improves both precisions' local update times and +reduces INT8's steady allocated peak from 2.616 to 2.259 GiB. Setup peak remains +2.616 GiB; graphs reduce steady allocation further but increase reserved memory +and leave INT8 slower than graph float in all three trials. This is a synthetic +capacity/execution result, not dataset quality or target-hardware acceptance. + The earlier floating-checkpoint TensorRT timing comparison uses matched builder settings and three fresh paired processes per head. At batches 1 and 8, INT8 did not beat FP16 in local host latency. Engines are approximately 43% smaller, while reported execution From aaa90a15d7ab3660c3a583ec16f98fa4ce925008 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 19:27:02 +0200 Subject: [PATCH 084/155] docs: report three-seed hierarchical INT8 training comparisons --- docs/benchmarks.md | 76 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 22 ++++++++--- 2 files changed, 92 insertions(+), 6 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 07a7043..f503905 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2405,6 +2405,82 @@ step checks described above. Hierarchical, larger-class and target-hardware conclusions require their own corresponding measurements. +### Isolated three-seed hierarchical training comparison + +At revision `4c0ef1b`, twelve fresh-process runs extend the preceding flat study +to the normalized symmetric `HierarchicalClassifier`. The pretrained +EfficientNetV2-S backbone, reviewed Blair splits (3,704/912/1,161 images), seeds +42/43/44, five epochs, batch 32, 128px inputs, BF16 AMP, MuonAuxAdamW at learning +rate 0.01, CPU caching, zero workers and one PyTorch thread match that study. +The hierarchical model uses its two-level training loss and 25 leaf/15 parent +classes. Full and parameter-frozen training are separate; frozen parameters retain +training-mode BatchNorm/dropout. Model and optimizer compilation are disabled. +Seed 43 reverses training-mode and precision order; all processes run sequentially +without competing benchmarks/tests. GPU clocks, thermals and background OS +activity remain uncontrolled on the local RTX 3080 Ti Laptop GPU. + +All twelve runs completed training, checkpoint reload and finite held-out +inference. Source hashes match across runs. All six paired evaluations passed +configuration and dataset checks, and class mappings, sample identities, labels +and recorded image hashes match across seeds/modes. Every final resume checkpoint +records epoch 4, scheduler position 575 and Adam step 575. The maintained adapter +and mini_metrics evaluator use fixed argmax predictions at threshold zero, with +all five metrics evaluated independently at both levels; no held-out selection. + +| Mode | Seed | Float train epoch (s) | INT8 train epoch (s) | INT8 / float | Float training call (s) | INT8 training call (s) | +| --- | ---: | ---: | ---: | ---: | ---: | ---: | +| Full | 42 | 12.393 | 15.555 | 1.255 | 72.289 | 88.361 | +| Full | 43 | 13.809 | 13.932 | 1.009 | 84.263 | 86.504 | +| Full | 44 | 14.200 | 14.217 | 1.001 | 89.357 | 86.489 | +| Parameter frozen | 42 | 5.856 | 6.272 | 1.071 | 41.789 | 44.750 | +| Parameter frozen | 43 | 6.916 | 6.099 | 0.882 | 46.162 | 44.124 | +| Parameter frozen | 44 | 7.012 | 6.238 | 0.890 | 44.907 | 45.750 | + +Train times are medians of epochs 3–5, including loading, preprocessing, compute +and batch logging. Whole training calls include setup, validation, figures and +checkpoint writing, but exclude final held-out inference. Full INT8 train phases +were tied with or slower than float, while frozen timings were mixed. Median +paired train/call ratios were 1.009/1.027 full and 0.890/1.019 frozen: faster +frozen train phases in two seeds did not consistently shorten the whole call. + +Scoped allocated peaks match across all seeds: full training used 1,230.659 MiB +float versus 1,189.365 MiB INT8 (**3.4% lower**), and frozen training used 206.902 +versus 170.035 MiB (**17.8% lower**). These are allocator measurements, not total +process memory. Convolutions, gradients and optimizer state remain floating. + +The table gives observed INT8-minus-float differences multiplied by 100, spanning +three paired seeds. These ranges are not confidence intervals. Coverage is 1.0 +throughout. + +| Mode / level | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Full / leaf | −3.450 to +4.970 | −3.622 to +5.164 | −3.979 to +5.054 | 0 | −1.936 to +2.599 | +| Full / parent | −5.099 to −0.638 | −5.454 to +0.463 | −5.378 to −2.794 | 0 | −1.660 to +0.794 | +| Parameter frozen / leaf | −2.215 to +0.100 | −3.081 to +0.630 | −1.215 to +1.496 | 0 | −1.911 to +0.999 | +| Parameter frozen / parent | −2.770 to +0.427 | −2.580 to 0.000 | −3.073 to +1.550 | 0 | −1.301 to +0.178 | + +Full-training parent Macro-F1 and Macro-Precision fall in every seed, with losses +up to about five points. This is a less favorable result than the flat study and +does not justify recommending the current five-epoch hierarchical recipe for its +3.4% memory saving. Leaf changes are mixed and cannot establish general quality +superiority. Longer matched-quality training and investigation of the hierarchy's +optimization sensitivity are warranted before production recommendations. + +Absolute leaf Macro-F1 is 0.692–0.725 float / 0.672–0.742 INT8 for full training, +versus 0.480–0.489 / 0.467–0.481 frozen. Parent Macro-F1 is 0.838–0.860 / +0.809–0.841 full, versus 0.624–0.639 / 0.597–0.643 frozen. As in the flat study, +frozen training at this budget is not an equal-quality replacement for full +training. These runs neither establish target-hardware gains nor compare trained +ONNX deployment candidates. + +Ignored `tmp-isolated-hierarchical-training/` retains the protocol, drivers, logs, +final/resume checkpoints, predictions, six quality bundles and metric reports, +`comparison.json` and `checkpoint-checks.json`. Completed epoch-zero checkpoints +were removed, retaining approximately 4 GiB of evidence. Validation consists of +the twelve real training runs, six two-level metric evaluations and the identity, +configuration and checkpoint checks above. No library code changed, so the +previous full-suite result remains separate from this empirical study. + ### Isolated 100k-class optimizer compilation comparison Revision `aff0807` exposes the existing optimizer compilation and graph options diff --git a/docs/quantization-status.md b/docs/quantization-status.md index c27780f..671781d 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the 100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison). +recorded in [the hierarchical training study](benchmarks.md#isolated-three-seed-hierarchical-training-comparison). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -45,8 +45,18 @@ allocated peak memory. Parameter-frozen INT8 saved 17.8%, with mixed timings and a largest observed quality loss of 2.082 points in Macro-Precision. All five metrics and checkpoint step counts were checked. The absolute frozen-model quality remained well below full training at five epochs, and timing drift -prevents claiming a stable speedup. Hierarchical and target-hardware replication -and time-to-useful-quality evidence remain outstanding. +prevents claiming a stable speedup. The hierarchical replication below is now +complete; target-hardware replication and time-to-useful-quality evidence remain +outstanding. + +The [hierarchical three-seed study](benchmarks.md#isolated-three-seed-hierarchical-training-comparison) +completes twelve corresponding training/reload runs and six two-level metric +comparisons. Memory savings match the flat study, with mixed timings. Full-training +parent Macro-F1 and Macro-Precision fall in every seed, by up to 5.10 and 5.38 +points respectively; leaf changes are mixed. This negative result argues against +recommending the five-epoch recipe for its 3.4% full-training memory saving. +Longer matched-quality budgets and investigation of hierarchical optimization +sensitivity remain necessary. All final checkpoint step checks passed. The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) now separates eager, compiled and graph execution in eighteen isolated processes. @@ -132,9 +142,9 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur [corrected allocation comparison](benchmarks.md#correcting-the-frozen-head-allocation-comparison) shows a 6.6% local peak reduction for both heads. Repeat the corrected probe across seeds and with realistic pretrained features and longer integrated training. - The flat pretrained Blair profile now has isolated three-seed comparisons - against full-backbone baselines. Extend that comparison to hierarchical models - and longer matched-quality budgets; the current five-epoch frozen models do + Both flat and hierarchical pretrained Blair profiles now have isolated + three-seed comparisons against full-backbone baselines. Extend them to longer + matched-quality budgets; the current five-epoch frozen models do not reach the full-training quality level. Investigate measured execution costs and larger-head regimes rather than inferring a general speedup from these runs. From b6a05b06df404c0aa8178a921b67d247c409c38a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 19:40:04 +0200 Subject: [PATCH 085/155] fix: retain epoch statistics in training benchmarks --- dev/benchmarks/README.md | 18 ++++++++++++++++++ dev/benchmarks/run.py | 2 +- docs/quantization-status.md | 18 ++++++++++++++++++ tests/test_benchmark_datasets.py | 5 +++++ tests/test_benchmark_synthetic.py | 10 ++++++++++ 5 files changed, 52 insertions(+), 1 deletion(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index e25d19a..052db4f 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1441,3 +1441,21 @@ This is a saved-prediction quality comparison. It does not establish runtime performance, checkpoint-to-score parity, source-image integrity beyond the recorded hashes, or production acceptance. Preserve the evaluator's report and CSV/manifest bundle alongside the original training reports for continuous runs. + +### Epoch statistics for convergence comparisons + +The dataset runner retains the existing `MetricLogger` so +`training/logs/summary.csv` records actual train/validation statistics for every +epoch, including per-level hierarchical losses and accuracies. These are the +logger's unweighted means of batch statistics, not the held-out mini_metrics +Macro-F1/Recall/Precision/Coverage/Theil's U results. Use the paired prediction +adapter and quality evaluator for those metrics. + +Earlier runs through `aaa90a1` passed `logger_cls=[]`: their CSV statistics are +zero placeholders and cannot support convergence claims, and their `best.pt` +selection saw a constant validation statistic. The dataset runner explicitly +reloads `last.pt` for held-out predictions, so those independently calculated +quality results and the recorded timing/memory measurements remain valid within +their stated scope. Do not treat the historical `best.pt` files as validated +best-epoch choices. Restoring statistics changes logging overhead; rerun both +precisions together before comparing new timing results with one another. diff --git a/dev/benchmarks/run.py b/dev/benchmarks/run.py index 1de4c2b..ea09436 100644 --- a/dev/benchmarks/run.py +++ b/dev/benchmarks/run.py @@ -209,7 +209,7 @@ def run( criterion_builder_kwargs={"label_smoothing": 0.0}, regularizer_builder_kwargs={"strength": 0.0}, lr_schedule_builder_kwargs={"warmup_epochs": 0.0}, - logger_builder_kwargs={"logger_cls": [], "measurement_device": device}, + logger_builder_kwargs={"measurement_device": device}, ) if target_device.type == "cuda": torch.cuda.synchronize(target_device) diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 671781d..d03ec59 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -58,6 +58,24 @@ recommending the five-epoch recipe for its 3.4% full-training memory saving. Longer matched-quality budgets and investigation of hierarchical optimization sensitivity remain necessary. All final checkpoint step checks passed. +A subsequent harness audit found that all statistic loggers were disabled. The +historical epoch-summary CSVs therefore contain zeros and cannot reveal training +curves; historical `best.pt` selection also used a constant validation statistic. +The five-metric results above use independent predictions from `last.pt`, so this +does not explain their quality differences. The runner now retains its default +metric logger to support real convergence diagnostics. New timing comparisons +must include this overhead in both baselines; see the +[epoch-statistics contract](../dev/benchmarks/README.md#epoch-statistics-for-convergence-comparisons). +The correction passed static checks and the full CPU-default suite: 503 passed, +160 skipped and one known EMA expected failure. Focused tests verify all epoch +rows, positive finite losses, oracle accuracy and both hierarchical loss levels. +A one-epoch pretrained hierarchical BF16/native INT8 fine-tuning run also completed +on CUDA, reloaded its checkpoint and recorded actual statistics at both levels. +Its artifacts are in ignored `tmp-epoch-logger-int8/`; CPU checks overlapped, so +its timings are excluded from performance claims. This restores the prerequisite +for a longer convergence study; it does not supply the missing historical curves. + + The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) now separates eager, compiled and graph execution in eighteen isolated processes. Ordinary optimizer compilation improves both precisions' local update times and diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index b329c7b..f38466e 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -1,3 +1,4 @@ +import csv import json import shutil from pathlib import Path @@ -257,6 +258,10 @@ def inspect_optimizer(model, *args, **kwargs): captured.clear() finally: torch.set_num_threads(previous_threads) + with (tmp_path / "hierarchical/training/logs/summary.csv").open() as stream: + summaries = list(csv.DictReader(stream)) + assert [row["type"] for row in summaries] == ["train", "eval"] + assert all(float(row[level_loss]) > 0 for row in summaries for level_loss in ("loss/lvl0", "loss/lvl1")) flat, hierarchical = reports assert flat["dataset_manifest_sha256"] == hierarchical["dataset_manifest_sha256"] assert flat["class_mapping"] == spec["cls2idx"]["0"] diff --git a/tests/test_benchmark_synthetic.py b/tests/test_benchmark_synthetic.py index 0b636fc..8db38fe 100644 --- a/tests/test_benchmark_synthetic.py +++ b/tests/test_benchmark_synthetic.py @@ -1,3 +1,5 @@ +import csv + import numpy as np from PIL import Image @@ -37,6 +39,14 @@ def test_synthetic_training_matches_oracle_and_repeats(tmp_path): assert all(phase["seconds"] >= 0 and phase["peak_cuda_allocated_bytes"] is None for phase in first["phase_measurements"]) assert first["peak_cuda_allocated_bytes"] is None assert first["test_accuracy"] == second["test_accuracy"] == 1.0 + for name in ("first", "second"): + with (tmp_path / name / "training/logs/summary.csv").open() as stream: + summaries = list(csv.DictReader(stream)) + assert [(int(row["epoch"]), row["type"]) for row in summaries] == [ + (epoch, phase) for epoch in range(12) for phase in ("train", "eval") + ] + assert all(np.isfinite(float(row["loss"])) and float(row["loss"]) > 0 for row in summaries) + assert float(summaries[-1]["acc1"]) == 100.0 assert first["dataset_manifest_sha256"] == second["dataset_manifest_sha256"] with np.load(tmp_path / "first/predictions.npz") as a, np.load(tmp_path / "second/predictions.npz") as b: np.testing.assert_array_equal(a["scores"], b["scores"]) From 5010b22309a6c2eee3962af61954a34a37cec2d6 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 19:52:53 +0200 Subject: [PATCH 086/155] feat: compose representative paired training and quality benchmarks --- dev/benchmarks/README.md | 57 ++++++++++++++++ dev/benchmarks/summarize.py | 30 ++++++++- dev/check-benchmarks.sh | 83 +++++++++++++++++++++++- docs/quantization-status.md | 11 ++++ tests/test_benchmark_datasets.py | 108 +++++++++++++++++++++++++++++++ 5 files changed, 287 insertions(+), 2 deletions(-) diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 052db4f..b343a6d 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1459,3 +1459,60 @@ quality results and the recorded timing/memory measurements remain valid within their stated scope. Do not treat the historical `best.pt` files as validated best-epoch choices. Restoring statistics changes logging overhead; rerun both precisions together before comparing new timing results with one another. + +### Representative paired EfficientNetV2 training profile + +`qt-efficientnet` composes the existing dataset runner, training-prediction adapter +and five-metric evaluator. By default it trains pretrained EfficientNetV2-S with +symmetric normalized flat and hierarchical heads, full and parameter-frozen +backbones, seeds 42/43/44, BF16 and native INT8: 24 sequential training processes +and twelve held-out comparisons. It uses the reviewed Blair taxonomy, five epochs, +batch 32, 128px images, MuonAuxAdamW at learning rate 0.01, CPU caching and zero +workers. Model/optimizer compilation is disabled. Odd seeds reverse precision and +full/frozen ordering. Metric evaluation starts after every training process has +finished, preserving isolation from that evaluation workload. + +Use explicitly prepared training and mini_metrics environments; the command +installs nothing and does not assume a sibling checkout. The metrics interpreter +is checked before training begins. Run from an otherwise idle allocation and +record its actual hardware/runtime in the retained reports: + +```bash +CUDA_VISIBLE_DEVICES=0 PYTHONHASHSEED=0 \ +BENCHMARK_DATA_ROOT=/path/to/examples \ +BLAIR_CLASS_SPEC=/path/to/reviewed/class_spec.json \ +BENCHMARK_METRICS_PYTHON=/path/to/metrics-env/bin/python \ +bash dev/check-benchmarks.sh qt-efficientnet fresh-results +``` + +The following environment variables select a bounded diagnostic or longer budget: + +| Variable | Default | Accepted values | +| --- | --- | --- | +| `BENCHMARK_HEAD` | `both` | `both`, `flat`, `hierarchical` | +| `BENCHMARK_TRAINING_MODE` | `both` | `both`, `full`, `frozen` | +| `BENCHMARK_SEEDS` | `42 43 44` | Space-separated unique nonnegative integers | +| `BENCHMARK_EPOCHS` | `5` | Positive integer; each model starts fresh with this schedule budget | +| `BENCHMARK_PYTHON` | `.venv/bin/python` | Existing training interpreter | + +For example, set `BENCHMARK_HEAD=hierarchical`, `BENCHMARK_TRAINING_MODE=full`, +`BENCHMARK_SEEDS=42` and `BENCHMARK_EPOCHS=20` for a fixed-seed longer-budget pair. +This is a diagnostic, not a replacement for repeated-seed qualification. Choose +the budget before examining held-out results; the command always evaluates final +`last.pt` checkpoints and does not select a seed or threshold on the test split. + +The results directory must be new. It retains each training report, epoch CSV, +checkpoints, predictions, pair input manifests, mini_metrics reports and logs. +`summary.md` contains both training measurements and per-level quality differences +for Macro-F1, Macro-Recall, Macro-Precision, Coverage and Theil's U. Differences +are multiplied by 100, including Theil's U; undefined metrics remain explicit. +Training or pairing/evaluation failures make the command return nonzero and are +retained in reports and the summary. Other pairs still run so failures cannot +silently remove difficult configurations from the evidence. + +The default matrix is substantially larger than the TinyConv `qt-real` profile. +Size the allocation and artifact storage accordingly. This command is suitable +for a configured target runner but is not yet wired into the scheduled GPU job; +durable result hosting and joint quality/resource acceptance gates remain open. +Local CUDA success does not establish HPC throughput, desktop ONNX performance +or ARM inference support. diff --git a/dev/benchmarks/summarize.py b/dev/benchmarks/summarize.py index fb96cf9..c1f31bb 100644 --- a/dev/benchmarks/summarize.py +++ b/dev/benchmarks/summarize.py @@ -15,8 +15,14 @@ def summarize(directory: Path) -> str: "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", ] reports = sorted(directory.rglob("report.json")) + quality_reports = [] + training_reports = 0 for path in reports: report = json.loads(path.read_text()) + if "models" in report or report.get("benchmark_kind") == "paired_quality": + quality_reports.append((path, report)) + continue + training_reports += 1 accuracy = ", ".join(f"{value:.2%}" for value in report.get("level_accuracies", [])) or "—" seconds = report.get("training_wall_seconds") duration = f"{seconds:.2f}s" if seconds is not None else "—" @@ -49,8 +55,30 @@ def summarize(directory: Path) -> str: f"| {name} | {report['status']} | {device} | {quantization} | {accuracy} | {parameter_bytes} | " f"{peak_memory} | {later_duration} | {duration} | {'frozen' if report.get('fine_tune') else 'trainable'} |" ) - if not reports: + if not training_reports: lines.append("| No reports produced | incomplete | — | — | — | — | — | — | — | — |") + if quality_reports: + lines.extend( + [ + "", + "## Paired held-out quality", + "", + "Differences are candidate minus baseline, multiplied by 100; they are not acceptance gates.", + "", + "| Pair / level | Status | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U |", + "| --- | --- | ---: | ---: | ---: | ---: | ---: |", + ] + ) + for path, report in quality_reports: + name = path.parent.relative_to(directory).as_posix() + if report["status"] != "evaluated": + lines.append(f"| {name} | {report['status']} | — | — | — | — | — |") + continue + for level in report["levels"]: + values = [level["candidate_minus_baseline"].get(metric) for metric in ("f1", "recall", "precision", "coverage", "theilU")] + formatted = " | ".join("undefined" if value is None else f"{100 * value:+.3f}" for value in values) + label = str(level["name"]).replace("|", "\\|").replace("\n", " ").replace("\r", " ") + lines.append(f"| {name} / {label} | evaluated | {formatted} |") lines.extend( [ "", diff --git a/dev/check-benchmarks.sh b/dev/check-benchmarks.sh index 35b2c11..5eafb7a 100644 --- a/dev/check-benchmarks.sh +++ b/dev/check-benchmarks.sh @@ -9,7 +9,33 @@ if [[ -e "$results" ]]; then echo 'Results directory must be new.' >&2 exit 2 fi -case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch|qt-cudagraphs|qt-optimizer-cudagraphs) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense, qt-large-batch, qt-cudagraphs or qt-optimizer-cudagraphs.' >&2; exit 2 ;; esac +case "$mode" in cpu|gpu|real|qt|qt-real|qt-dense|qt-large-batch|qt-cudagraphs|qt-optimizer-cudagraphs|qt-efficientnet) ;; *) echo 'Mode must be cpu, gpu, real, qt, qt-real, qt-dense, qt-large-batch, qt-cudagraphs, qt-optimizer-cudagraphs or qt-efficientnet.' >&2; exit 2 ;; esac +if [[ "$mode" == qt-efficientnet ]]; then + : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing blair/}" + : "${BLAIR_CLASS_SPEC:?Set BLAIR_CLASS_SPEC to a reviewed Blair class specification}" + : "${BENCHMARK_METRICS_PYTHON:?Set BENCHMARK_METRICS_PYTHON to an environment with mini_metrics installed}" + epochs="${BENCHMARK_EPOCHS:-5}" + [[ "$epochs" =~ ^[1-9][0-9]*$ ]] || { echo 'BENCHMARK_EPOCHS must be a positive integer.' >&2; exit 2; } + read -r -a seeds <<< "${BENCHMARK_SEEDS:-42 43 44}" + [[ "${#seeds[@]}" -gt 0 ]] || { echo 'BENCHMARK_SEEDS must not be empty.' >&2; exit 2; } + declare -A seen_seeds=() + for seed in "${seeds[@]}"; do + [[ "$seed" =~ ^(0|[1-9][0-9]*)$ ]] || { echo 'BENCHMARK_SEEDS must contain nonnegative integers.' >&2; exit 2; } + [[ -z "${seen_seeds[$seed]:-}" ]] || { echo 'BENCHMARK_SEEDS must be unique.' >&2; exit 2; } + seen_seeds[$seed]=1 + done + case "${BENCHMARK_HEAD:-both}" in + both) heads=(flat hierarchical) ;; + flat|hierarchical) heads=("$BENCHMARK_HEAD") ;; + *) echo 'BENCHMARK_HEAD must be both, flat or hierarchical.' >&2; exit 2 ;; + esac + case "${BENCHMARK_TRAINING_MODE:-both}" in + both) training_modes=(full frozen) ;; + full|frozen) training_modes=("$BENCHMARK_TRAINING_MODE") ;; + *) echo 'BENCHMARK_TRAINING_MODE must be both, full or frozen.' >&2; exit 2 ;; + esac + "$BENCHMARK_METRICS_PYTHON" -c 'import mini_metrics' # Fail before training if the optional environment is unavailable. +fi mkdir -p -- "$results" status=0 run_profile() { @@ -80,6 +106,61 @@ elif [[ "$mode" == qt-large-batch || "$mode" == qt-cudagraphs || "$mode" == qt-o --allow-nondeterministic "${quantization[@]}" done done +elif [[ "$mode" == qt-efficientnet ]]; then + for seed in "${seeds[@]}"; do + precisions=(float int8) + modes=("${training_modes[@]}") + if [[ "$seed" =~ [13579]$ ]]; then + precisions=(int8 float) + if [[ "${#modes[@]}" == 2 ]]; then modes=(frozen full); fi + fi + for head in "${heads[@]}"; do + for training_mode in "${modes[@]}"; do + extra=() + if [[ "$training_mode" == frozen ]]; then extra=(--fine-tune); fi + for precision in "${precisions[@]}"; do + quantization=() + if [[ "$precision" == int8 ]]; then quantization=(--quantized-training); fi + run_profile "blair-$head-$training_mode-$precision-seed$seed" \ + --dataset blair --data-root "$BENCHMARK_DATA_ROOT/blair" --class-spec "$BLAIR_CLASS_SPEC" \ + --backbone efficientnet_v2_s --head "$head" --hidden symmetric --normalized --pretrained \ + --seed "$seed" --epochs "$epochs" --batch-size 32 --image-size 128 --optimizer muon --learning-rate 0.01 \ + --device cuda:0 --dtype bfloat16 --cache CPU --cache-workers 0 --num-workers 0 --threads 1 \ + "${extra[@]}" "${quantization[@]}" + done + done + done + done + # Evaluate only after all training timings, in the explicitly selected metrics environment. + for seed in "${seeds[@]}"; do + for head in "${heads[@]}"; do + for training_mode in "${training_modes[@]}"; do + pair="blair-$head-$training_mode-seed$seed" + if "$benchmark_python" -m dev.benchmarks.training_predictions \ + --baseline "$results/blair-$head-$training_mode-float-seed$seed" \ + --candidate "$results/blair-$head-$training_mode-int8-seed$seed" \ + --output "$results/$pair-inputs" > "$results/$pair-quality.log" 2>&1 && \ + PYTHONHASHSEED=0 OMP_NUM_THREADS=1 "$BENCHMARK_METRICS_PYTHON" -m dev.benchmarks.quality_compare \ + --manifest "$results/$pair-inputs/manifest.json" --output "$results/$pair-quality" \ + >> "$results/$pair-quality.log" 2>&1; then + : + else + echo "Quality evaluation failed: $pair; see $results/$pair-quality.log" >&2 + status=1 + "$benchmark_python" - "$results/$pair-quality/report.json" <<'PY_QUALITY_FAILURE' +import json +import sys +from pathlib import Path +path = Path(sys.argv[1]) +if not path.exists(): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps({"schema_version": 1, "benchmark_kind": "paired_quality", "status": "failed", + "error": "Prediction preparation or evaluation failed; see the pair quality log."}) + "\n") +PY_QUALITY_FAILURE + fi + done + done + done elif [[ "$mode" == qt-real ]]; then : "${BENCHMARK_DATA_ROOT:?Set BENCHMARK_DATA_ROOT to the directory containing mnist/ and blair/}" : "${BLAIR_CLASS_SPEC:?Set BLAIR_CLASS_SPEC to a reviewed Blair class specification}" diff --git a/docs/quantization-status.md b/docs/quantization-status.md index d03ec59..a749614 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -146,6 +146,17 @@ checks and **470 tests**, with **152 skips** and **one known EMA expected failur five-metric evaluator without checkpoint or image transfers. It reproduced the pretrained fine-tuning metrics exactly; continuous orchestration and durable result publication remain to be completed. + The [representative training profile](../dev/benchmarks/README.md#representative-paired-efficientnetv2-training-profile) + now composes paired EfficientNetV2 training, prediction preparation, all five + metrics and visible summaries in one shared command, with selectable heads, + full/frozen modes, seeds and epoch budgets. Scheduled runner wiring, durable + hosting and deployment orchestration remain unfinished. + Validation passed static checks and 508 CPU-default tests, with 160 skips and + the known EMA expected failure. A bounded one-epoch hierarchical frozen + float/INT8 Blair pair completed through the shared command, including both + levels of mini_metrics and the combined summary. Reports remain in ignored + `tmp-representative-profile-smoke/`. CPU checks overlapped this smoke run, so + it supplies integration evidence only, not new performance or acceptance claims. 2. **Find and verify the beneficial workload regimes.** Pair INT8 with practical FP16/BF16 baselines while varying batch size, resolution and head size in a diff --git a/tests/test_benchmark_datasets.py b/tests/test_benchmark_datasets.py index f38466e..0757b17 100644 --- a/tests/test_benchmark_datasets.py +++ b/tests/test_benchmark_datasets.py @@ -278,3 +278,111 @@ def inspect_optimizer(model, *args, **kwargs): assert indices[0]["path"] == indices[1]["path"] assert indices[0]["split"] == indices[1]["split"] assert indices[0]["class"] == [labels[0] for labels in indices[1]["class"]] + + +def test_representative_profile_retains_training_and_quality_failures(tmp_path): + import os + import shlex + import subprocess + import sys + + runner = tmp_path / "python-wrapper" + runner.write_text( + "#!/usr/bin/env bash\n" + 'if [[ "$1" == "-c" ]]; then exit 0; fi\n' + 'if [[ "$1" == "-m" && "$2" == "dev.benchmarks.run" ]]; then exit 134; fi\n' + f'exec {shlex.quote(sys.executable)} "$@"\n' + ) + runner.chmod(0o755) + output = tmp_path / "reports" + result = subprocess.run( + ["bash", "dev/check-benchmarks.sh", "qt-efficientnet", str(output)], + env={ + **os.environ, + "BENCHMARK_PYTHON": str(runner), + "BENCHMARK_METRICS_PYTHON": str(runner), + "BENCHMARK_DATA_ROOT": str(tmp_path), + "BLAIR_CLASS_SPEC": str(tmp_path / "spec.json"), + "BENCHMARK_HEAD": "hierarchical", + "BENCHMARK_TRAINING_MODE": "frozen", + "BENCHMARK_SEEDS": "43", + "BENCHMARK_EPOCHS": "20", + }, + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 1 + for precision in ("float", "int8"): + report = json.loads((output / f"blair-hierarchical-frozen-{precision}-seed43/report.json").read_text()) + assert report["status"] == "failed" and report["error"]["exit_code"] == 134 + args = report["arguments"] + for option, value in ( + ("--backbone", "efficientnet_v2_s"), + ("--head", "hierarchical"), + ("--hidden", "symmetric"), + ("--epochs", "20"), + ("--seed", "43"), + ("--dtype", "bfloat16"), + ): + assert args[args.index(option) + 1] == value + assert "--fine-tune" in args and "--pretrained" in args and "--normalized" in args + assert ("--quantized-training" in args) == (precision == "int8") + quality = json.loads((output / "blair-hierarchical-frozen-seed43-quality/report.json").read_text()) + assert quality["status"] == "failed" + summary = (output / "summary.md").read_text() + assert "Paired held-out quality" in summary + assert "blair-hierarchical-frozen-seed43-quality | failed" in summary + + +def test_summary_renders_paired_metrics_without_treating_them_as_training(tmp_path): + output = tmp_path / "pair" + output.mkdir() + (output / "report.json").write_text( + json.dumps( + { + "status": "evaluated", + "models": {}, + "levels": [ + { + "name": "parent", + "candidate_minus_baseline": { + "f1": -0.051, + "recall": 0.02, + "precision": None, + "coverage": 0, + "theilU": -0.01, + }, + } + ], + } + ) + ) + summary = summarize(tmp_path) + assert "| pair / parent | evaluated | -5.100 | +2.000 | undefined | +0.000 | -1.000 |" in summary + assert "| pair | evaluated | ? / ?" not in summary + assert "not acceptance gates" in summary + + +@pytest.mark.parametrize("seeds", [" ", "42 42", "-1"]) +def test_representative_profile_rejects_invalid_seeds_before_output(tmp_path, seeds): + import os + import subprocess + import sys + + output = tmp_path / "reports" + result = subprocess.run( + ["bash", "dev/check-benchmarks.sh", "qt-efficientnet", str(output)], + env={ + **os.environ, + "BENCHMARK_PYTHON": sys.executable, + "BENCHMARK_METRICS_PYTHON": sys.executable, + "BENCHMARK_DATA_ROOT": str(tmp_path), + "BLAIR_CLASS_SPEC": str(tmp_path / "spec.json"), + "BENCHMARK_SEEDS": seeds, + }, + capture_output=True, + text=True, + timeout=30, + ) + assert result.returncode == 2 and "BENCHMARK_SEEDS" in result.stderr + assert not output.exists() From eb4c8ddb80ceee434eae99a719fd1e79a72b755a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 20:09:04 +0200 Subject: [PATCH 087/155] docs: report longer hierarchical INT8 convergence diagnostic --- docs/benchmarks.md | 65 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 12 ++++++- 2 files changed, 76 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index f503905..403c245 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2481,6 +2481,71 @@ the twelve real training runs, six two-level metric evaluations and the identity configuration and checkpoint checks above. No library code changed, so the previous full-suite result remains separate from this empirical study. +### Twenty-epoch hierarchical convergence diagnostic + +Revision `5010b22` runs a fresh 20-epoch full-training pair at seed 42 through +`qt-efficientnet`, using the restored epoch logger. Both models use pretrained +EfficientNetV2-S, the normalized symmetric hierarchical head, BF16 AMP, the same +reviewed Blair splits, batch 32, 128px images and MuonAuxAdamW at learning rate +0.01. Compilation is disabled. Float runs before native INT8; no benchmark/test +jobs overlap. GPU clocks, thermals and background OS activity remain uncontrolled +on the local RTX 3080 Ti Laptop GPU. This is one diagnostic seed, not a new +three-seed qualification or evidence about the intended deployment hardware. + +Both runs completed 20 epochs, checkpoint-reloaded inference and two-level +mini_metrics evaluation. Final resume checkpoints record epoch 19, scheduler +position 2,300 and Adam step 2,300. Source and dataset hashes match within the +pair; the held-out identities, labels, ordered class mappings and image hashes +also match the earlier five-epoch seed-42 evaluation. Every epoch now has real +train/validation statistics. Final `last.pt` checkpoints are evaluated without +held-out seed, threshold or epoch selection. + +| Measurement | Float | INT8 | +| --- | ---: | ---: | +| Whole training call (s) | 320.502 | 349.918 | +| Median train phase, epochs 3–20 (s) | 13.919 | 14.836 | +| Scoped peak allocated memory (MiB) | 1,230.659 | 1,189.365 | +| Held-out leaf Macro-F1 | 0.75854 | 0.76184 | +| Held-out parent Macro-F1 | 0.86738 | 0.86656 | + +INT8 takes 9.2% longer for the training call and 6.6% longer in median later train +phases, with 3.4% lower peak allocation. A single ordered pair does not establish +a stable timing ratio. These scoped allocator peaks are not total process/device +memory, and this result does not demonstrate a time-to-useful-quality advantage. + +All five held-out metric differences below are INT8 minus float, multiplied by +100. Coverage is 1.0 throughout; none of the metrics is undefined. + +| Level | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Leaf | +0.331 | −0.279 | +1.612 | 0 | −0.252 | +| Parent | −0.082 | −0.992 | +1.559 | 0 | +0.323 | + +The parent Macro-F1 gap is small at this budget in seed 42. This does not erase +the earlier five-epoch, three-seed regressions or establish quality superiority. +The new run has a fresh 20-epoch cosine schedule as well as more updates, and the +restored logger adds overhead. It is not an isolated extra-epochs intervention +or a directly comparable timing baseline for the older unlogged runs. + +The curves expose a separate concern: INT8 training loss falls steadily, but its +validation loss spikes in the first half of training. At epochs 5 and 10 it is +6.405 and 4.807, versus float 1.571 and 1.378. By epochs 15 and 20 it is 0.856 +and 0.824, versus float 0.944 and 0.934. Final batch-averaged validation leaf/parent +accuracy is 86.53%/94.18% INT8 versus 84.91%/92.35% float. These validation batch +means are distinct from held-out macro metrics. The spikes warrant replaying +intermediate checkpoints and inspecting train/eval state before assuming that +simply extending training is a reliable production solution. Their cause has not +been established by this experiment. + +The exact environment/command record is `tmp-hierarchical-long-budget/protocol.json`. +Use the documented representative profile with `BENCHMARK_HEAD=hierarchical`, +`BENCHMARK_TRAINING_MODE=full`, `BENCHMARK_SEEDS=42` and `BENCHMARK_EPOCHS=20`. +The same ignored directory retains reports, epoch CSVs, all checkpoints needed +for the instability investigation, predictions, the paired quality bundle, +`checkpoint-checks.json`, and `curves.png` with its plotting script. No library +code changed; validation comprises the real paired run, metric evaluation, +cross-study held-out identity checks and checkpoint-step checks above. + ### Isolated 100k-class optimizer compilation comparison Revision `aff0807` exposes the existing optimizer compilation and graph options diff --git a/docs/quantization-status.md b/docs/quantization-status.md index a749614..50637b8 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the hierarchical training study](benchmarks.md#isolated-three-seed-hierarchical-training-comparison). +recorded in [the twenty-epoch hierarchical diagnostic](benchmarks.md#twenty-epoch-hierarchical-convergence-diagnostic). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -76,6 +76,16 @@ its timings are excluded from performance claims. This restores the prerequisite for a longer convergence study; it does not supply the missing historical curves. +A [fresh twenty-epoch hierarchical pair](benchmarks.md#twenty-epoch-hierarchical-convergence-diagnostic) +at seed 42 now completes with actual epoch statistics. Held-out INT8 leaf/parent +Macro-F1 differences are +0.331/−0.082 points, with mixed changes in other metrics. +INT8 retains a 3.4% allocated-memory saving but takes 9.2% longer for the local +training call. Both checkpoints record 2,300 optimizer/scheduler updates. Strong +INT8 validation-loss spikes in the first half of training settle later; replaying +retained intermediate checkpoints and examining train/eval state is the next +focused diagnostic. One longer-budget seed neither resolves the multi-seed +quality question nor establishes target-hardware benefit. + The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) now separates eager, compiled and graph execution in eighteen isolated processes. Ordinary optimizer compilation improves both precisions' local update times and From fe15d3a888c46edac6369ab08dcac14a27f4bd7f Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 20:18:48 +0200 Subject: [PATCH 088/155] docs: isolate BatchNorm state in hierarchical validation instability --- docs/benchmarks.md | 59 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 18 ++++++++--- 2 files changed, 73 insertions(+), 4 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 403c245..501b38b 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2546,6 +2546,65 @@ for the instability investigation, predictions, the paired quality bundle, code changed; validation comprises the real paired run, metric evaluation, cross-study held-out identity checks and checkpoint-step checks above. +### Hierarchical validation replay and BatchNorm-state diagnostic + +At revision `eb4c8dd`, fresh CUDA loads of the preceding 20-epoch study's retained +float and native INT8 checkpoints replay validation at epochs 6, 11 and 20. +The original sequential validation loader, 128px preprocessing, batch 32, BF16 +AMP and two-level loss are reused over all 912 validation images. Each checkpoint +reconstructs its model, so the replay does not inherit live training caches. +Using the logger's exact `topk=(1, 5)` accuracy convention, all six replays match +both recorded accuracies exactly; the largest absolute loss difference is +1.85e-7. The early INT8 loss spike therefore survives checkpoint reconstruction. +An initial argmax-based diagnostic differed on tied scores; its separate report +is retained, and the matched replay supersedes that accuracy comparison. + +A controlled ablation then refreshes the 110 backbone `BatchNorm2d` modules on +copies of epochs 6 and 20, for both precisions. Parameters remain fixed. Running +statistics are reset and accumulated with `momentum=None` over one shuffled, +training-only pass: 115 batches of 32 images (3,680 images; the training loader +drops its final partial batch). The shuffle seed is fixed at 42. Other modules +remain in evaluation mode, including dropout; preprocessing and BF16 execution +match the original run. No validation or test images update the statistics. + +| Precision / epoch | Validation loss before → after | Leaf accuracy before → after (%) | Parent accuracy before → after (%) | +| --- | ---: | ---: | ---: | +| Float / 6 | 1.567 → 1.090 | 70.15 → 76.72 | 80.28 → 87.28 | +| INT8 / 6 | 5.865 → 1.283 | 39.33 → 72.74 | 48.81 → 85.45 | +| Float / 20 | 0.934 → 0.917 | 84.91 → 85.56 | 92.35 → 93.10 | +| INT8 / 20 | 0.824 → 0.820 | 86.53 → 87.07 | 94.18 → 94.07 | + +Every ablation verifies exact equality of all physical parameter arrays before +and after, including INT8 data and scales. A separate early-INT8 repeat also +verifies exact equality of every non-BatchNorm buffer array; it reproduces the +same before/after metrics. The first buffer-inspection attempt used an unsupported +generic CPU conversion for a quantized cache; comparing its physical data/scales +resolved that diagnostic issue without changing the model implementation. + +This isolates backbone running-statistic refresh as sufficient to remove much +of the early checkpoint's validation degradation. It implicates a mismatch in +saved running statistics, rather than a cache retained from training, as a major +contributor in this case. It does not establish why the mismatch developed, +explain every fluctuation, or show that INT8 rounding directly caused it. Both +precisions improve at epoch 6, and final INT8 parent accuracy slightly decreases +after refresh. Universal recalibration or a changed default is not justified by +this single-seed diagnostic. + +These are validation loss/accuracy batch means, not the five held-out +mini_metrics measures or a production acceptance study. Before adopting a recipe, +qualify training-only refresh on additional seeds and heads, evaluate all five +metrics on an untouched held-out split, and include the extra pass in training +cost and checkpoint/export provenance. Inspect the original running-statistic +updates and training/evaluation behavior before choosing a permanent intervention. +No refreshed checkpoint was saved and no training defaults changed. + +Ignored `tmp-hierarchical-replay/` retains the replay/ablation scripts, the +argmax and matched replay reports, `batchnorm-report.json` and the stronger +`batchnorm-buffer-check.json`. The original intermediate checkpoints remain in +`tmp-hierarchical-long-budget/`. Validation consists of six matched real-model +replays, four controlled refresh ablations and one stronger state-preservation +repeat; no new speed, memory or target-hardware claim is made. + ### Isolated 100k-class optimizer compilation comparison Revision `aff0807` exposes the existing optimizer compilation and graph options diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 50637b8..898cee4 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the twenty-epoch hierarchical diagnostic](benchmarks.md#twenty-epoch-hierarchical-convergence-diagnostic). +recorded in [the hierarchical BatchNorm-state diagnostic](benchmarks.md#hierarchical-validation-replay-and-batchnorm-state-diagnostic). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -81,11 +81,21 @@ at seed 42 now completes with actual epoch statistics. Held-out INT8 leaf/parent Macro-F1 differences are +0.331/−0.082 points, with mixed changes in other metrics. INT8 retains a 3.4% allocated-memory saving but takes 9.2% longer for the local training call. Both checkpoints record 2,300 optimizer/scheduler updates. Strong -INT8 validation-loss spikes in the first half of training settle later; replaying -retained intermediate checkpoints and examining train/eval state is the next -focused diagnostic. One longer-budget seed neither resolves the multi-seed +INT8 validation-loss spikes in the first half of training settle later; the +checkpoint replay and state ablation below investigate that behavior. One longer-budget seed neither resolves the multi-seed quality question nor establishes target-hardware benefit. +The [checkpoint replay and BatchNorm ablation](benchmarks.md#hierarchical-validation-replay-and-batchnorm-state-diagnostic) +reproduce all six logged validation results after fresh loads. Refreshing only +backbone BatchNorm running statistics on training images reduces early INT8 loss +from 5.865 to 1.283 and raises parent accuracy from 48.81% to 85.45%, with all +parameters and non-BatchNorm buffers unchanged in the verified repeat. Float also +improves; final-checkpoint effects are much smaller and not uniformly positive. +This identifies running-statistic sensitivity as a substantial contributor in +this checkpoint, but not its origin or a universally beneficial production recipe. +Next inspect the training-state updates and qualify any proposed refresh on +additional seeds/heads with five held-out metrics and its extra execution cost. + The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) now separates eager, compiled and graph execution in eighteen isolated processes. Ordinary optimizer compilation improves both precisions' local update times and From 554313a90e57a7b2c8ed487794e70c955f44aabe Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 20:26:56 +0200 Subject: [PATCH 089/155] docs: verify BatchNorm updates and control refresh behavior --- docs/benchmarks.md | 53 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 11 +++++++- 2 files changed, 63 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 501b38b..28168bf 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2605,6 +2605,59 @@ argmax and matched replay reports, `batchnorm-report.json` and the stronger replays, four controlled refresh ablations and one stronger state-preservation repeat; no new speed, memory or target-hardware claim is made. +### BatchNorm update equations and refresh controls + +At revision `fe15d3a`, an eager BF16 audit checks all 110 backbone BatchNorm +modules in the float and native INT8 epoch-6 and epoch-20 checkpoints. The +pretrained checkpoint contains 671,327 tracked batches per module. Both training +precisions contain exactly 672,017 at epoch 6 and 673,627 at epoch 20: the +pretrained count plus 115 updates per training epoch. No unexplained counter +offset is present in these retained states. + +Forward hooks capture each layer's actual input statistics for one training +batch. Every module executes once, increments its counter once, and follows +momentum 0.1 for its running mean and unbiased variance. Expected values computed +from the inputs pass `rtol=1e-4, atol=2e-5`; maximum absolute differences are +3.81e-6 for means and 4.88e-4 for variances, consistent with floating reduction +rounding at the observed scales. A subsequent evaluation forward leaves all +running means, variances and counters exactly unchanged. No optimizer update is +performed in this probe. This finds no update-equation or mode-handling defect +in the examined eager path; it does not audit compiled graphs or distributed +training, nor reconstruct every historical batch. + +Six additional epoch-6 refresh controls hold all weights fixed and vary averaging +policy and stochastic-layer mode. Each resets BatchNorm state and processes the +same training-only 115 batches, as in the preceding ablation. Full training mode +retains normal stochastic depth/dropout; BatchNorm-only mode disables those +stochastic layers. All parameter and non-BatchNorm buffer arrays remain exactly +unchanged in every control. + +| Precision | Refresh mode | Momentum | Validation loss after | Leaf accuracy after (%) | Parent accuracy after (%) | +| --- | --- | --- | ---: | ---: | ---: | +| Float | BatchNorm only | 0.1 | 1.108 | 76.19 | 87.39 | +| Float | Full training mode | 0.1 | 1.110 | 76.19 | 87.39 | +| Float | Full training mode | Cumulative | 1.097 | 76.51 | 86.85 | +| INT8 | BatchNorm only | 0.1 | 1.310 | 71.77 | 84.81 | +| INT8 | Full training mode | 0.1 | 1.287 | 72.31 | 85.45 | +| INT8 | Full training mode | Cumulative | 1.256 | 72.95 | 85.88 | + +The original validation losses were 1.567 float and 5.865 INT8. Recovery therefore +occurs even with the normal training-mode stochastic layers and original +momentum. It is not specific to cumulative averaging or disabling stochastic +layers. Lag between the changing model and its running statistics is a working +explanation consistent with these controls; the experiment does not fully +establish how that lag arose or quantify each source of training instability. +There is no evidence here for changing BatchNorm's update equations or patching +its counters. + +Next qualify an explicit training-only refresh on additional heads/seeds with +all five held-out metrics and the added pass cost, before proposing a default or +production deployment recipe. These controls use validation batch-mean loss and +accuracy, not held-out acceptance or target-hardware performance measurements. +Scripts, per-module equation checks, pretrained counter evidence and six refresh +reports are retained in ignored `tmp-batchnorm-update-audit/`. No library or +training-default changes were made. + ### Isolated 100k-class optimizer compilation comparison Revision `aff0807` exposes the existing optimizer compilation and graph options diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 898cee4..25a04a1 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the hierarchical BatchNorm-state diagnostic](benchmarks.md#hierarchical-validation-replay-and-batchnorm-state-diagnostic). +recorded in [the BatchNorm update audit](benchmarks.md#batchnorm-update-equations-and-refresh-controls). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -96,6 +96,15 @@ this checkpoint, but not its origin or a universally beneficial production recip Next inspect the training-state updates and qualify any proposed refresh on additional seeds/heads with five held-out metrics and its extra execution cost. +The [BatchNorm update audit and refresh controls](benchmarks.md#batchnorm-update-equations-and-refresh-controls) +find correct eager update equations, exactly expected inherited batch counts, +and unchanged running state during evaluation. The early recovery also occurs +with normal training-mode stochastic layers and the original momentum 0.1 while +all weights remain fixed. Lagging statistics remain a plausible explanation; +these results do not justify patching BatchNorm counters or update equations. +Multi-seed/head held-out qualification and accounting for refresh cost remain +necessary before adopting an explicit refresh recipe. + The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) now separates eager, compiled and graph execution in eighteen isolated processes. Ordinary optimizer compilation improves both precisions' local update times and From 8b8a9369e72146b2d6bc4884f4b21397a22a4e64 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 20:39:15 +0200 Subject: [PATCH 090/155] docs: qualify BatchNorm refresh across heads and seeds --- docs/benchmarks.md | 70 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 13 ++++++- 2 files changed, 82 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 28168bf..da82d73 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -2658,6 +2658,76 @@ Scripts, per-module equation checks, pretrained counter evidence and six refresh reports are retained in ignored `tmp-batchnorm-update-audit/`. No library or training-default changes were made. +### Held-out qualification of training-only BatchNorm refresh + +At revision `554313a`, the fixed BatchNorm-only cumulative refresh is applied to +all twelve retained five-epoch full-training checkpoints: flat/hierarchical +EfficientNetV2-S, float/native INT8, seeds 42/43/44. This extends the preceding +early-checkpoint validation diagnostic to final-checkpoint held-out quality. +The procedure, seeds, thresholds and checkpoint epochs are fixed before +collecting the new held-out predictions. + +Each case reconstructs its original checkpoint, collects fresh before/after test +predictions and verifies exact equality of parameters and every non-BatchNorm +buffer array. Only the 110 backbone BatchNorm modules enter training mode. Their +statistics are reset and accumulated over 115 training-only batches (3,680 images, +batch 32, shuffle seed 42, BF16 AMP); other modules stay in evaluation mode. +Current contents of all 5,777 source images were checked against the recorded +hashes before the study. All twelve pre-refresh five-metric evaluations exactly +reproduce the previous held-out results, and all saved buffer-override hashes +were verified. The 1,161 test images never update running statistics. + +All twelve refresh cases and eighteen mini_metrics comparisons complete: twelve +before/after comparisons, plus six float/INT8 comparisons after refreshing both. +All five metrics are finite; coverage is 1.0 throughout. Tables show observed +three-seed ranges of candidate-minus-baseline differences multiplied by 100, +not confidence intervals. + +| Refresh effect on model | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat float / leaf | −0.802 to +1.602 | −0.725 to +1.660 | −1.330 to +1.364 | 0 | +0.186 to +0.582 | +| Flat INT8 / leaf | −0.330 to +0.990 | −0.636 to +0.911 | −0.115 to +1.758 | 0 | −0.226 to +0.531 | +| Hierarchical float / leaf | −0.876 to +0.314 | −0.840 to −0.157 | −0.748 to +0.979 | 0 | −0.523 to +0.274 | +| Hierarchical float / parent | −0.445 to −0.163 | −1.604 to −0.238 | −0.995 to +1.164 | 0 | −0.700 to +0.096 | +| Hierarchical INT8 / leaf | −0.491 to +2.034 | −0.640 to +1.673 | −0.500 to +1.610 | 0 | +0.029 to +0.451 | +| Hierarchical INT8 / parent | −0.362 to +1.038 | −0.883 to +0.730 | −0.176 to +1.483 | 0 | −0.026 to +0.162 | + +Refresh has mixed effects on final-checkpoint quality. In particular, it lowers +hierarchical float parent Macro-F1 in every seed. The strong early validation +recovery therefore does not establish a generally beneficial final-model recipe. + +| INT8 minus float after both are refreshed | Macro-F1 | Macro-Recall | Macro-Precision | Coverage | Theil's U | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat / leaf | −2.370 to +3.640 | −3.070 to +4.635 | −0.854 to +2.139 | 0 | −1.107 to +2.731 | +| Hierarchical / leaf | −3.516 to +5.356 | −3.050 to +5.363 | −3.829 to +5.303 | 0 | −1.765 to +3.249 | +| Hierarchical / parent | −3.652 to −0.276 | −3.120 to +0.204 | −5.059 to −2.075 | 0 | −1.110 to +0.681 | + +Parent Macro-F1 and Macro-Precision still trail float in all three hierarchical +pairs. After refresh, parent Macro-F1 spans 0.8340–0.8561 float versus +0.8196–0.8376 INT8. Refresh does not resolve the earlier short-budget quality +concern or establish accuracy superiority. Automatic refresh is not adopted. + +Each GPU case runs in a fresh sequential process; seed 43 reverses precision +order. No tests, benchmarks or metric jobs overlap the refresh timings. The +synchronized refresh pass takes **4.474–5.494 seconds**, with a separate +**0.996–1.340 seconds** for training-cache construction on this laptop. The pass +includes buffer reset, cached loading, preprocessing, forward computation and +mode restoration; it excludes checkpoint/model loading, state verification and +held-out inference. These local costs do not establish target-machine overhead, +a cold-start measurement, added peak memory or an end-to-end training benefit. +GPU clocks, thermals and background OS activity remain uncontrolled. + +Ignored `tmp-bn-refresh-qualification/` retains the exact driver and command +protocol, source provenance, before/after scores and CSVs, all eighteen metric +reports, timing/state checks and `verification.json`. Small +`batchnorm-overrides.pt` files preserve the changed buffers without duplicating +full checkpoints; apply them over the identified original checkpoint to reproduce +the candidate state. The full study occupies about 36 MiB. No library defaults +or original checkpoints changed. Frozen-backbone models, longer-budget repeated +seeds and target hardware are not qualified by this study. The result favors +keeping refresh as an explicit diagnostic and returning efficiency work to the +large-head and deployment regimes where substantial benefits remain plausible. + ### Isolated 100k-class optimizer compilation comparison Revision `aff0807` exposes the existing optimizer compilation and graph options diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 25a04a1..d30ba90 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the BatchNorm update audit](benchmarks.md#batchnorm-update-equations-and-refresh-controls). +recorded in [the held-out refresh qualification](benchmarks.md#held-out-qualification-of-training-only-batchnorm-refresh). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -105,6 +105,17 @@ these results do not justify patching BatchNorm counters or update equations. Multi-seed/head held-out qualification and accounting for refresh cost remain necessary before adopting an explicit refresh recipe. +The [held-out refresh qualification](benchmarks.md#held-out-qualification-of-training-only-batchnorm-refresh) +now covers all twelve five-epoch full-training checkpoints across both heads and +three seeds, with eighteen paired five-metric evaluations. Effects are mixed; +hierarchical parent Macro-F1 still trails float in every seed after both models +are refreshed (−3.652 to −0.276 points), and parent precision losses reach 5.059 +points. The extra local cached pass costs 4.474–5.494 seconds, excluding model +loading and inference. Automatic refresh is not adopted. This closes the bounded +final-checkpoint qualification with a negative general-recipe result; it does not +qualify frozen/longer-budget models or target hardware. Prioritize large-head +training/deployment efficiency and the remaining target-runner/reporting work. + The [100k-class optimizer study](benchmarks.md#isolated-100k-class-optimizer-compilation-comparison) now separates eager, compiled and graph execution in eighteen isolated processes. Ordinary optimizer compilation improves both precisions' local update times and From bb7a93df9ab7104b44a2c70e35878cfd33ec446a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 20:59:36 +0200 Subject: [PATCH 091/155] docs: record 10000-class TensorRT capacity and latency comparison --- docs/benchmarks.md | 73 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 10 ++++- 2 files changed, 82 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index da82d73..52dca87 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3101,3 +3101,76 @@ messages and external-weight provenance. See the Validation passed static checks and 406 CPU-default tests (149 skips and the known EMA expected failure), plus all 16 focused checks in the prepared CUDA/ TensorRT environment and both real-model reference comparisons above. + +### Ten-thousand-class TensorRT capacity comparison + +A bounded capacity study uses a pretrained EfficientNetV2-S backbone and a seeded +random symmetric normalized flat head with 10,000 classes (seed 42, 14,462,240 +head parameters). This tests deployment capacity and execution; the random head +cannot establish trained classification quality. All 128 training calibration +images were checked against this model's preprocessing, with exact input-array +equality. Inputs have shape `[batch, 3, 128, 128]`. + +The initial FP32 ONNX export **failed the default numerical tolerance**. The +study explicitly used `rtol=1e-4, atol=1e-4` on retry; public defaults were not +changed. Additional CPU comparisons at real batches 1, 2, 4 and 8 had maximum +absolute score error 2.134e-5, respectively 7/9/8/12 scores outside the default +tolerance, and no argmax changes. This limited check does not qualify default +export parity or establish trained quality. The default failure remains an open +large-head export qualification issue. + +The maintained calibration command used percentile 99.9, signed symmetric INT8 +activations, per-channel INT8 weights and floating biases, with asymmetric +histogram collection. Both TensorRT engines used profiles min/opt/max 1/8/8, +1 GiB workspace, optimization level 1, TF32 disabled, and **FP16 enabled in both**. +Thus remaining floating operations in the INT8 candidate can execute in FP16; +this differs from earlier studies that disabled FP16 for the INT8 engine. +Build success establishes parsing and finite execution, not numerical acceptance. +Inspection shows all 170 convolutions and both head GEMMs execute with INT8 in +the candidate; the baseline uses half precision for those operations. + +| Artifact requirement | FP16 baseline | INT8 candidate | Reduction | +| --- | ---: | ---: | ---: | +| Serialized engine bytes | 71,603,924 | 39,234,916 | 45.2% | +| Reported context memory bytes | 5,669,888 | 4,276,224 | 24.6% | + +Context requirements exclude weights and other process/runtime allocations; +these are not measurements of total GPU memory savings. + +Three fresh paired timing processes per batch used the maintained +`dev.benchmarks.tensorrt_pair` command, ten warmups and 31 alternating paired +samples each. Trial two reversed the initial engine order and batch order. +The GPU was idle before starting, with no competing GPU workload observed. +Host IO was pageable. Timing includes preallocated H2D input copies, execution, +D2H output copies and stream synchronization; it excludes preprocessing, +allocation and loading. Both engines coexist. Hardware was the local RTX 3080 Ti +Laptop GPU under WSL, with TensorRT 10.16.1.11 and one PyTorch intra-op thread. + +| Batch | Trial | FP16 median ms | INT8 median ms | Median paired INT8 / FP16 | +| --- | --- | ---: | ---: | ---: | +| 1 | 1 | 3.488 | 3.808 | 1.134 | +| 1 | 2 | 3.214 | 3.714 | 1.170 | +| 1 | 3 | 3.465 | 4.199 | 1.165 | +| 8 | 1 | 3.642 | 4.079 | 1.158 | +| 8 | 2 | 3.732 | 4.235 | 1.158 | +| 8 | 3 | 3.775 | 4.216 | 1.120 | + +All six reports passed and retained finite outputs of shape `[batch, 10000]`. +Engine hashes were verified across every trial. The INT8 candidate was 12–17% +slower in these paired measurements. This configuration therefore supplies no +local speed benefit; its smaller artifact does not establish the user's joint +quality/efficiency acceptance criterion. It also does not predict behavior on +Spark, the intended RTX desktop, HPC GPUs or ARM CPUs. A larger bounded head +and isolated runtime-memory measurements remain useful next capacity checks. + +Inputs, checkpoint, exact build/timing commands, calibration provenance, engine +inspection, raw timings and outputs are retained under ignored +`tmp-large-head-deployment/`; `capacity.json` records the export override and +`timing-summary.json` records the verified trials. Source checkpoint SHA256 is +`30af5dd007656f2e5639012874e17e8b27231de99d3d063eb764de6077e3bc97`. +FP16 engine SHA256 is +`eba430f9ff1da76ed333cdb1bb145aba4bfb311dce2cfb7e8fd4ed6f4690cafa`; +INT8 engine SHA256 is +`53d3020e506e10c388cda6f1ae09c7bbf7554f95670c405fdb36dcf0f02f774d`. +No implementation change or new trained mini_metrics comparison accompanies +this capacity milestone. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index d30ba90..dcfe373 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,13 +3,21 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the held-out refresh qualification](benchmarks.md#held-out-qualification-of-training-only-batchnorm-refresh). +recorded in [the 10,000-class TensorRT capacity study](benchmarks.md#ten-thousand-class-tensorrt-capacity-comparison). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost or memory benefit. This does not make a quality-only result, smaller engine file, or integer operator count sufficient evidence of production readiness. +A 10,000-class seeded random normalized head now builds and executes through +calibrated TensorRT, including 170 INT8 convolutions and both INT8 head GEMMs. +Across six fresh paired trials, INT8 was 12–17% slower than FP16 locally, despite +a 45.2% smaller engine and 24.6% lower reported context requirement. Total runtime +memory and trained quality were not measured. FP32 export needed an explicit +absolute-tolerance override from 1e-5 to 1e-4; default parity qualification remains +open. This capacity evidence is not target-hardware certification. + ## Current evidence | Workstream | Verified locally | What remains unproven | From b375d985a756413a1c95291c7f68f9b3cf374f8a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 21:02:51 +0200 Subject: [PATCH 092/155] docs: investigate large-head ONNX numerical parity --- docs/benchmarks.md | 35 +++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 6 +++++- 2 files changed, 40 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 52dca87..2d8e56e 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3174,3 +3174,38 @@ INT8 engine SHA256 is `53d3020e506e10c388cda6f1ae09c7bbf7554f95670c405fdb36dcf0f02f774d`. No implementation change or new trained mini_metrics comparison accompanies this capacity milestone. + +#### Numerical attribution of the capacity export failure + +A follow-up CPU diagnostic used the same checkpoint and real batch eight, +comparing PyTorch FP32 to a copied FP64 model, exposing the ONNX backbone output, +and running the PyTorch head on those exact ONNX features. Classifier caches were +invalidated after conversion to FP64. All comparisons use the original +`atol=1e-5, rtol=1e-4` criterion; FP64 is a higher-precision reference, not a proof +of exact mathematical output. + +| Comparison | Maximum absolute error | Scores outside tolerance | +| --- | ---: | ---: | +| PyTorch FP32 versus FP64 | 1.121e-5 | 0 | +| ORT all optimizations versus PyTorch FP32 | 2.134e-5 | 12 | +| ORT all optimizations versus PyTorch FP64 | 1.610e-5 | 0 | +| ORT optimizations disabled versus PyTorch FP32 | 2.557e-5 | 29 | +| ORT optimizations disabled versus PyTorch FP64 | 2.245e-5 | 6 | +| ORT head versus PyTorch head on identical ORT features, all optimizations | 1.287e-5 | 0 | + +Basic-only optimization matched the disabled result. Backbone features also pass +this tolerance (maximum error 6.050e-6 with all optimizations). Thus disabling +optimization does not resolve this failure. The combined observations are +consistent with accumulated backbone/head rounding differences: separate head +and feature checks pass, but the end-to-end comparison does not. They do not +isolate a faulty operator or justify changing normalized score semantics, turning +off optimization, or weakening the default export gate. In particular, both +implementations passing against FP64 does not make their mutual parity pass. + +The diagnostic only added an intermediate graph output to a separate file; it +did not change the deployment graph, engines, checkpoint or library behavior. +The optimized diagnostic reproduces the original batch-eight maximum error and +12 failing scores. Script and detailed measurements are retained in +`tmp-large-head-deployment/parity-diagnostic.py` and `parity-diagnostic.json`. +The process completed successfully. No training or GPU test was needed for this +research-only increment; default large-head export qualification remains open. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index dcfe373..18b3c7f 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -16,7 +16,11 @@ Across six fresh paired trials, INT8 was 12–17% slower than FP16 locally, desp a 45.2% smaller engine and 24.6% lower reported context requirement. Total runtime memory and trained quality were not measured. FP32 export needed an explicit absolute-tolerance override from 1e-5 to 1e-4; default parity qualification remains -open. This capacity evidence is not target-hardware certification. +open. A [CPU numerical diagnostic](benchmarks.md#numerical-attribution-of-the-capacity-export-failure) +found that disabling ONNX optimization worsens parity, while both optimized ORT +and PyTorch FP32 separately pass against a FP64 reference on batch eight. This +narrows the investigation without establishing a faulty operator or relaxing the +export gate. This capacity evidence is not target-hardware certification. ## Current evidence From 8166d370e8b238f978561d22d61d675055c231a1 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 21:05:15 +0200 Subject: [PATCH 093/155] docs: measure isolated TensorRT deployment memory --- docs/benchmarks.md | 42 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 7 +++++-- 2 files changed, 47 insertions(+), 2 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 2d8e56e..9df70d1 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3209,3 +3209,45 @@ The optimized diagnostic reproduces the original batch-eight maximum error and `tmp-large-head-deployment/parity-diagnostic.py` and `parity-diagnostic.json`. The process completed successfully. No training or GPU test was needed for this research-only increment; default large-head export qualification remains open. + +#### Isolated deployment memory snapshots + +Six fresh processes loaded one of the exact 10,000-class engines above, created +its context and pageable IO buffers for batch eight, and completed 20 synchronized +inferences. There were three trials per engine, with order reversed in trial two. +The GPU was idle before the study and no compute process was listed. Every run +passed finite `[8, 10000]` output checks and engine-hash verification. + +CUDA free/total snapshots were collected after CUDA initialization, engine load, +context/IO creation and warm execution. These are **device-wide snapshots**, not +per-process GPU accounting or transient allocation peaks. The initialization +snapshot was 1,176.5 MiB in all six processes. PyTorch allocated/reserved memory +was recorded separately; it does not account for TensorRT's own allocations. +Host RSS/PSS came from each process's `/proc/self/smaps_rollup`. + +| Engine | Trial | Warm device-used MiB | Increase from CUDA initialization MiB | Warm host RSS MiB | Warm host PSS MiB | +| --- | --- | ---: | ---: | ---: | ---: | +| FP16 | 1 | 1306.5 | 130.0 | 957.2 | 950.2 | +| INT8 | 1 | 1272.5 | 96.0 | 872.5 | 865.5 | +| FP16 | 2 | 1306.5 | 130.0 | 954.8 | 947.8 | +| INT8 | 2 | 1272.5 | 96.0 | 874.8 | 867.7 | +| FP16 | 3 | 1306.5 | 130.0 | 956.6 | 949.5 | +| INT8 | 3 | 1272.5 | 96.0 | 870.4 | 863.3 | + +The consistent device difference is 34 MiB: 26.2% of the FP16 increment after +initialization, but only 2.6% of its warm device-wide snapshot including the +common runtime/background footprint. Host RSS is about 8.4–9.0% lower. These +measurements demonstrate why the 45.2% engine-file reduction must not be described +as a corresponding total deployment-memory reduction. Combined with the measured +12–17% latency regression, this local 10,000-class profile has not demonstrated +a compelling production trade-off. A larger head may change the balance but +requires a new measurement, and target hardware remains unverified. + +The probe uses the maintained TensorRT IO-buffer helper with one engine per +process; it is an experimental driver, not yet a maintained memory command. +`tmp-large-head-deployment/memory-probe.py`, `measure-memory.py`, +`memory-protocol.json`, `memory-summary.json`, and all six reports/output archives +retain its protocol and evidence. WSL device-memory accounting and unobserved +background activity limit attribution. No claim of measured total GPU peak memory +or minimal standalone TensorRT host footprint is made: this probe imports PyTorch +for CUDA and IO management. This research increment changes no runtime code. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 18b3c7f..2303a90 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -13,8 +13,11 @@ or integer operator count sufficient evidence of production readiness. A 10,000-class seeded random normalized head now builds and executes through calibrated TensorRT, including 170 INT8 convolutions and both INT8 head GEMMs. Across six fresh paired trials, INT8 was 12–17% slower than FP16 locally, despite -a 45.2% smaller engine and 24.6% lower reported context requirement. Total runtime -memory and trained quality were not measured. FP32 export needed an explicit +a 45.2% smaller engine and 24.6% lower reported context requirement. [Isolated memory snapshots](benchmarks.md#isolated-deployment-memory-snapshots) +subsequently showed a consistent 34 MiB lower device reading: 26.2% of the +increment after CUDA initialization, but 2.6% of the warm device-wide footprint. +Host RSS was 8.4–9.0% lower. Total GPU transient peaks and trained quality remain +unmeasured. FP32 export needed an explicit absolute-tolerance override from 1e-5 to 1e-4; default parity qualification remains open. A [CPU numerical diagnostic](benchmarks.md#numerical-attribution-of-the-capacity-export-failure) found that disabling ONNX optimization worsens parity, while both optimized ORT From a13b9fc3dc31da7c1d3b71fd8ef2038ceff21d3d Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 21:24:30 +0200 Subject: [PATCH 094/155] feat: add isolated TensorRT memory measurement command --- dev/benchmarks/README.md | 39 +++++++ dev/benchmarks/tensorrt_memory.py | 140 ++++++++++++++++++++++++ docs/benchmarks.md | 30 +++++ docs/quantization-status.md | 2 +- tests/test_benchmark_tensorrt_memory.py | 90 +++++++++++++++ 5 files changed, 300 insertions(+), 1 deletion(-) create mode 100644 dev/benchmarks/tensorrt_memory.py create mode 100644 tests/test_benchmark_tensorrt_memory.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index b343a6d..56cd84c 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1516,3 +1516,42 @@ for a configured target runner but is not yet wired into the scheduled GPU job; durable result hosting and joint quality/resource acceptance gates remain open. Local CUDA success does not establish HPC throughput, desktop ONNX performance or ARM inference support. + +### Isolated TensorRT memory command + +Use `dev.benchmarks.tensorrt_memory` in a **fresh Linux process for each engine +and trial**, after preparing TensorRT/CUDA explicitly. The command installs +nothing and accepts the same arbitrary named NPZ inputs as the build and paired +latency commands. Engines must be rebuilt for the intended target environment. + +```bash +CUDA_VISIBLE_DEVICES=0 python -m dev.benchmarks.tensorrt_memory \ + --engine results/fp16/model.engine --inputs inputs-8.npz \ + --output results/memory-fp16-trial1 --runs 20 --threads 1 +CUDA_VISIBLE_DEVICES=0 python -m dev.benchmarks.tensorrt_memory \ + --engine results/int8/model.engine --inputs inputs-8.npz \ + --output results/memory-int8-trial1 --runs 20 --threads 1 +``` + +Repeat in fresh processes and alternate engine order. Stop competing GPU jobs +before measuring. `--pinned` enables pinned host buffers and nonblocking copies; +keep this choice, input shapes and run counts matched across candidates. +`--device` selects the visible CUDA device, with engine optimization profile zero. +Each output directory must be new. Reports retain engine/input/output hashes, +versions, GPU identity, IO contracts, TensorRT warnings, and available snapshots +if deserialization, input validation or execution fails. + +Snapshots cover host memory before runtime imports, then host/device memory after +CUDA initialization, engine load, context/IO allocation and synchronized repeated +execution. Every inference output must be finite. The command records Linux +RSS/PSS/swap and the approximate host high-water mark, device-wide free/total +memory, and PyTorch allocated/reserved counters separately. Snapshots precede +output serialization. Model decoding/preprocessing is external. + +Interpret the fields according to their scope: device-wide changes can include +other processes and driver accounting; PyTorch counters omit TensorRT-owned +allocations; context requirements omit engine weights and other runtime costs. +These snapshots do **not** measure total transient GPU allocation peaks. Importing +PyTorch for CUDA/IO management also affects this process's footprint. Use paired +latency and full `mini_metrics` quality evaluation separately before accepting a +memory/quality trade-off. CPU/WSL evidence does not establish target GPU results. diff --git a/dev/benchmarks/tensorrt_memory.py b/dev/benchmarks/tensorrt_memory.py new file mode 100644 index 0000000..78faf3e --- /dev/null +++ b/dev/benchmarks/tensorrt_memory.py @@ -0,0 +1,140 @@ +"""Collect single-engine TensorRT memory snapshots in a fresh Linux process.""" + +import hashlib +import io +import json +import platform +from argparse import ArgumentParser +from pathlib import Path + +import numpy as np + +from .onnx_cpu_memory import resident_memory +from .onnx_inference import file_hash +from .tensorrt_pair import buffers_for + + +def memory_snapshot(torch, device): + """Keep device-wide observations separate from process and allocator counters.""" + torch.cuda.synchronize(device) + free, total = torch.cuda.mem_get_info(device) + return { + "device_used_bytes": total - free, + "device_total_bytes": total, + "torch_allocated_bytes": torch.cuda.memory_allocated(device), + "torch_reserved_bytes": torch.cuda.memory_reserved(device), + "host": resident_memory(), + } + + +def measure(engine, inputs, output, runs=20, threads=1, device=0, pinned=False): + """Call the CLI separately for every engine/trial; no environment installation.""" + if runs < 1 or threads < 1 or device < 0: + raise ValueError("Require positive runs/threads and a nonnegative device") + output = Path(output) + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "environment": {"platform": platform.platform(), "python": platform.python_version()}, + "settings": {"runs": runs, "threads": threads, "device": device, "pinned_host_io": pinned, "profile": 0}, + "memory": {}, + "messages": [], + "scope": ( + "Run one engine per fresh Linux process. Device used/free is device-wide, not per-process or a transient peak; " + "other activity and WSL accounting can affect attribution. PyTorch counters exclude TensorRT-owned allocations. " + "Host RSS/PSS includes interpreter, imports, inputs, engine and IO; host peak is approximate since exec. " + "No latency or quality acceptance claim. Snapshots precede output serialization." + ), + } + try: + report["memory"]["before_runtime_import"] = {"host": resident_memory()} + try: + import tensorrt as trt + import torch + except ImportError as error: + raise ImportError("Use an explicitly prepared TensorRT and CUDA PyTorch environment; this command installs nothing") from error + torch.set_num_threads(threads) + report["versions"] = {"tensorrt": trt.__version__, "torch": torch.__version__, "numpy": np.__version__} + payload = Path(inputs).read_bytes() + report["inputs"] = {"path": str(inputs), "sha256": hashlib.sha256(payload).hexdigest()} + with np.load(io.BytesIO(payload), allow_pickle=False) as archive: + feeds = {name: archive[name].copy(order="C") for name in archive.files} + del payload + if not feeds or any(not np.isfinite(a).all() for a in feeds.values()): + raise ValueError("Supply finite named input arrays") + + class Logger(trt.ILogger): + def log(self, severity, message): + if severity <= trt.ILogger.WARNING: + report["messages"].append({"severity": str(severity), "message": message}) + + with torch.cuda.device(device): + stream = torch.cuda.Stream(device=device) + report["environment"].update( + gpu=torch.cuda.get_device_name(device), compute_capability=list(torch.cuda.get_device_capability(device)) + ) + report["memory"]["cuda_initialized"] = memory_snapshot(torch, device) + logger = Logger() + trt.init_libnvinfer_plugins(logger, "") + runtime = trt.Runtime(logger) + serialized = Path(engine).read_bytes() + report["engine"] = {"path": str(engine), "sha256": hashlib.sha256(serialized).hexdigest(), "bytes": len(serialized)} + model = runtime.deserialize_cuda_engine(serialized) + del serialized + if model is None: + raise RuntimeError("Could not deserialize engine") + report["memory"]["engine_loaded"] = memory_snapshot(torch, device) + context = model.create_execution_context() + if context is None: + raise RuntimeError("Could not create execution context") + with torch.cuda.stream(stream): + buffers, input_names = buffers_for(model, context, feeds, torch, trt, device, pinned) + report["io"] = { + name: {"shape": list(host.shape), "dtype": str(host.numpy().dtype), "input": name in input_names} + for name, (_, host) in buffers.items() + } + if all(item["input"] for item in report["io"].values()): + raise ValueError("Require at least one engine output") + report["context_memory_bytes"] = model.device_memory_size_v2 + report["memory"]["context_and_io"] = memory_snapshot(torch, device) + for _ in range(runs): + with torch.cuda.stream(stream): + for name in input_names: + gpu, host = buffers[name] + gpu.copy_(host, non_blocking=pinned) + if not context.execute_async_v3(stream.cuda_stream): + raise RuntimeError("Engine execution failed") + for name, (gpu, host) in buffers.items(): + if name not in input_names: + host.copy_(gpu, non_blocking=pinned) + stream.synchronize() + if any(not np.isfinite(host.numpy()).all() for name, (_, host) in buffers.items() if name not in input_names): + raise ValueError("Nonfinite engine outputs") + report["memory"]["warm"] = memory_snapshot(torch, device) + arrays = {name: host.numpy() for name, (_, host) in buffers.items() if name not in input_names} + np.savez(output / "outputs.npz", **arrays) + report["outputs_sha256"] = file_hash(output / "outputs.npz") + report["status"] = "passed" + except Exception as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + raise + finally: + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + for name in ("engine", "inputs", "output"): + parser.add_argument("--" + name, type=Path, required=True) + parser.add_argument("--runs", type=int, default=20) + parser.add_argument("--threads", type=int, default=1) + parser.add_argument("--device", type=int, default=0) + parser.add_argument("--pinned", action="store_true") + measure(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 9df70d1..71b1517 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3251,3 +3251,33 @@ retain its protocol and evidence. WSL device-memory accounting and unobserved background activity limit attribution. No claim of measured total GPU peak memory or minimal standalone TensorRT host footprint is made: this probe imports PyTorch for CUDA and IO management. This research increment changes no runtime code. + +### Maintained single-engine TensorRT memory measurement + +`dev.benchmarks.tensorrt_memory` makes the isolated memory experiment reusable on +Linux GPU hosts. It accepts arbitrary named NPZ inputs and an existing engine, +reuses the paired runner's shape/dtype/device IO checks, and keeps engine/input/ +output hashes, environment, IO contracts and TensorRT warnings. Each run requires +a fresh output directory, and failures retain the available report and snapshots. +There is no model-name or class-count allowlist. + +The command records device-wide free/total memory and PyTorch allocator counters +separately from Linux process RSS/PSS/swap and approximate host high-water marks. +Snapshots occur after CUDA initialization, engine load, context/IO creation and +repeated synchronized execution. Every inference output must be finite; output +serialization follows the final snapshot. Device-wide differences are not +per-process accounting, PyTorch counters omit TensorRT-owned allocations, and +none of these snapshots establish total transient GPU peaks. Use fresh processes +and quiescent hardware for comparisons, and retain the separate latency and +five-metric quality checks. See the [command documentation](../dev/benchmarks/README.md#isolated-tensorrt-memory-command). + +CPU contracts check lazy CLI imports, actionable missing-runtime failures, +retained failure reports, protection against overwriting results, and invalid +settings. Intentional CUDA tests cover exact named outputs with multiple inputs, +pageable/pinned IO, available memory snapshots and retained input-contract errors. + +Validation passed static checks and the full CPU-default suite: 513 passed, +162 skipped and the known EMA expected failure. All seven focused checks also +passed with CUDA/TensorRT explicitly enabled. These validate the command's +contracts; the preceding experimental memory values retain their original probe +and provenance and are not silently reattributed to the maintained command. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 2303a90..843eb27 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -32,7 +32,7 @@ export gate. This capacity evidence is not target-hardware certification. | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions; maintained 100k-class BF16 capacity comparison: full-model peak 17.4% lower; corrected frozen probe peak 6.6% lower after releasing an obsolete floating parameter; mixed timing; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | | QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, isolated memory snapshots, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | | CPU/edge inference | Full Blair metrics and isolated process-memory/timing on x86; unsigned CPU recipe executes 170 integer convolutions and two head GEMMs, with 44–54% lower warm batch-one latency and 45–48% lower resident memory in three trials per head | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging; larger-class qualification | | Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | diff --git a/tests/test_benchmark_tensorrt_memory.py b/tests/test_benchmark_tensorrt_memory.py new file mode 100644 index 0000000..e8e0f48 --- /dev/null +++ b/tests/test_benchmark_tensorrt_memory.py @@ -0,0 +1,90 @@ +import json +import os +import subprocess +import sys + +import numpy as np +import pytest + +from dev.benchmarks.tensorrt_memory import measure + + +def test_help_does_not_import_gpu_libraries(): + subprocess.run( + [ + sys.executable, + "-c", + """ +import runpy, sys +sys.argv = ['tensorrt_memory', '--help'] +try: + runpy.run_module('dev.benchmarks.tensorrt_memory', run_name='__main__') +except SystemExit as error: + assert error.code == 0 +assert 'torch' not in sys.modules and 'tensorrt' not in sys.modules +""", + ], + check=True, + capture_output=True, + ) + + +def test_missing_runtime_retains_failure_and_does_not_overwrite(tmp_path, monkeypatch): + monkeypatch.setitem(sys.modules, "tensorrt", None) + output = tmp_path / "report" + with pytest.raises(ImportError, match="explicitly prepared"): + measure("absent.engine", "absent.npz", output) + report = json.loads((output / "report.json").read_text()) + assert report["status"] == "failed" and report["error"].startswith("ImportError:") + with pytest.raises(FileExistsError): + measure("absent.engine", "absent.npz", output) + + +@pytest.mark.parametrize("settings", [{"runs": 0}, {"threads": 0}, {"device": -1}]) +def test_invalid_settings_fail_before_output_creation(tmp_path, settings): + with pytest.raises(ValueError): + measure("absent.engine", "absent.npz", tmp_path / "report", **settings) + assert not (tmp_path / "report").exists() + + +@pytest.mark.parametrize("pinned", [False, True]) +def test_real_engine_memory_and_retained_failure(tmp_path, pinned): + if os.environ.get("RUN_CUDA_TESTS") != "1": + pytest.skip("Set RUN_CUDA_TESTS=1 in an explicitly prepared GPU environment") + pytest.importorskip("tensorrt") + onnx = pytest.importorskip("onnx") + from dev.benchmarks.tensorrt_build import build + + graph = onnx.helper.make_graph( + [onnx.helper.make_node("Add", ["x", "offset"], ["scores"])], + "two-input", + [ + onnx.helper.make_tensor_value_info("x", onnx.TensorProto.FLOAT, [2, 2]), + onnx.helper.make_tensor_value_info("offset", onnx.TensorProto.FLOAT, [2]), + ], + [onnx.helper.make_tensor_value_info("scores", onnx.TensorProto.FLOAT, [2, 2])], + ) + model = tmp_path / "model.onnx" + onnx.save(onnx.helper.make_model(graph, opset_imports=[onnx.helper.make_opsetid("", 18)], ir_version=10), model) + inputs = tmp_path / "inputs.npz" + x = np.arange(4, dtype=np.float32).reshape(2, 2) + offset = np.array([0.5, -0.5], dtype=np.float32) + np.savez(inputs, x=x, offset=offset) + build(model, inputs, tmp_path / "build", optimization=0) + engine = tmp_path / "build/model.engine" + result = measure(engine, inputs, tmp_path / "memory", runs=2, pinned=pinned) + assert result["status"] == "passed" + assert result == json.loads((tmp_path / "memory/report.json").read_text()) + for stage in ("cuda_initialized", "engine_loaded", "context_and_io", "warm"): + snapshot = result["memory"][stage] + assert 0 <= snapshot["device_used_bytes"] <= snapshot["device_total_bytes"] + assert snapshot["host"]["resident_bytes"] > 0 + with np.load(tmp_path / "memory/outputs.npz") as outputs: + assert outputs.files == ["scores"] + np.testing.assert_array_equal(outputs["scores"], x + offset) + bad_inputs = tmp_path / "wrong.npz" + np.savez(bad_inputs, wrong=x) + with pytest.raises(ValueError, match="Input names"): + measure(engine, bad_inputs, tmp_path / "failed") + failure = json.loads((tmp_path / "failed/report.json").read_text()) + assert failure["status"] == "failed" and "engine_loaded" in failure["memory"] From 2ee441c13c5c13dfc5bd5cb4c349e1e3311e41f0 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 21:27:08 +0200 Subject: [PATCH 095/155] docs: report 100000-class TensorRT latency and memory trade-offs --- docs/benchmarks.md | 90 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 12 ++++- 2 files changed, 101 insertions(+), 1 deletion(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 71b1517..905a8de 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3281,3 +3281,93 @@ Validation passed static checks and the full CPU-default suite: 513 passed, passed with CUDA/TensorRT explicitly enabled. These validate the command's contracts; the preceding experimental memory values retain their original probe and provenance and are not silently reattributed to the maintained command. + +### Hundred-thousand-class TensorRT capacity comparison + +The bounded deployment study now reaches 100,000 classes using the same +pretrained EfficientNetV2-S and symmetric normalized flat-head configuration. +Seed 42 produces 129,842,240 head parameters and 600,078,912 total parameter +bytes. All 780 backbone tensors and both hidden-layer tensors were checked for +exact equality against the 10,000-class checkpoint; the final class projection +changes. The head is randomly initialized, so this is capacity/execution evidence, +not a trained 100,000-class quality comparison. + +All 128 training calibration inputs again exactly matched this model's +preprocessing. Export passed the declared study tolerance `rtol=1e-4, atol=1e-4`; +the stricter default was not tested in this run. Calibration used the same +percentile-99.9 signed symmetric activations, asymmetric histograms, per-channel +INT8 weights and floating biases as the 10,000-class study. Both engines enabled +FP16, disabled TF32, and used profile 1/8/8 at 128x128, optimization level 1 and +1 GiB workspace. Builds overlapped CPU correctness checks; the inference +measurements below started after those checks completed, with an idle GPU and +no other compute process listed. Build duration is not a performance result. + +Inspection confirms 170 INT8 convolutions and both INT8 head GEMMs in the +candidate. The FP16-enabled baseline has 170 half-weight convolutions and head +GEMMs with half-precision inputs; its selected hidden GEMM uses a different +accumulation/output tactic from the 10,000-class build. Engine sizes are +302,529,468 bytes for FP16 and 155,355,740 bytes for INT8, a 48.6% reduction. +Reported context requirements are 5,771,776 and 4,303,360 bytes respectively; +these do not include total runtime memory. + +Six fresh processes used `dev.benchmarks.tensorrt_pair`, three per batch, with +10 warmups and 31 alternating paired samples. Trial two reversed the initial +engine and batch order. Pageable host IO, synchronized H2D/execution/D2H timing, +one PyTorch intra-op thread, the RTX 3080 Ti Laptop GPU and TensorRT 10.16.1.11 +match the earlier timing protocol. Both engines coexist during paired timing. + +| Batch | Trial | FP16 median ms | INT8 median ms | Median paired INT8 / FP16 | +| --- | --- | ---: | ---: | ---: | +| 1 | 1 | 3.712 | 4.749 | 1.148 | +| 1 | 2 | 3.775 | 4.029 | 1.084 | +| 1 | 3 | 4.032 | 4.215 | 1.091 | +| 8 | 1 | 4.362 | 4.747 | 1.049 | +| 8 | 2 | 4.401 | 4.665 | 1.064 | +| 8 | 3 | 4.711 | 4.884 | 1.086 | + +Ratios are medians of adjacent paired observations, not ratios of the displayed +marginal medians. INT8 remains 8.4–14.8% slower at batch one and 4.9–8.6% slower +at batch eight. These small-batch results do not establish large-batch throughput +or benefit on the intended desktop/Spark hardware. + +Six additional fresh processes used the maintained +`dev.benchmarks.tensorrt_memory` command at revision `a13b9fc`, one engine per +process, 20 runs at batch eight, pageable IO and one thread. Engine order reversed +in trial two. CUDA initialization snapshots were 1,176.5 MiB in all six runs. + +| Engine | Trial | Warm device-used MiB | Increase from CUDA initialization MiB | Warm host RSS MiB | Warm host PSS MiB | +| --- | --- | ---: | ---: | ---: | ---: | +| FP16 | 1 | 1496.5 | 320.0 | 975.2 | 968.3 | +| INT8 | 1 | 1366.5 | 190.0 | 973.4 | 966.4 | +| FP16 | 2 | 1496.5 | 320.0 | 977.3 | 970.4 | +| INT8 | 2 | 1366.5 | 190.0 | 971.3 | 964.3 | +| FP16 | 3 | 1496.5 | 320.0 | 975.0 | 968.0 | +| INT8 | 3 | 1366.5 | 190.0 | 975.6 | 968.5 | + +The consistent device difference is 130 MiB: 40.6% of the increase after CUDA +initialization, or 8.7% of the entire warm device-wide snapshot. Host RSS is +essentially unchanged. This is a stronger incremental memory result than the +10,000-class study, but device snapshots remain subject to WSL/driver and +background-process accounting and do not capture total transient GPU peaks. +The maintained probe also checks finite outputs every iteration, whereas the +older experimental memory driver checked its final output; retain that protocol +distinction when comparing host observations across studies. + +All six timing and six memory reports passed; independent checks verified engine +hashes and finite `[batch, 100000]` outputs throughout the retained final-output +archives. Raw paired samples, memory stages, outputs, source/preprocessing checks, +calibration, build inspection, drivers and exact command records remain under +ignored `tmp-100k-head-deployment/`, with `summary.json` holding the verified +aggregate. Source checkpoint SHA256 is +`ba1d4c5cc9d2c45e73fc2b4b8dbbe736a82724770201c0b573bf041e38482ab3`. +FP16 engine SHA256 is +`e418de6af2a73999379502865fcbad70fc8106854bbb875fd259bc8b68f392d1`; +INT8 engine SHA256 is +`09eecbd2c34a5cf477082084ea66eeb382253c5a687b6a1eab3e0bf2df1295b2`. + +This closes a bounded 100,000-class flat deployment capacity check. It does not +qualify a trained large-class dataset, hierarchical deployment at this class +count, larger batches, native dynamic-quantizer GPU export, HPC training, or ARM +inference. The observed memory saving warrants further joint quality/efficiency +qualification; the latency regression must remain visible. No million-class +model or new production acceptance claim is introduced. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 843eb27..5fffbb7 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the 10,000-class TensorRT capacity study](benchmarks.md#ten-thousand-class-tensorrt-capacity-comparison). +recorded in [the 100,000-class TensorRT capacity study](benchmarks.md#hundred-thousand-class-tensorrt-capacity-comparison). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -25,6 +25,16 @@ and PyTorch FP32 separately pass against a FP64 reference on batch eight. This narrows the investigation without establishing a faulty operator or relaxing the export gate. This capacity evidence is not target-hardware certification. +The subsequent 100,000-class flat capacity study also executes 170 INT8 +convolutions and both INT8 head GEMMs. Six fresh paired timing runs remain +4.9–14.8% slower than FP16. Six isolated runs of the maintained memory command +show a 130 MiB lower device snapshot: 40.6% of the increment after CUDA +initialization and 8.7% of the warm device-wide footprint, with essentially +unchanged host RSS. This supplies stronger incremental memory evidence, but the +random head cannot qualify trained quality and snapshots do not measure total +GPU peaks. Hierarchical large-class deployment, larger batches and target +hardware remain outstanding. + ## Current evidence | Workstream | Verified locally | What remains unproven | From 5ebfe5d31031ea23d52fa44fa1722de93d403e3d Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 21:32:23 +0200 Subject: [PATCH 096/155] docs: qualify larger-batch 100000-class TensorRT deployment --- docs/benchmarks.md | 71 +++++++++++++++++++++++++++++++++++++ docs/quantization-status.md | 12 +++++-- 2 files changed, 80 insertions(+), 3 deletions(-) diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 905a8de..cf8811b 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3371,3 +3371,74 @@ count, larger batches, native dynamic-quantizer GPU export, HPC training, or ARM inference. The observed memory saving warrants further joint quality/efficiency qualification; the latency regression must remain visible. No million-class model or new production acceptance claim is introduced. + +### Larger-batch hundred-thousand-class deployment + +A follow-up rebuild keeps the exact 100,000-class floating and calibrated INT8 +ONNX artifacts from the preceding study, using matching TensorRT profiles with +min/opt/max batch 1/32/64. Batches 32 and 64 concatenate the first four/eight +retained calibration batches; all input-file hashes were verified and the 64 +images are distinct. The existing 128-image preprocessing equivalence check still +applies. These training inputs measure execution; the seeded random class head +still supplies no trained classification-quality result. + +Both builds enable FP16, disable TF32, use optimization level 1 and 1 GiB workspace, +and pass finite batch-32 smoke execution. Inspection confirms 170 half-weight +convolutions and half-input head GEMMs in the baseline, versus 170 INT8 +convolutions and both INT8 head GEMMs in the candidate. FP16/INT8 engine sizes are +302,384,868/155,584,412 bytes; context requirements are 46,174,208/34,426,880 bytes. +These are new engines with new tactics; the profile change is part of the study. +No competing compute process was listed before builds or measurement. + +Six fresh processes use the maintained paired timing command, three per batch, +10 warmups, 31 alternating paired observations, with initial engine and batch +order reversed in trial two. The same RTX 3080 Ti Laptop GPU, TensorRT 10.16.1.11, +pageable host IO and one PyTorch intra-op thread apply. Timings include H2D, +execution, D2H and stream synchronization, excluding loading and preprocessing. +Both engines coexist during timing. + +| Batch | Trial | FP16 median ms | INT8 median ms | Median paired INT8 / FP16 | +| --- | --- | ---: | ---: | ---: | +| 32 | 1 | 7.334 | 7.864 | 1.045 | +| 32 | 2 | 7.879 | 8.384 | 1.035 | +| 32 | 3 | 7.778 | 7.820 | 1.022 | +| 64 | 1 | 12.913 | 12.775 | 0.998 | +| 64 | 2 | 13.047 | 13.210 | 1.011 | +| 64 | 3 | 12.654 | 12.515 | 0.996 | + +INT8 is 2.2–4.5% slower at batch 32. At batch 64 the paired ratios span +0.996–1.011, effectively tied in this study, with no stable speedup established. +The wider batch regime reduces the observed relative latency penalty, but this +includes a profile/tactic rebuild and is not an isolated attribution to head IO. + +Six further fresh processes run the maintained memory command at batch 64, +20 inferences each, with alternating engine order between trials. All six begin +at the same 1,176.5 MiB device-used snapshot after CUDA initialization. + +| Engine | Trial | Warm device-used MiB | Increase from CUDA initialization MiB | Warm host RSS MiB | Warm host PSS MiB | +| --- | --- | ---: | ---: | ---: | ---: | +| FP16 | 1 | 1554.5 | 378.0 | 1017.4 | 1010.5 | +| INT8 | 1 | 1414.5 | 238.0 | 968.8 | 961.7 | +| FP16 | 2 | 1554.5 | 378.0 | 1022.1 | 1015.2 | +| INT8 | 2 | 1414.5 | 238.0 | 971.2 | 964.1 | +| FP16 | 3 | 1554.5 | 378.0 | 1017.6 | 1010.5 | +| INT8 | 3 | 1414.5 | 238.0 | 968.8 | 961.8 | + +The device difference is consistently 140 MiB: 37.0% of the post-initialization +increment and 9.0% of the warm device-wide footprint. Host RSS is 4.8–5.0% lower. +This makes batch 64 a local memory-saving candidate with nearly tied measured +latency. The same device-wide/WSL accounting and transient-peak limitations apply; +without trained quality it cannot satisfy the joint production acceptance rule. + +All twelve reports passed. Independent checks verified engine hashes, finite +outputs and exact `[batch, 100000]` final-output shapes for every trial. No code +change accompanies this research increment. Ignored `tmp-100k-batch-deployment/` +retains input provenance, build/inspection, commands, raw timings, memory reports, +outputs and `summary.json`. FP16 engine SHA256 is +`1d74f21747f2b583c37a7f121b4caa3ff1d38ea069d796c1126b3692ec2c4095`; +INT8 engine SHA256 is +`6745ceb732333e22779b016b9ec78d7a09032495f41ec73ef47bdda58f780134`. +The ONNX/checkpoint hashes remain those of the preceding 100,000-class study. +These bounded local results support moving to reproducible target-runner and +joint-quality qualification rather than claiming laptop speed gains or extending +the sweep to a million classes. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 5fffbb7..41d41df 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -3,7 +3,7 @@ The goal remains **incomplete**. There are working native INT8 training and calibrated INT8 inference paths, but the required benefit on the intended deployment regimes has not been established. The latest completed comparison is -recorded in [the 100,000-class TensorRT capacity study](benchmarks.md#hundred-thousand-class-tensorrt-capacity-comparison). +recorded in [the larger-batch 100,000-class study](benchmarks.md#larger-batch-hundred-thousand-class-deployment). The acceptance principle is a joint trade-off: the user tolerates metric losses of a few percentage points when accompanied by a substantial inference speed/cost @@ -32,8 +32,14 @@ show a 130 MiB lower device snapshot: 40.6% of the increment after CUDA initialization and 8.7% of the warm device-wide footprint, with essentially unchanged host RSS. This supplies stronger incremental memory evidence, but the random head cannot qualify trained quality and snapshots do not measure total -GPU peaks. Hierarchical large-class deployment, larger batches and target -hardware remain outstanding. +GPU peaks. A subsequent rebuild for batches 32/64 leaves INT8 2.2–4.5% slower +at batch 32 but effectively tied at batch 64 (paired ratios 0.996–1.011). At +batch 64 it saves 140 MiB in device snapshots: 37.0% of the post-initialization +increment and 9.0% of the warm device-wide footprint, with 4.8–5.0% lower host +RSS. This identifies a local memory-saving candidate without a stable speedup; +trained quality, hierarchical large-class deployment and target hardware remain +outstanding. Prioritize composed target-runner and joint-quality qualification +over further laptop capacity sweeps. ## Current evidence From 413fa7f165b4605a4c3d3da2d747fde5678ec93a Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 21:46:48 +0200 Subject: [PATCH 097/155] feat: compose TensorRT quality and resource evaluation --- dev/benchmarks/README.md | 52 +++++ dev/benchmarks/tensorrt_deployment.py | 247 ++++++++++++++++++++ docs/benchmarks.md | 49 ++++ docs/quantization-status.md | 12 +- tests/test_benchmark_tensorrt_deployment.py | 174 ++++++++++++++ 5 files changed, 532 insertions(+), 2 deletions(-) create mode 100644 dev/benchmarks/tensorrt_deployment.py create mode 100644 tests/test_benchmark_tensorrt_deployment.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 56cd84c..2c67f01 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1555,3 +1555,55 @@ These snapshots do **not** measure total transient GPU allocation peaks. Importi PyTorch for CUDA/IO management also affects this process's footprint. Use paired latency and full `mini_metrics` quality evaluation separately before accepting a memory/quality trade-off. CPU/WSL evidence does not establish target GPU results. + +### Composed TensorRT deployment evaluation + +`dev.benchmarks.tensorrt_deployment` joins the maintained held-out collector and +`mini_metrics` comparison with build inspection, fresh-process paired latency and +isolated single-engine memory trials. Prepare the two engines on the intended +hardware with `dev.benchmarks.tensorrt_build`, then supply its **build directories** +containing `model.engine`, `report.json`, and `layers.json`: + +```bash +CUDA_VISIBLE_DEVICES=0 OMP_NUM_THREADS=1 python -m dev.benchmarks.tensorrt_deployment \ + --baseline-build results/fp16 --candidate-build results/int8 \ + --manifest heldout/manifest.json --inputs heldout/batch-00000.npz \ + --output results/deployment-comparison \ + --metrics-python /path/to/metrics-environment/bin/python \ + --trials 3 --warmup 10 --repeats 31 --memory-runs 20 --threads 1 +``` + +The active interpreter must have explicitly prepared TensorRT/CUDA dependencies; +the separate metrics interpreter must provide `mini_metrics`. Nothing is +installed or synchronized. The collector manifest declares preprocessing, +held-out sample identities, class order, output names and score semantics. Use +representative preprocessed timing inputs with shapes inside both engines' +profiles. Any number of declared classification levels is supported by the +existing collector contract; there is no backbone/head name allowlist. + +Each build's engine and layer-inspection hashes must match its successful build +report. Quality collection runs first in fresh processes, followed by all five +metrics. Engine bytes, manifest bytes, ordered consumed batches and runtime +identity must agree before resource measurement begins. Each resource trial runs +adjacent paired latency in a new process, then one fresh memory process per +engine; trial order alternates. Resource input and engine hashes are checked +against the declared artifacts. `--device` and optional `--pinned` apply to both +resource runners. Memory uses the same NPZ inputs as timing and profile zero. + +Retain the entire output directory: it contains stage commands/logs, report +hashes, canonical prediction CSVs, quality deltas and undefined metrics, raw +paired timings, memory stages, and `summary.md`. Keep the source build bundles +and input manifests/arrays with those artifacts for reproduction; the comparison +does not copy the source engines or datasets. Failed stages return nonzero and +retain the available reports instead of continuing into a misleading comparison. +The CLI summary can be included in a CI job summary and the directory uploaded +as an artifact; remote GPU orchestration and durable publication are separate. + +`status=evaluated` means the protocol completed, **not** that the candidate passed +a production quality/resource threshold. Read the metric deltas and undefined +values together with measured benefit. The inspection inventory does not itself +prove a speedup or complete integer coverage. Host latency excludes preprocessing +and loading, and both engines coexist during timing. Memory comes from separate +processes; device-wide snapshots are not per-process or transient peak readings. +Use quiescent target hardware and record deployment conditions before making +HPC, desktop/Spark or edge acceptance claims. diff --git a/dev/benchmarks/tensorrt_deployment.py b/dev/benchmarks/tensorrt_deployment.py new file mode 100644 index 0000000..2cd55e5 --- /dev/null +++ b/dev/benchmarks/tensorrt_deployment.py @@ -0,0 +1,247 @@ +"""Compose TensorRT held-out quality, inspection, paired latency and isolated memory.""" + +import json +import os +import subprocess +import sys +from argparse import ArgumentParser +from collections import Counter +from pathlib import Path + +from .inference_pair import run_pair +from .onnx_inference import file_hash + + +def build_bundle(path): + """Bind retained inspection to the exact engine measured by every stage.""" + path = Path(path).resolve(strict=True) + report = json.loads((path / "report.json").read_text()) + engine, layers = path / "model.engine", path / "layers.json" + if report["status"] != "passed" or file_hash(engine) != report["engine"]["sha256"]: + raise ValueError("Build report must describe the exact successful engine") + if file_hash(layers) != report["engine"]["layers_sha256"]: + raise ValueError("Layer inspection differs from the build report") + inspection = json.loads(layers.read_text())["Layers"] + return engine, { + "directory": str(path), + "report_sha256": file_hash(path / "report.json"), + "engine": report["engine"], + "settings": report["settings"], + "weight_types": dict(Counter(layer["Weights"]["Type"] for layer in inspection if "Weights" in layer)), + "gemms": [layer for layer in inspection if layer.get("LayerType") == "gemm"], + } + + +def evaluate( + baseline_build, + candidate_build, + manifest, + inputs, + output, + trials=3, + warmup=10, + repeats=31, + memory_runs=20, + threads=1, + device=0, + pinned=False, + metrics_python=sys.executable, +): + if min(trials, repeats, memory_runs, threads) < 1 or min(warmup, device) < 0: + raise ValueError("Require positive trials/repeats/memory_runs/threads and nonnegative warmup/device") + manifest, inputs = Path(manifest).resolve(strict=True), Path(inputs).resolve(strict=True) + output = Path(output).resolve() + output.mkdir(parents=True, exist_ok=False) + report = { + "schema_version": 1, + "status": "running", + "runner_sha256": file_hash(__file__), + "manifest": {"path": str(manifest), "sha256": file_hash(manifest)}, + "inputs": {"path": str(inputs), "sha256": file_hash(inputs)}, + "settings": { + "trials": trials, + "warmup": warmup, + "repeats": repeats, + "memory_runs": memory_runs, + "threads": threads, + "device": device, + "pinned_host_io": pinned, + }, + "builds": {}, + "stages": [], + "trials": [], + "scope": ( + "Paired held-out mini_metrics quality, build-bound inspection, adjacent paired host latency and isolated memory. " + "Evaluated is not production acceptance; inspect all five metrics, undefined values and resource trade-offs. " + "Latency excludes preprocessing/loading; device memory is device-wide snapshots, not per-process or transient peaks. " + "Run on quiescent target hardware; no automatic thermal control or disk-cache eviction." + ), + } + + def save(): + (output / "report.json").write_text(json.dumps(report, indent=2) + "\n") + + def child(name, module, arguments): + command = [sys.executable, "-m", f"dev.benchmarks.{module}", *map(str, arguments), "--output", str(output / name)] + stage = {"name": name, "command": command, "status": "running", "log": f"{name}.log"} + report["stages"].append(stage) + save() + with (output / stage["log"]).open("w") as log: + process = subprocess.run( + command, + cwd=Path(__file__).resolve().parents[2], + env={**os.environ, "PYTHONHASHSEED": "0", "OMP_NUM_THREADS": str(threads)}, + stdout=log, + stderr=subprocess.STDOUT, + check=False, + ) + stage["returncode"] = process.returncode + if process.returncode: + raise RuntimeError(f"{name} failed; see {output / stage['log']}") + result = json.loads((output / name / "report.json").read_text()) + if result["status"] != "passed": + raise RuntimeError(f"Unexpected {name} status: {result['status']}") + stage.update(status="passed", report_sha256=file_hash(output / name / "report.json")) + return result + + def identity(role, engine_info): + if engine_info["sha256"] != report["builds"][role]["engine"]["sha256"]: + raise ValueError(f"{role} engine changed between build and evaluation stages") + + def resource_inputs(result): + if result["inputs"]["sha256"] != report["inputs"]["sha256"]: + raise ValueError("Resource inputs changed between stages") + + try: + report["phase"] = "inspection" + engines = {} + for role, path in (("baseline", baseline_build), ("candidate", candidate_build)): + engines[role], report["builds"][role] = build_bundle(path) + report["phase"] = "quality" + save() + runtime = {"backend": "tensorrt", "device": device} + quality = run_pair(manifest, engines["baseline"], engines["candidate"], output / "quality", runtime, runtime, metrics_python) + report["quality"] = { + "report_sha256": file_hash(output / "quality/report.json"), + "levels": quality["levels"], + "undefined_metrics": quality["undefined_metrics"], + "models": quality["models"], + } + collectors = {} + for role in engines: + result = json.loads((output / f"quality/{role}/report.json").read_text()) + if len(result["model_files"]) != 1: + raise ValueError("Require one TensorRT engine per collector") + identity(role, result["model_files"][0]) + if result["manifest"]["sha256"] != report["manifest"]["sha256"]: + raise ValueError("Quality manifest changed between stages") + collectors[role] = result + if collectors["baseline"]["batches"] != collectors["candidate"]["batches"]: + raise ValueError("Quality input batches differ between engines") + if collectors["baseline"]["runtime"] != collectors["candidate"]["runtime"]: + raise ValueError("Quality runtime environments differ") + report["phase"] = "resources" + save() + flags = ["--device", device, *(["--pinned"] if pinned else [])] + for trial in range(trials): + record = {"trial": trial} + report["trials"].append(record) + latency = child( + f"latency-{trial}", + "tensorrt_pair", + [ + "--baseline", + engines["baseline"], + "--candidate", + engines["candidate"], + "--inputs", + inputs, + "--warmup", + warmup, + "--repeats", + repeats, + *flags, + *(["--reverse"] if trial % 2 else []), + ], + ) + resource_inputs(latency) + for role in engines: + identity(role, latency["models"][role]) + record["latency"] = latency["summary"] + memory = {} + order = ["baseline", "candidate"] if trial % 2 == 0 else ["candidate", "baseline"] + record["memory_order"] = order + for role in order: + result = child( + f"memory-{trial}-{role}", + "tensorrt_memory", + ["--engine", engines[role], "--inputs", inputs, "--runs", memory_runs, "--threads", threads, *flags], + ) + resource_inputs(result) + identity(role, result["engine"]) + memory[role] = result + record.setdefault("memory", {})[role] = result["memory"] + a, b = memory["baseline"], memory["candidate"] + if any(a[key] != b[key] for key in ("settings", "versions", "environment")): + raise ValueError("Memory trial environments or settings differ") + for value in memory.values(): + if any(value["versions"][key] != latency["versions"][key] for key in ("torch", "tensorrt", "numpy")): + raise ValueError("Latency and memory runtime versions differ") + if value["environment"]["gpu"] != latency["environment"]["gpu"]: + raise ValueError("Latency and memory GPU identities differ") + if any(latency["versions"][key] != collectors["baseline"]["runtime"][key] for key in ("tensorrt", "torch")): + raise ValueError("Quality and resource runtime versions differ") + if latency["environment"]["gpu"] != collectors["baseline"]["runtime"]["gpu"]: + raise ValueError("Quality and resource GPU identities differ") + save() + lines = [ + "# TensorRT deployment comparison", + "", + report["scope"], + "", + "| Trial | Paired latency ratio | Baseline warm device bytes | Candidate warm device bytes | " + "Baseline host RSS bytes | Candidate host RSS bytes |", + "| --- | ---: | ---: | ---: | ---: | ---: |", + ] + for trial in report["trials"]: + a, b = [trial["memory"][role]["warm"] for role in ("baseline", "candidate")] + lines.append( + f"| {trial['trial']} | {trial['latency']['median_paired_ratio']:.4f} | " + f"{a['device_used_bytes']} | {b['device_used_bytes']} | " + f"{a['host']['resident_bytes']} | {b['host']['resident_bytes']} |" + ) + lines.extend( + [ + "", + "Latency ratios are candidate / baseline. Device readings include shared/runtime overhead; " + "inspect initialization snapshots in report.json.", + "", + (output / "quality/summary.md").read_text(), + ] + ) + (output / "summary.md").write_text("\n".join(lines) + "\n") + report.update(status="evaluated", phase="complete") + except BaseException as error: + report.update(status="failed", error=f"{type(error).__name__}: {error}") + for stage in report["stages"]: + if stage["status"] == "running": + stage["status"] = "failed" + raise + finally: + save() + return report + + +def main(): + parser = ArgumentParser(description=__doc__) + for name in ("baseline-build", "candidate-build", "manifest", "inputs", "output"): + parser.add_argument("--" + name, type=Path, required=True) + for name, default in (("trials", 3), ("warmup", 10), ("repeats", 31), ("memory-runs", 20), ("threads", 1), ("device", 0)): + parser.add_argument("--" + name, type=int, default=default) + parser.add_argument("--pinned", action="store_true") + parser.add_argument("--metrics-python", default=sys.executable) + evaluate(**vars(parser.parse_args())) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index cf8811b..36172a6 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3442,3 +3442,52 @@ The ONNX/checkpoint hashes remain those of the preceding 100,000-class study. These bounded local results support moving to reproducible target-runner and joint-quality qualification rather than claiming laptop speed gains or extending the sweep to a million classes. + +### Composed TensorRT deployment evaluation + +`dev.benchmarks.tensorrt_deployment` joins the existing TensorRT held-out collector +and paired `mini_metrics` evaluator with build-bound inspection, fresh-process +adjacent paired latency and single-engine memory snapshots. It accepts two +maintained build bundles and arbitrary declared output levels; it has no +backbone/head allowlist. Engine and inspection hashes must agree with each +successful build report. The consumed engine, manifest and ordered batch +identities are checked through quality collection, and resource stages must +consume the same engine bytes and declared timing inputs. Reports retain +runtime/environment identity, all five metric deltas, undefined values, stage +commands, logs and hashes. Failures preserve the available evidence. + +The [command documentation](../dev/benchmarks/README.md#composed-tensorrt-deployment-evaluation) +provides the target-runner invocation. `summary.md` presents quality and resource +readings together, but `status=evaluated` is not an automatic production gate. +Build bundles and original manifests/arrays must be retained for reproduction; +this command does not install runtimes or copy source engines/datasets. Remote +runner orchestration and durable publication remain separate work. + +Real smoke runs used the previously trained native INT8 Blair checkpoints' +materialized FP16 and calibrated INT8 TensorRT engines, both flat and +hierarchical, with all 912 held-out images. Each composed run completed quality, +one paired timing process and two isolated memory processes. The smoke settings +were one warmup, three paired samples and two memory executions. CPU correctness +tests overlapped, so **these resource readings are excluded from performance +claims**. Both build/inspection and cross-stage identity checks passed, with no +undefined quality metrics. + +| Head / level | Macro-F1 delta pp | Macro-Recall delta pp | Macro-Precision delta pp | Coverage delta pp | Theil U delta ×100 | +| --- | ---: | ---: | ---: | ---: | ---: | +| Flat leaf | -0.3030 | -0.2660 | -0.7119 | 0 | -0.6427 | +| Hierarchical leaf | +0.4091 | +0.8540 | -2.2414 | 0 | +0.1912 | +| Hierarchical parent | -0.1056 | +0.1340 | -0.3391 | 0 | -0.4782 | + +Deltas are candidate minus baseline. All predicted labels match the retained +previous collection. Three of the four prediction CSVs are byte-identical; +the hierarchical baseline differs only in two confidence values, by at most +4.278e-12. No universal bitwise floating-output promise follows. These verify +composition using existing trained artifacts, not new training improvements. +Reports, full collector/metric outputs, commands and smoke resource outputs remain +under ignored `tmp-composed-trt-{flat,hierarchical}-smoke/`. + +Validation passed static checks, all eight focused orchestration tests, and the +full CPU-default suite: 521 passed, 162 skipped and the known EMA expected failure. +The two real GPU smoke runs above additionally exercised full held-out collection, +separate `mini_metrics` evaluation, paired latency and fresh-process memory stages. +No new production performance claim accompanies this harness milestone. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index 41d41df..ad3b9eb 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -41,6 +41,14 @@ trained quality, hierarchical large-class deployment and target hardware remain outstanding. Prioritize composed target-runner and joint-quality qualification over further laptop capacity sweeps. +The [composed TensorRT deployment command](benchmarks.md#composed-tensorrt-deployment-evaluation) +now binds inspection, held-out quality, adjacent paired timing and isolated memory +into one retained report. Real flat/hierarchical Blair smoke runs passed all +stages and all five metrics across 912 held-out images; their resource readings +are excluded because CPU correctness checks overlapped. The command is ready for +target-runner qualification, while remote CI orchestration, durable reporting and +production acceptance remain unfinished. + ## Current evidence | Workstream | Verified locally | What remains unproven | @@ -48,9 +56,9 @@ over further laptop capacity sweeps. | Native PyTorch INT8 training | INT8 Linear weights and saved inputs, normalized symmetric flat/hierarchical heads, optimizer/AMP/checkpoint regressions; maintained 100k-class BF16 capacity comparison: full-model peak 17.4% lower; corrected frozen probe peak 6.6% lower after releasing an obsolete floating parameter; mixed timing; bounded preparation and initialization | Benefit on A40/A100/B300-class hardware; representative end-to-end training gains; integer convolution training; distributed QT | | Native QT ONNX export | Generic integer-forward export; full Blair predictions and five metrics preserved on the tested hybrid CUDA/CPU path, with strict score differences | Integer head execution on GPU: CUDA falls back to CPU for MatMulInteger; TensorRT rejects the native representation | | QT checkpoint to calibrated GPU deployment | Explicit materialization of the matched trained INT8 checkpoints, training-only calibration, TensorRT INT8 convolution/head execution, and full Blair comparison against native and FP16 baselines | Broader configuration/large-head qualification; target-machine quality and cost/runtime-memory benefit; exact native dynamic quantization is not preserved | -| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, isolated memory snapshots, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16; target desktop/Spark results; composed continuous orchestration | +| Calibrated TensorRT inference | Both representative heads execute 170 INT8 convolutions and two INT8 head GEMMs; full 912-image Blair evaluation; maintained input preparation, calibration, build/inspection/smoke, paired timing, isolated memory snapshots, full-dataset collection and paired quality commands | Significant speed/runtime-memory benefit against FP16 on target desktop/Spark; continuous runner integration | | CPU/edge inference | Full Blair metrics and isolated process-memory/timing on x86; unsigned CPU recipe executes 170 integer convolutions and two head GEMMs, with 44–54% lower warm batch-one latency and 45–48% lower resident memory in three trials per head | Raspberry Pi/ARM numerical behavior, sustained latency, throughput, process memory and deployment packaging; larger-class qualification | -| Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Latest TensorRT/paired-engine experiments in maintained commands and continuous profiles; durable cross-device result history and acceptance gates | +| Continuous validation/reporting | CPU safeguards, optional GPU training workflow, visible job summaries and 90-day artifacts | Remote execution of the composed TensorRT deployment command; durable cross-device result history and acceptance gates | Native training currently leaves convolutions, gradients and optimizer states floating. Calibrated TensorRT inference now also has an explicit route from trained diff --git a/tests/test_benchmark_tensorrt_deployment.py b/tests/test_benchmark_tensorrt_deployment.py new file mode 100644 index 0000000..77e5c51 --- /dev/null +++ b/tests/test_benchmark_tensorrt_deployment.py @@ -0,0 +1,174 @@ +import json +import subprocess +import sys +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from dev.benchmarks import tensorrt_deployment as deployment +from dev.benchmarks.onnx_inference import file_hash + + +@pytest.fixture +def pipeline(tmp_path, monkeypatch): + builds = [] + for role in ("baseline", "candidate"): + path = tmp_path / role + path.mkdir() + (path / "model.engine").write_bytes(role.encode()) + (path / "layers.json").write_text(json.dumps({"Layers": []})) + (path / "report.json").write_text( + json.dumps( + { + "status": "passed", + "engine": {"sha256": file_hash(path / "model.engine"), "layers_sha256": file_hash(path / "layers.json")}, + "settings": {}, + } + ) + ) + builds.append(path) + manifest, inputs = tmp_path / "manifest.json", tmp_path / "inputs.npz" + manifest.write_text("{}") + inputs.write_bytes(b"input payload") + versions = {"tensorrt": "test", "torch": "test", "numpy": "test"} + runtime = {"tensorrt": "test", "torch": "test", "gpu": "test"} + calls = [] + + def quality(manifest, baseline, candidate, output, *args): + output.mkdir() + result = {"levels": [], "models": {}, "undefined_metrics": ["retained undefined metric"]} + (output / "report.json").write_text(json.dumps(result)) + (output / "summary.md").write_text("undefined metric remains visible") + for role, engine in [("baseline", baseline), ("candidate", candidate)]: + (output / role).mkdir() + (output / role / "report.json").write_text( + json.dumps( + { + "model_files": [{"sha256": file_hash(engine)}], + "manifest": {"sha256": file_hash(manifest)}, + "batches": [{"sha256": "same input"}], + "runtime": runtime, + } + ) + ) + return result + + def child(command, **kwargs): + calls.append(command) + out = Path(command[command.index("--output") + 1]) + out.mkdir() + result = { + "status": "passed", + "inputs": {"sha256": file_hash(inputs)}, + "versions": versions, + "environment": {"gpu": "test"}, + "settings": {}, + } + if command[2].endswith("tensorrt_pair"): + result.update( + models={role: {"sha256": file_hash(path / "model.engine")} for role, path in zip(("baseline", "candidate"), builds)}, + summary={"median_paired_ratio": 1.0}, + ) + else: + engine = Path(command[command.index("--engine") + 1]) + result.update(engine={"sha256": file_hash(engine)}, memory={"warm": {"device_used_bytes": 100, "host": {"resident_bytes": 50}}}) + (out / "report.json").write_text(json.dumps(result)) + return SimpleNamespace(returncode=0) + + monkeypatch.setattr(deployment, "run_pair", quality) + monkeypatch.setattr(deployment.subprocess, "run", child) + return builds, manifest, inputs, calls, quality, child + + +def test_pipeline_alternates_memory_and_retains_quality(pipeline, tmp_path): + builds, manifest, inputs, calls, _, _ = pipeline + output = tmp_path / "result" + result = deployment.evaluate(*builds, manifest, inputs, output, trials=2) + assert result["status"] == "evaluated" + assert result["quality"]["undefined_metrics"] + assert "undefined" in (output / "summary.md").read_text() + assert [stage["name"] for stage in result["stages"]] == [ + "latency-0", + "memory-0-baseline", + "memory-0-candidate", + "latency-1", + "memory-1-candidate", + "memory-1-baseline", + ] + assert "--reverse" not in calls[0] and "--reverse" in calls[3] + with pytest.raises(FileExistsError): + deployment.evaluate(*builds, manifest, inputs, output) + + +@pytest.mark.parametrize("artifact", ["model.engine", "layers.json"]) +def test_mismatched_build_artifacts_fail_before_runtime(pipeline, tmp_path, artifact): + builds, manifest, inputs, calls, _, _ = pipeline + (builds[1] / artifact).write_bytes(b"changed") + with pytest.raises(ValueError): + deployment.evaluate(*builds, manifest, inputs, tmp_path / "result") + report = json.loads((tmp_path / "result/report.json").read_text()) + assert report["status"] == "failed" and report["phase"] == "inspection" and not calls + + +@pytest.mark.parametrize("mutation", ["input", "engine", "process"]) +def test_failed_or_changed_resource_stops_pipeline(pipeline, tmp_path, monkeypatch, mutation): + builds, manifest, inputs, calls, _, child = pipeline + + def change(command, **kwargs): + result = child(command, **kwargs) + path = Path(command[command.index("--output") + 1]) / "report.json" + report = json.loads(path.read_text()) + if mutation == "input": + report["inputs"]["sha256"] = "changed" + elif mutation == "engine": + report["models"]["candidate"]["sha256"] = "changed" + else: + return SimpleNamespace(returncode=17) + path.write_text(json.dumps(report)) + return result + + monkeypatch.setattr(deployment.subprocess, "run", change) + with pytest.raises((ValueError, RuntimeError)): + deployment.evaluate(*builds, manifest, inputs, tmp_path / "result") + result = json.loads((tmp_path / "result/report.json").read_text()) + assert result["status"] == "failed" and result["phase"] == "resources" + assert len(calls) == 1 and (tmp_path / "result/quality/report.json").exists() + + +def test_changed_quality_batches_stop_before_resources(pipeline, tmp_path, monkeypatch): + builds, manifest, inputs, calls, quality, _ = pipeline + + def changed(*args): + result = quality(*args) + path = args[3] / "candidate/report.json" + data = json.loads(path.read_text()) + data["batches"][0]["sha256"] = "changed" + path.write_text(json.dumps(data)) + return result + + monkeypatch.setattr(deployment, "run_pair", changed) + with pytest.raises(ValueError, match="input batches differ"): + deployment.evaluate(*builds, manifest, inputs, tmp_path / "result") + assert not calls + assert json.loads((tmp_path / "result/report.json").read_text())["status"] == "failed" + + +def test_help_is_runtime_independent(): + subprocess.run( + [ + sys.executable, + "-c", + """ +import runpy, sys +sys.argv = ['tensorrt_deployment', '--help'] +try: + runpy.run_module('dev.benchmarks.tensorrt_deployment', run_name='__main__') +except SystemExit as error: + assert error.code == 0 +assert 'torch' not in sys.modules and 'tensorrt' not in sys.modules +""", + ], + check=True, + capture_output=True, + ) From 01b68bebd1bb3797c881b7e5d839e4d74fccff19 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 22:01:57 +0200 Subject: [PATCH 098/155] ci: add opt-in TensorRT target deployment workflow --- .github/workflows/tensorrt-deployment.yml | 57 +++++++++++++++ .gitignore | 1 + README.md | 3 + dev/benchmarks/README.md | 58 +++++++++++++++ dev/check-tensorrt-deployment.sh | 43 +++++++++++ docs/benchmarks.md | 44 ++++++++++++ docs/quantization-status.md | 7 +- tests/test_tensorrt_deployment_harness.py | 88 +++++++++++++++++++++++ 8 files changed, 298 insertions(+), 3 deletions(-) create mode 100644 .github/workflows/tensorrt-deployment.yml create mode 100644 dev/check-tensorrt-deployment.sh create mode 100644 tests/test_tensorrt_deployment_harness.py diff --git a/.github/workflows/tensorrt-deployment.yml b/.github/workflows/tensorrt-deployment.yml new file mode 100644 index 0000000..c3d77e8 --- /dev/null +++ b/.github/workflows/tensorrt-deployment.yml @@ -0,0 +1,57 @@ +name: TensorRT deployment evaluation + +on: + workflow_dispatch: + schedule: + - cron: "23 5 * * 5" + +permissions: + contents: read + +concurrency: + group: tensorrt-deployment-${{ vars.TRT_RUNNER_LABEL || 'gpu' }} + cancel-in-progress: false + +jobs: + deployment: + if: github.event_name == 'workflow_dispatch' || vars.ENABLE_TENSORRT_BENCHMARKS == 'true' + runs-on: [self-hosted, linux, "${{ vars.TRT_RUNNER_LABEL || 'gpu' }}"] + timeout-minutes: 60 + env: + BENCHMARK_PYTHON: ${{ vars.TRT_BENCHMARK_PYTHON }} + BENCHMARK_METRICS_PYTHON: ${{ vars.BENCHMARK_METRICS_PYTHON }} + TRT_BASELINE_MODEL: ${{ vars.TRT_BASELINE_MODEL }} + TRT_CANDIDATE_MODEL: ${{ vars.TRT_CANDIDATE_MODEL }} + TRT_INFERENCE_MANIFEST: ${{ vars.TRT_INFERENCE_MANIFEST }} + TRT_INPUTS: ${{ vars.TRT_INPUTS }} + TRT_PROFILES: ${{ vars.TRT_PROFILES }} + TRT_OPTIMIZATION: ${{ vars.TRT_OPTIMIZATION || '1' }} + TRT_WORKSPACE_MIB: ${{ vars.TRT_WORKSPACE_MIB || '1024' }} + BENCHMARK_THREADS: ${{ vars.TRT_THREADS || '1' }} + CUDA_VISIBLE_DEVICES: ${{ vars.TRT_CUDA_VISIBLE_DEVICES || '0' }} + steps: + - uses: actions/checkout@v6 + - name: Build and evaluate on the target + run: bash dev/check-tensorrt-deployment.sh tensorrt-results + - name: Visible quality and resource summary + if: always() + shell: bash + run: | + if [[ -f tensorrt-results/evaluation/summary.md ]]; then + cat tensorrt-results/evaluation/summary.md >> "$GITHUB_STEP_SUMMARY" + else + echo 'TensorRT evaluation did not complete. Inspect the retained status and stage logs.' >> "$GITHUB_STEP_SUMMARY" + if [[ -f tensorrt-results/status.json ]]; then + echo '```json' >> "$GITHUB_STEP_SUMMARY" + cat tensorrt-results/status.json >> "$GITHUB_STEP_SUMMARY" + echo '```' >> "$GITHUB_STEP_SUMMARY" + fi + fi + - name: Retain engines, inspection, quality and resource evidence + if: always() + uses: actions/upload-artifact@v7 + with: + name: tensorrt-deployment-${{ github.run_id }}-${{ github.run_attempt }} + retention-days: 90 + if-no-files-found: error + path: tensorrt-results/ diff --git a/.gitignore b/.gitignore index bb3adfe..25679e3 100644 --- a/.gitignore +++ b/.gitignore @@ -22,6 +22,7 @@ **/__pycache__/** /tmp* /tmp/** +/tensorrt-results/ **.pyspy # Specific cases diff --git a/README.md b/README.md index b0e812c..30f47a8 100644 --- a/README.md +++ b/README.md @@ -127,6 +127,9 @@ Follow the [benchmark results and coverage](docs/benchmarks.md) and The suite progresses from an exact synthetic oracle to MNIST and hierarchical Blair, with separate CPU and GPU profiles, visible summaries, and retained reproduction artifacts. +For configured GPU runners, the opt-in [TensorRT deployment workflow](dev/benchmarks/README.md#opt-in-target-gpu-workflow) +rebuilds engines on the target and reports paired quality, latency and memory. + ## Temporarily unsupported feature EMA (`--ema` / `ema=True`) is currently nonfunctional: classifier caches populated diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 2c67f01..9d20d19 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1607,3 +1607,61 @@ and loading, and both engines coexist during timing. Memory comes from separate processes; device-wide snapshots are not per-process or transient peak readings. Use quiescent target hardware and record deployment conditions before making HPC, desktop/Spark or edge acceptance claims. + +### Opt-in target GPU workflow + +`.github/workflows/tensorrt-deployment.yml` runs the same target command as a +local or manually allocated Linux GPU session: + +```bash +export BENCHMARK_PYTHON=/absolute/path/to/prepared-tensorrt-env/bin/python +export BENCHMARK_METRICS_PYTHON=/absolute/path/to/metrics-env/bin/python +export TRT_BASELINE_MODEL=/absolute/path/to/float/model.onnx +export TRT_CANDIDATE_MODEL=/absolute/path/to/calibrated/model.onnx +export TRT_INFERENCE_MANIFEST=/absolute/path/to/heldout/manifest.json +export TRT_INPUTS=/absolute/path/to/heldout/batch-00000.npz +export TRT_PROFILES=/absolute/path/to/profiles.json +CUDA_VISIBLE_DEVICES=0 bash dev/check-tensorrt-deployment.sh tensorrt-results +``` + +Keep ONNX external-weight files beside each model. Both models must implement +the manifest's class order and score semantics. Supply a calibrated candidate; +this command does not select calibration data or fit a quantizer. It checks +runtime availability, rebuilds both engines on the actual runner, then invokes +the composed evaluator. Both builds allow FP16, disable TF32, share the profile +JSON, and default to optimization level 1 and 1 GiB workspace. Override those last +two settings with `TRT_OPTIMIZATION` and `TRT_WORKSPACE_MIB`. Resource threads +default to one (`BENCHMARK_THREADS`) and visible device index to zero +(`BENCHMARK_DEVICE`). No environment installation, synchronization or deletion +occurs. Relative paths resolve from the repository root. + +The GitHub workflow requires a configured self-hosted Linux GPU runner and these +repository variables: + +| Repository variable | Meaning | +| --- | --- | +| `TRT_BENCHMARK_PYTHON` | Prepared TensorRT/CUDA Python; maps to `BENCHMARK_PYTHON` | +| `BENCHMARK_METRICS_PYTHON` | Prepared `mini_metrics` Python | +| `TRT_BASELINE_MODEL`, `TRT_CANDIDATE_MODEL` | Absolute source ONNX paths | +| `TRT_INFERENCE_MANIFEST`, `TRT_INPUTS`, `TRT_PROFILES` | Absolute held-out contract, resource inputs and build profile paths | +| `TRT_RUNNER_LABEL` | Additional self-hosted runner label; defaults to `gpu` | +| `TRT_CUDA_VISIBLE_DEVICES` | Explicit visible GPU selection; defaults to `0` | +| `TRT_THREADS`, `TRT_OPTIMIZATION`, `TRT_WORKSPACE_MIB` | Optional matching execution/build settings | +| `ENABLE_TENSORRT_BENCHMARKS` | Set exactly `true` to enable weekly scheduled execution | + +Prepared environments and source artifacts should live outside the checkout, +which the checkout action can clean between runs. Use quiescent, allocated target +hardware; the workflow's concurrency group prevents overlapping executions of +this workflow with the same runner label, but does not coordinate unrelated jobs. +Manual dispatch is available independently of the schedule opt-in. GitHub requires +the workflow on the default branch for manual/scheduled activation; scheduled +start times can be delayed. See [GitHub's event documentation](https://docs.github.com/en/actions/reference/workflows-and-actions/events-that-trigger-workflows). + +Every run retains the repository revision, harness hash, preflight/build/evaluation +logs, failing phase/exit code, rebuilt engines and inspections, predictions, +metrics, paired timings and memory reports. A completed comparison appears in +the job summary. Failures instead show their status and point to retained logs. +Artifacts are named by run ID and attempt and retained for 90 days. This supplies +visible per-run evidence, **not durable cross-run history**; permanent publication +and target acceptance still require additional work. Source models, datasets and +prepared environments are not uploaded by the workflow. diff --git a/dev/check-tensorrt-deployment.sh b/dev/check-tensorrt-deployment.sh new file mode 100644 index 0000000..da81bff --- /dev/null +++ b/dev/check-tensorrt-deployment.sh @@ -0,0 +1,43 @@ +#!/usr/bin/env bash +# Build engines on the target and evaluate them using explicitly prepared environments. +set -euo pipefail +cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." +results="${1:?Supply a new results directory}" +if [[ -e "$results" || -L "$results" ]]; then + echo 'Results directory must be new.' >&2 + exit 2 +fi +mkdir -p -- "$results" +phase=configuration +trap 'code=$?; printf "{\"phase\":\"%s\",\"exit_code\":%d}\n" "$phase" "$code" > "$results/status.json"' EXIT +: "${BENCHMARK_PYTHON:?Set BENCHMARK_PYTHON to an explicitly prepared TensorRT/CUDA Python}" +: "${BENCHMARK_METRICS_PYTHON:?Set BENCHMARK_METRICS_PYTHON to an environment with mini_metrics}" +: "${TRT_BASELINE_MODEL:?Set TRT_BASELINE_MODEL to the floating ONNX model}" +: "${TRT_CANDIDATE_MODEL:?Set TRT_CANDIDATE_MODEL to the calibrated candidate ONNX model}" +: "${TRT_INFERENCE_MANIFEST:?Set TRT_INFERENCE_MANIFEST to the held-out inference manifest}" +: "${TRT_INPUTS:?Set TRT_INPUTS to representative named preprocessed NPZ inputs}" +: "${TRT_PROFILES:?Set TRT_PROFILES to the shared TensorRT profile JSON}" +export OMP_NUM_THREADS="${BENCHMARK_THREADS:-1}" +export PYTHONHASHSEED=0 +phase=preflight +git rev-parse HEAD > "$results/revision.txt" +sha256sum -- dev/check-tensorrt-deployment.sh > "$results/harness.sha256" +"$BENCHMARK_PYTHON" -c 'import onnx, numpy, tensorrt, torch; assert torch.cuda.is_available(), "CUDA must be available"' > "$results/preflight.log" 2>&1 +"$BENCHMARK_METRICS_PYTHON" -c 'from mini_metrics.data import MetricDF; from mini_metrics.metrics import MacroF1, evaluate_file' >> "$results/preflight.log" 2>&1 +for role in baseline candidate; do + phase="build-$role" + model="$TRT_BASELINE_MODEL" + if [[ "$role" == candidate ]]; then model="$TRT_CANDIDATE_MODEL"; fi + "$BENCHMARK_PYTHON" -m dev.benchmarks.tensorrt_build \ + --model "$model" --inputs "$TRT_INPUTS" --profiles "$TRT_PROFILES" \ + --output "$results/$role" --fp16 --device "${BENCHMARK_DEVICE:-0}" \ + --optimization "${TRT_OPTIMIZATION:-1}" --workspace-mib "${TRT_WORKSPACE_MIB:-1024}" \ + > "$results/build-$role.log" 2>&1 +done +phase=evaluation +"$BENCHMARK_PYTHON" -m dev.benchmarks.tensorrt_deployment \ + --baseline-build "$results/baseline" --candidate-build "$results/candidate" \ + --manifest "$TRT_INFERENCE_MANIFEST" --inputs "$TRT_INPUTS" --output "$results/evaluation" \ + --metrics-python "$BENCHMARK_METRICS_PYTHON" --threads "$OMP_NUM_THREADS" \ + --device "${BENCHMARK_DEVICE:-0}" > "$results/evaluation.log" 2>&1 +phase=complete diff --git a/docs/benchmarks.md b/docs/benchmarks.md index 36172a6..c65f64e 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3491,3 +3491,47 @@ full CPU-default suite: 521 passed, 162 skipped and the known EMA expected failu The two real GPU smoke runs above additionally exercised full held-out collection, separate `mini_metrics` evaluation, paired latency and fresh-process memory stages. No new production performance claim accompanies this harness milestone. + +### Target GPU workflow entry point + +`dev/check-tensorrt-deployment.sh` now runs explicit environment preflight, +matching FP16-enabled baseline/candidate builds on the target, and the composed +quality/resource evaluator. It records the source revision, harness hash and +failure phase/exit code, preserves stage logs and rejects existing output paths. +Its configuration uses quoted arguments rather than evaluating command strings; +paths with spaces and shell metacharacters remain literal. + +The opt-in `tensorrt-deployment.yml` workflow uses that same command on a +configured self-hosted Linux GPU runner. Manual invocation and an explicitly +enabled weekly schedule produce job summaries and 90-day artifacts including +engines, inspection, quality and resource evidence. It uses prepared environments +without dependency synchronization, has read-only repository permissions, and +rebuilds engines from configured ONNX sources instead of reusing laptop binaries. +See the [runner setup](../dev/benchmarks/README.md#opt-in-target-gpu-workflow). +This is a runner handoff, not evidence that unavailable hardware has passed, and +90-day artifacts do not solve durable result publication. + +The exact shared command completed locally on the retained trained flat Blair +ONNX sources: two fresh engine builds, 912-image paired quality, three paired +latency processes and six isolated memory processes. Both builds used matching +1/8/8 profiles, FP16 enabled, TF32 disabled, optimization level 1 and 1 GiB +workspace. The top-level status records `phase=complete, exit_code=0` and the +composed report records `evaluated`. All five metrics are defined; candidate minus +baseline changes are -0.2768 pp Macro-F1, -0.4905 pp Macro-Recall, -0.0246 pp +Macro-Precision, zero Coverage change and -0.5290 for Theil's U multiplied by 100. +These rebuilt engines differ from earlier experiments; candidate FP16 is now +allowed. No inference cost claim is made because CPU checks overlapped this +correctness run. Full evidence is retained under ignored +`tmp-target-trt-harness-smoke/`. + +Shell syntax, workflow YAML structure and embedded shell syntax passed local +checks. Six harness tests cover build-before-evaluation ordering, matching flags, +literal paths, output preservation, missing configuration and failure propagation +from both environments, candidate build and evaluation. No remote GitHub job was +launched. Default-branch activation, configured runner execution and durable +publication remain unverified; local parser checks are not a GitHub execution test. + +The full CPU-default suite passed: 527 tests passed, 162 skipped and the known +EMA expected failure, alongside the local real-data GPU workflow-command run. +Static repository checks passed. The workflow itself has only local YAML/shell +validation; execution by GitHub Actions on the intended runner is still required. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index ad3b9eb..ef08b0d 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -45,9 +45,10 @@ The [composed TensorRT deployment command](benchmarks.md#composed-tensorrt-deplo now binds inspection, held-out quality, adjacent paired timing and isolated memory into one retained report. Real flat/hierarchical Blair smoke runs passed all stages and all five metrics across 912 held-out images; their resource readings -are excluded because CPU correctness checks overlapped. The command is ready for -target-runner qualification, while remote CI orchestration, durable reporting and -production acceptance remain unfinished. +are excluded because CPU correctness checks overlapped. An opt-in target GPU workflow now rebuilds both engines and invokes this command +using explicitly configured environments, ONNX sources and held-out inputs. +Its job summary and 90-day artifacts provide per-run visibility; remote execution, +durable cross-run reporting and production acceptance remain unverified. ## Current evidence diff --git a/tests/test_tensorrt_deployment_harness.py b/tests/test_tensorrt_deployment_harness.py new file mode 100644 index 0000000..55c6f53 --- /dev/null +++ b/tests/test_tensorrt_deployment_harness.py @@ -0,0 +1,88 @@ +import json +import os +import subprocess +import sys +from pathlib import Path + +import pytest + + +@pytest.fixture +def environment(tmp_path): + executable = tmp_path / "prepared python" + executable.write_text( + f"#!{sys.executable}\n" + + """ +import json, os, sys +from pathlib import Path +args = sys.argv[1:] +with Path(os.environ['HARNESS_COMMANDS']).open('a') as log: + log.write(json.dumps(args) + '\\n') +if args[0] == '-c': + stage = 'metrics' if 'mini_metrics' in args[1] else 'preflight' +else: + stage = 'candidate' if '--model' in args and args[args.index('--model')+1] == os.environ['TRT_CANDIDATE_MODEL'] else 'baseline' + if args[1].endswith('tensorrt_deployment'): stage = 'evaluation' +if stage == os.environ.get('FAIL_STAGE'): + print('requested stage failure', file=sys.stderr) + raise SystemExit(19) +""" + ) + executable.chmod(0o755) + env = { + **os.environ, + "BENCHMARK_PYTHON": str(executable), + "BENCHMARK_METRICS_PYTHON": str(executable), + "TRT_BASELINE_MODEL": "float model.onnx", + "TRT_CANDIDATE_MODEL": "candidate $(not-a-command).onnx", + "TRT_INFERENCE_MANIFEST": "held out/manifest.json", + "TRT_INPUTS": "held out/inputs.npz", + "TRT_PROFILES": "profiles with spaces.json", + "HARNESS_COMMANDS": str(tmp_path / "commands.jsonl"), + } + return env + + +def invoke(tmp_path, environment): + output = tmp_path / "new results" + result = subprocess.run(["bash", "dev/check-tensorrt-deployment.sh", str(output)], env=environment, capture_output=True, text=True) + commands = Path(environment["HARNESS_COMMANDS"]) + calls = [json.loads(line) for line in commands.read_text().splitlines()] if commands.exists() else [] + return result, output, calls + + +def test_harness_builds_before_evaluation_and_preserves_literal_paths(tmp_path, environment): + result, output, calls = invoke(tmp_path, environment) + assert result.returncode == 0, result.stderr + assert len(calls) == 5 + assert [call[1] for call in calls[2:]] == ["dev.benchmarks.tensorrt_build"] * 2 + ["dev.benchmarks.tensorrt_deployment"] + for role, call in zip(("BASELINE", "CANDIDATE"), calls[2:4], strict=True): + assert call[call.index("--model") + 1] == environment[f"TRT_{role}_MODEL"] + assert call[call.index("--profiles") + 1] == environment["TRT_PROFILES"] + assert "--fp16" in call and "--tf32" not in call + assert calls[4][calls[4].index("--metrics-python") + 1] == environment["BENCHMARK_METRICS_PYTHON"] + assert json.loads((output / "status.json").read_text()) == {"phase": "complete", "exit_code": 0} + assert (output / "revision.txt").read_text().strip() + original = (output / "status.json").read_bytes() + again, _, repeated = invoke(tmp_path, environment) + assert again.returncode == 2 and repeated == calls and (output / "status.json").read_bytes() == original + + +@pytest.mark.parametrize( + "stage,phase,count", + [("preflight", "preflight", 1), ("metrics", "preflight", 2), ("candidate", "build-candidate", 4), ("evaluation", "evaluation", 5)], +) +def test_harness_retains_failed_phase_and_stops(tmp_path, environment, stage, phase, count): + environment["FAIL_STAGE"] = stage + result, output, calls = invoke(tmp_path, environment) + assert result.returncode == 19 and len(calls) == count + assert json.loads((output / "status.json").read_text()) == {"phase": phase, "exit_code": 19} + assert any("requested stage failure" in path.read_text() for path in output.glob("*.log")) + + +def test_missing_configuration_stops_before_importing_runtimes(tmp_path, environment): + del environment["TRT_PROFILES"] + result, output, calls = invoke(tmp_path, environment) + assert result.returncode != 0 and not calls + assert json.loads((output / "status.json").read_text())["phase"] == "configuration" + assert "Set TRT_PROFILES" in result.stderr From 2371b5969313411a9ffb3fed1062ae0b0f80bd88 Mon Sep 17 00:00:00 2001 From: asgersvenning Date: Wed, 9 Sep 2026 22:20:53 +0200 Subject: [PATCH 099/155] feat: add immutable deployment report history --- dev/benchmarks/README.md | 48 ++++ dev/benchmarks/report_history.py | 294 +++++++++++++++++++++++++ docs/benchmarks.md | 38 ++++ docs/quantization-status.md | 8 + tests/test_benchmark_report_history.py | 122 ++++++++++ 5 files changed, 510 insertions(+) create mode 100644 dev/benchmarks/report_history.py create mode 100644 tests/test_benchmark_report_history.py diff --git a/dev/benchmarks/README.md b/dev/benchmarks/README.md index 9d20d19..ca0dad7 100644 --- a/dev/benchmarks/README.md +++ b/dev/benchmarks/README.md @@ -1665,3 +1665,51 @@ Artifacts are named by run ID and attempt and retained for 90 days. This supplie visible per-run evidence, **not durable cross-run history**; permanent publication and target acceptance still require additional work. Source models, datasets and prepared environments are not uploaded by the workflow. + +### Compact report history and local dashboard + +`dev.benchmarks.report_history` archives version-one composed TensorRT reports as +immutable, compact JSON records and renders a standalone HTML history. Use the +repository's supported Python (3.12 or newer); the command itself only needs the +standard library and does not load PyTorch, TensorRT or datasets. + +```bash +.venv/bin/python -m dev.benchmarks.report_history archive \ + --report tensorrt-results/evaluation/report.json --history tmp-report-history \ + --run-id run-123-attempt-1 --revision FULL_SOURCE_COMMIT_HASH \ + --profile 'Blair flat / target GPU / batch 8' \ + --run-url https://github.com/OWNER/REPO/actions/runs/123 \ + --note 'Describe workload and measurement conditions' +.venv/bin/python -m dev.benchmarks.report_history render --history tmp-report-history +``` + +Open `tmp-report-history/index.html`. Every record retains the source report +hash, source revision, input/manifest hashes, matching build settings, aggregate +quality values/deltas, engine identities and resource snapshots. Runtime identity +comes from a retained latency report whose hash must match the evaluation report. +Keep that child report beside the source evaluation report when archiving. +Prediction rows, class/sample labels, local artifact paths and exception text are +not copied into the compact record. Profile/note/run URL are explicitly supplied +publication metadata; review them before hosting. + +Run IDs accept a restricted filename-safe alphabet. Repeating the identical run +is idempotent; changing its evidence or metadata under the same ID fails without +overwriting the record. New records are published with a no-overwrite filesystem +operation. The HTML page is derived and can be regenerated from `records/`. +Retain that directory in persistent storage; creating these files alone is not a +backup or permanent hosting service. Original reproduction artifacts remain +necessary; compact records do not replace engine files, inputs or raw reports. + +Resource comparisons are **excluded by default**. Supply `--performance-valid` +only when the run's conditions justify their use, and describe those conditions +in `--note`. This is a publisher declaration, not an automatic certification. +Failed runs never display eligible performance comparisons. For a preflight/build +failure before an evaluation report exists, pass the shared command's nonzero +`status.json` instead. A successful top-level status alone is insufficient: +archive its detailed evaluation report. Missing quality and undefined metrics +remain visible, and completed evaluation is not displayed as production acceptance. + +The renderer currently supports TensorRT deployment records. CPU/ARM and training +history adapters, persistent hosted storage and publication automation remain +unfinished. No remote uploads occur from either command. Local HTML structure +and escaping have tests; browser rendering still needs visual qualification. diff --git a/dev/benchmarks/report_history.py b/dev/benchmarks/report_history.py new file mode 100644 index 0000000..599f4cd --- /dev/null +++ b/dev/benchmarks/report_history.py @@ -0,0 +1,294 @@ +"""Archive compact TensorRT comparison records and render a standalone HTML history.""" + +import hashlib +import html +import json +import math +import os +import re +import tempfile +from argparse import ArgumentParser +from datetime import UTC, datetime +from pathlib import Path +from urllib.parse import urlsplit + +METRICS = ("f1", "recall", "precision", "coverage", "theilU") +METRIC_NAMES = dict(zip(METRICS, ("Macro-F1", "Macro-Recall", "Macro-Precision", "Coverage", "Theil U"), strict=True)) +IDENTIFIER = re.compile(r"[A-Za-z0-9][A-Za-z0-9._-]{0,127}") +REVISION = re.compile(r"(?:[0-9a-f]{40}|[0-9a-f]{64})") + + +def number(value): + """Undefined metrics stay undefined; malformed/nonfinite values cannot be published.""" + if value is None: + return None + if type(value) not in (float, int) or not math.isfinite(value): + raise ValueError("Expected a finite numeric value or null") + return value + + +def project(source): + """Select aggregate evidence without paths, predictions, labels, or exception text.""" + if "exit_code" in source: + if type(source["exit_code"]) is not int or source["exit_code"] == 0: + raise ValueError("A completed target command requires its evaluation report") + return { + "status": "failed", + "phase": source["phase"], + "exit_code": source["exit_code"], + "engines": {}, + "quality": [], + "resources": [], + } + if source.get("schema_version") != 1 or source.get("status") not in ("evaluated", "failed") or "builds" not in source: + raise ValueError("Expected a version-one composed TensorRT deployment report") + result = {"status": source["status"], "phase": source.get("phase"), "engines": {}, "quality": [], "resources": []} + for role, build in source["builds"].items(): + if role not in ("baseline", "candidate"): + raise ValueError("Unexpected model role") + engine = build["engine"] + result["engines"][role] = {key: engine[key] for key in ("sha256", "bytes", "context_memory_bytes")} + quality = source.get("quality", {}) + for index, level in enumerate(quality.get("levels", [])): + item = {key: level[key] for key in ("name", "samples", "prediction_changes")} + item["metrics"] = {} + for metric in METRICS: + a, b = [number(quality["models"][role]["metrics"][metric][str(index)]) for role in ("baseline", "candidate")] + delta = number(level["candidate_minus_baseline"][metric]) + if a is None or b is None: + if delta is not None: + raise ValueError("Undefined metrics require an undefined delta") + elif delta is None or not math.isclose(delta, b - a, rel_tol=1e-10, abs_tol=1e-12): + raise ValueError("Metric delta disagrees with baseline/candidate values") + item["metrics"][metric] = {"baseline": a, "candidate": b, "delta": delta} + result["quality"].append(item) + for trial in source.get("trials", []): + item = {"trial": trial["trial"], "memory": {}} + if "latency" in trial: + item["latency_ratio"] = number(trial["latency"]["median_paired_ratio"]) + item["latency_seconds"] = {role: number(trial["latency"]["median_seconds"][role]) for role in ("baseline", "candidate")} + for role, snapshots in trial.get("memory", {}).items(): + item["memory"][role] = { + name: { + "device_used_bytes": number(snapshots[name]["device_used_bytes"]), + "host_resident_bytes": number(snapshots[name]["host"]["resident_bytes"]), + } + for name in ("cuda_initialized", "warm") + } + result["resources"].append(item) + if source["status"] == "evaluated": + if set(result["engines"]) != {"baseline", "candidate"} or not result["quality"]: + raise ValueError("Completed comparison requires both engines and quality") + if len(result["resources"]) != source["settings"]["trials"] or not result["resources"]: + raise ValueError("Completed comparison has incomplete resource trials") + if any("latency_ratio" not in trial or set(trial["memory"]) != {"baseline", "candidate"} for trial in result["resources"]): + raise ValueError("Completed comparison has incomplete resource measurements") + return result + + +def runtime_metadata(source, root): + """Use a hash-verified latency report for the actual measured runtime identity.""" + for stage in source.get("stages", []): + name = stage["name"] + if not name.startswith("latency-") or stage["status"] != "passed": + continue + if not IDENTIFIER.fullmatch(name): + raise ValueError("Unsafe stage identifier") + payload = (root / name / "report.json").read_bytes() + if hashlib.sha256(payload).hexdigest() != stage["report_sha256"]: + raise ValueError("Latency evidence changed since evaluation") + child = json.loads(payload) + return { + "versions": {key: child["versions"][key] for key in ("torch", "tensorrt", "numpy")}, + "environment": {key: child["environment"][key] for key in ("gpu", "compute_capability", "platform", "python")}, + } + if source.get("status") == "evaluated": + raise ValueError("Completed comparison requires retained latency runtime evidence") + return None + + +def archive(report, history, run_id, revision, profile, run_url=None, performance_valid=False, note=""): + if not IDENTIFIER.fullmatch(run_id) or not REVISION.fullmatch(revision): + raise ValueError("Require a safe run identifier and full hexadecimal source revision") + if not profile.strip(): + raise ValueError("Supply a descriptive profile") + if run_url is not None: + url = urlsplit(run_url) + if url.scheme != "https" or not url.netloc or url.username or url.password: + raise ValueError("Run URL must be HTTPS without credentials") + payload = Path(report).read_bytes() + source = json.loads(payload) + evidence = project(source) + hardware = runtime_metadata(source, Path(report).parent) + record = { + "schema_version": 1, + "kind": "tensorrt_deployment", + "run_id": run_id, + "revision": revision, + "profile": profile, + "run_url": run_url, + "note": note, + "source_report_sha256": hashlib.sha256(payload).hexdigest(), + "runner_sha256": source.get("runner_sha256"), + "performance_valid": bool(performance_valid and evidence["status"] == "evaluated"), + "evidence": evidence, + "runtime": hardware, + "comparison": { + "inputs_sha256": source.get("inputs", {}).get("sha256"), + "manifest_sha256": source.get("manifest", {}).get("sha256"), + "settings": { + key: value + for key, value in source.get("settings", {}).items() + if key in ("trials", "warmup", "repeats", "memory_runs", "threads", "device", "pinned_host_io") + }, + "build_settings": { + role: { + key: value + for key, value in build.get("settings", {}).items() + if key in ("profiles", "fp16_allowed", "tf32_allowed", "workspace_bytes", "builder_optimization") + } + for role, build in source.get("builds", {}).items() + }, + }, + } + records = Path(history) / "records" + records.mkdir(parents=True, exist_ok=True) + destination = records / f"{run_id}.json" + if destination.exists(): + old = json.loads(destination.read_text()) + record["recorded_at"] = old["recorded_at"] + if old != record: + raise ValueError("Run identity already exists with different evidence or metadata") + return destination + record["recorded_at"] = datetime.now(UTC).isoformat() + encoded = (json.dumps(record, indent=2, allow_nan=False) + "\n").encode() + temporary = None + try: + with tempfile.NamedTemporaryFile(dir=records, suffix=".tmp", delete=False) as stream: + temporary = Path(stream.name) + stream.write(encoded) + stream.flush() + os.fsync(stream.fileno()) + # Publish without overwriting a concurrent writer's immutable record. + os.link(temporary, destination) + finally: + if temporary is not None: + temporary.unlink(missing_ok=True) + return destination + + +def escape(value): + return html.escape(str(value), quote=True) + + +def display(value, scale=1): + return "undefined" if value is None else f"{number(value) * scale:.4f}" + + +def render(history): + history = Path(history) + records = [] + for path in sorted((history / "records").glob("*.json")): + record = json.loads(path.read_text()) + if record.get("schema_version") != 1 or record.get("kind") != "tensorrt_deployment": + raise ValueError(f"Unsupported history record: {path.name}") + if not IDENTIFIER.fullmatch(record["run_id"]) or path.name != record["run_id"] + ".json": + raise ValueError("History filename and run identity differ") + records.append(record) + lines = [ + '', + '', + "Quantization evaluation history", + "", + "

Quantization evaluation history

", + "

Completed evaluation is not production acceptance. Compare quality and resource use on the intended hardware. " + "Device memory readings are device-wide snapshots, not per-process or transient peaks.

", + ] + for record in sorted(records, key=lambda r: (r["recorded_at"], r["run_id"]), reverse=True): + evidence = record["evidence"] + lines.append(f'
') + lines.append(f"

{escape(record['profile'])} — {escape(record['run_id'])}

") + lines.append( + f"

Status: {escape(evidence['status'])}; phase: {escape(evidence['phase'])}. " + f"Recorded: {escape(record['recorded_at'])}.

" + ) + lines.append( + f"

Revision {escape(record['revision'])}. " + f'Download compact record

' + ) + if record.get("run_url"): + url = urlsplit(record["run_url"]) + if url.scheme != "https" or not url.netloc or url.username or url.password: + raise ValueError("Unsafe history run URL") + lines.append(f'

Original workflow run

') + lines.append(f'

{escape(record["note"])}

') + hardware = record.get("runtime") + gpu = hardware["environment"]["gpu"] if hardware else "unrecorded" + lines.append(f"

Measured GPU: {escape(gpu)}. Source report {escape(record['source_report_sha256'])}.

") + for level in evidence["quality"]: + lines.append( + f"

{escape(level['name'])}: {escape(level['samples'])} samples, " + f"{escape(level['prediction_changes'])} changed predictions

" + ) + lines.append("") + for metric in METRICS: + values = level["metrics"][metric] + lines.append( + f"" + f"" + ) + lines.append("
MetricBaselineCandidateDelta ×100
{escape(METRIC_NAMES[metric])}{display(values['baseline'])}{display(values['candidate'])}{display(values['delta'], 100)}
") + if record["performance_valid"] and evidence["status"] == "evaluated": + lines.append("

Resource measurements marked usable by the publisher; this is not independently certified.

") + lines.append( + "" + "" + ) + for trial in evidence["resources"]: + a, b = [trial["memory"][role]["warm"] for role in ("baseline", "candidate")] + values = [ + trial["trial"], + display(trial["latency_ratio"]), + display(a["device_used_bytes"], 1 / 2**20), + display(b["device_used_bytes"], 1 / 2**20), + display(a["host_resident_bytes"], 1 / 2**20), + display(b["host_resident_bytes"], 1 / 2**20), + ] + lines.append("" + "".join(f"" for value in values) + "") + lines.append( + "
TrialPaired latency ratioBaseline device MiBCandidate device MiBBaseline host RSS MiBCandidate host RSS MiB
{escape(value)}

Latency ratios are candidate / baseline; below one is lower. " + "Inspect initialization snapshots in the compact record.

" + ) + else: + lines.append("

Resource readings excluded from performance comparisons.

") + if not evidence["quality"]: + lines.append("

Quality results unavailable.

") + lines.append("
") + if not records: + lines.append("

No archived runs.

") + lines.append("") + history.mkdir(parents=True, exist_ok=True) + (history / "index.html").write_text("\n".join(lines) + "\n") + return history / "index.html" + + +def main(): + parser = ArgumentParser(description=__doc__) + sub = parser.add_subparsers(dest="command", required=True) + add = sub.add_parser("archive") + for name in ("report", "history", "run-id", "revision", "profile"): + add.add_argument("--" + name, required=True) + add.add_argument("--run-url") + add.add_argument("--performance-valid", action="store_true") + add.add_argument("--note", default="") + sub.add_parser("render").add_argument("--history", required=True) + args = vars(parser.parse_args()) + command = args.pop("command") + archive(**args) if command == "archive" else render(**args) + + +if __name__ == "__main__": + main() diff --git a/docs/benchmarks.md b/docs/benchmarks.md index c65f64e..8cfbcd0 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -3535,3 +3535,41 @@ The full CPU-default suite passed: 527 tests passed, 162 skipped and the known EMA expected failure, alongside the local real-data GPU workflow-command run. Static repository checks passed. The workflow itself has only local YAML/shell validation; execution by GitHub Actions on the intended runner is still required. + +### Compact deployment history records + +A standard-library archive/renderer now creates immutable compact TensorRT +comparison records and a standalone HTML history. It preserves baseline/candidate +quality and deltas, input/manifest hashes, engine identities, build settings and +resource snapshots. Runtime metadata is taken from hash-verified retained latency +evidence. Prediction rows, local paths and exception details are excluded from +the public projection. Identical archive retries preserve the existing record; +conflicting evidence under the same run ID fails without overwriting it. + +The two completed flat/hierarchical Blair smoke reports were archived and rendered +locally under ignored `tmp-report-history-final/`. The page contains two runs and +three quality tables with explicit macro metric labels. Resource comparisons are +excluded, matching the overlapping CPU-check conditions of those runs. Input +identities and build settings remain available in each compact JSON record, and +checks confirm local filesystem paths are absent. This is local report packaging, +not evidence that a public history service has been deployed. + +Ten focused tests cover immutable/idempotent records, conflicting IDs, malformed +or nonfinite metric evidence, undefined values, changed runtime evidence, early +failures, unsafe names/links and HTML escaping. Local HTML parsing and content +checks passed; no browser renderer is available in this environment, so visual +browser qualification is not claimed. The system Python here is 3.10 and cannot +run this repository's supported-runtime code; the maintained `.venv` invocation +succeeds and is documented. See the [history commands](../dev/benchmarks/README.md#compact-report-history-and-local-dashboard). + +Persistent storage, hosted publication and CPU/training adapters remain necessary. +The archive's default exclusion of resource comparisons prevents these correctness +smokes from being presented as new performance evidence. Explicitly enabling +resource display remains a publisher assertion, not target-hardware or production +acceptance. + +Validation passed static checks, the ten focused tests on the final recorder and +renderer, and the full CPU-default suite: 537 passed, 162 skipped and the known +EMA expected failure. The real-report projection/render checks passed using the +maintained Python environment. No new GPU inference or hosting claim accompanies +this reporting milestone. diff --git a/docs/quantization-status.md b/docs/quantization-status.md index ef08b0d..5fb714b 100644 --- a/docs/quantization-status.md +++ b/docs/quantization-status.md @@ -50,6 +50,14 @@ using explicitly configured environments, ONNX sources and held-out inputs. Its job summary and 90-day artifacts provide per-run visibility; remote execution, durable cross-run reporting and production acceptance remain unverified. +A compact immutable TensorRT report archive and standalone HTML history now +render the real flat/hierarchical smoke reports locally. Hash-verified runtime +metadata, input identities, quality and resource scope survive the projection; +local paths and prediction rows do not. Resource comparisons are excluded by +default and remain excluded for these smokes. This prepares a reviewable reporting +format, but persistent storage, public hosting and CPU/training adapters are still +unimplemented. + ## Current evidence | Workstream | Verified locally | What remains unproven | diff --git a/tests/test_benchmark_report_history.py b/tests/test_benchmark_report_history.py new file mode 100644 index 0000000..8fbc44e --- /dev/null +++ b/tests/test_benchmark_report_history.py @@ -0,0 +1,122 @@ +import hashlib +import json + +import pytest + +from dev.benchmarks.report_history import METRICS, archive, render + + +@pytest.fixture +def report(tmp_path): + child = tmp_path / "latency-0" + child.mkdir() + (child / "report.json").write_text( + json.dumps( + { + "versions": {"torch": "test", "tensorrt": "test", "numpy": "test", "private": "/private/runtime"}, + "environment": {"gpu": "GPU", "compute_capability": [8, 6], "platform": "Linux", "python": "3.13"}, + } + ) + ) + snapshots = { + name: {"device_used_bytes": value, "host": {"resident_bytes": value}} for name, value in [("cuda_initialized", 100), ("warm", 200)] + } + data = { + "schema_version": 1, + "status": "evaluated", + "phase": "complete", + "settings": {"trials": 1}, + "builds": { + role: {"directory": "/private/model", "engine": {"sha256": "a" * 64, "bytes": 1, "context_memory_bytes": 2}} + for role in ("baseline", "candidate") + }, + "quality": { + "levels": [{"name": "leaf", "samples": 10, "prediction_changes": 2, "candidate_minus_baseline": dict.fromkeys(METRICS, 0.1)}], + "models": { + role: {"source_path": "/private/predictions.csv", "metrics": {key: {"0": value} for key in METRICS}} + for role, value in [("baseline", 0.5), ("candidate", 0.6)] + }, + }, + "trials": [ + { + "trial": 0, + "latency": {"median_paired_ratio": 0.5, "median_seconds": {"baseline": 0.2, "candidate": 0.1}}, + "memory": {role: snapshots for role in ("baseline", "candidate")}, + } + ], + "stages": [ + {"name": "latency-0", "status": "passed", "report_sha256": hashlib.sha256((child / "report.json").read_bytes()).hexdigest()} + ], + } + path = tmp_path / "report.json" + path.write_text(json.dumps(data)) + return path, data + + +def test_archive_is_idempotent_but_rejects_changed_identity(report, tmp_path): + path, _ = report + history = tmp_path / "history" + archived = archive(path, history, "run-1", "a" * 40, "profile") + before = archived.read_bytes() + assert archive(path, history, "run-1", "a" * 40, "profile").read_bytes() == before + with pytest.raises(ValueError, match="already exists"): + archive(path, history, "run-1", "b" * 40, "profile") + assert archived.read_bytes() == before and not list((history / "records").glob("*.tmp")) + assert b"/private" not in before + document = render(history).read_text() + assert "Resource readings excluded" in document and "Paired latency ratio" not in document + assert "Measured GPU: GPU" in document + + +@pytest.mark.parametrize("kind", ["delta", "nonfinite", "missing-trial"]) +def test_invalid_evidence_is_not_archived(report, tmp_path, kind): + path, data = report + if kind == "delta": + data["quality"]["levels"][0]["candidate_minus_baseline"]["f1"] = 0.9 + if kind == "nonfinite": + data["quality"]["models"]["baseline"]["metrics"]["f1"]["0"] = float("nan") + if kind == "missing-trial": + data["trials"] = [] + path.write_text(json.dumps(data)) + with pytest.raises(ValueError): + archive(path, tmp_path / "history", "run", "a" * 40, "profile") + assert not (tmp_path / "history").exists() + + +def test_hash_changed_runtime_evidence_is_rejected(report, tmp_path): + path, _ = report + (tmp_path / "latency-0/report.json").write_text("{}") + with pytest.raises(ValueError, match="evidence changed"): + archive(path, tmp_path / "history", "run", "a" * 40, "profile") + + +def test_undefined_metrics_and_escaped_labels_remain_visible(report, tmp_path): + path, data = report + data["quality"]["models"]["baseline"]["metrics"]["f1"]["0"] = None + data["quality"]["levels"][0]["candidate_minus_baseline"]["f1"] = None + path.write_text(json.dumps(data)) + history = tmp_path / "history" + archive(path, history, "run", "a" * 40, "", performance_valid=True, note="") + document = render(history).read_text() + assert "undefined" in document and "Paired latency ratio" in document + assert "