From ee4416ef919241be2e34281035d8ec82df88565f Mon Sep 17 00:00:00 2001 From: Imran Siddique Date: Thu, 27 Aug 2026 09:15:26 -0700 Subject: [PATCH] ci: one runner for the WCM demos, callable from the SDK repository weight-custody-manifest 0.27.0 shipped two correct security changes and both broke demos here: release began refusing manifests whose identity was not pinned out of band, and the memory-fingerprint challenge began requiring a signed sweep. Six demos that passed on 0.26.0 failed on 0.27.0, and nothing caught it. Nothing could have. The list of demos to run lived inside this repository's workflow file, where the SDK's CI cannot reach it, and this repository only ever tests against a version already on PyPI. So the earliest possible detection was after publishing, which is after the point where a version number can be taken back. run_demos.py moves the list next to the demos, where both repositories can call it. This one runs it against the published package, catching a demo somebody broke. The SDK repository runs it against a wheel built from the branch under review, catching an SDK change that breaks the demos, on the pull request that causes it. Demos are discovered rather than listed. Every top-level module that is not a test runs, so a new demo is covered the day it lands, which matters because the gap being closed is a change nobody happened to exercise. A demo that cannot run offline goes in REQUIRES_NETWORK with a reason and is reported as skipped on every run, so the exclusions stay visible instead of living in a comment nobody re-reads. Failures print after the summary rather than interleaved, because six demos failing the same way is a different problem from one failing alone and that is the first thing worth knowing. Both streams are captured: a demo printing its refusal to stdout and a trace to stderr is the normal shape here. Verified by reverting one demo to its pre-fix state and watching the runner single it out. Co-Authored-By: Claude Opus 5 (1M context) Claude-Session: https://claude.ai/code/session_014NL8o3PXq6kfs2SdmBv6ak --- .github/workflows/ci.yml | 40 ++------ weight-custody-manifest/README.md | 26 ++++- weight-custody-manifest/run_demos.py | 138 +++++++++++++++++++++++++++ 3 files changed, 172 insertions(+), 32 deletions(-) create mode 100644 weight-custody-manifest/run_demos.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 3df8131..c5505d4 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -183,34 +183,12 @@ jobs: - name: Install the WCM SDK from PyPI run: python -m pip install -r requirements.txt - - name: Run the 30-second refuse-and-wipe demo - run: python refuse_and_wipe.py - - - name: Run the end-to-end custody demo - run: python open_model_e2e.py - - - name: Run the sovereign threshold demo - run: python sovereign_self_custody.py - - - name: Replay a SEV-SNP quote offline - run: python snp_replay.py - - - name: Run the feature examples - run: | - for s in closed_model_e2e multi_stage_byom transparency_log post_quantum channel_binding revocation_kill_switch quote_verification load_guard; do - echo "== $s ==" - python "$s.py" - done - - - name: Run the model-signing provenance example - run: | - python -m pip install "weight-custody-manifest[model-signing]>=0.27.0" - python provenance_model_signing.py - - # These existed and were never run by any workflow. They cover the artifact - # digest, which is the value a manifest binds, so a change to it that no - # demo happens to exercise would otherwise land unnoticed. - - name: Run the offline unit tests - run: | - python -m pip install pytest - python -m pytest . -q + # One runner, discovered rather than listed, so a new demo is covered the + # day it lands. The list lived in this file and in nobody else's reach, + # which is why the SDK repository could not run these before cutting a + # release. It now calls the same script; see run_demos.py. + - name: Install the model-signing extra the provenance demo needs + run: python -m pip install "weight-custody-manifest[model-signing]>=0.27.0" pytest + + - name: Run every offline demo and the unit tests + run: python run_demos.py diff --git a/weight-custody-manifest/README.md b/weight-custody-manifest/README.md index 64872b2..bd5975a 100644 --- a/weight-custody-manifest/README.md +++ b/weight-custody-manifest/README.md @@ -90,7 +90,31 @@ python real_open_model.py ## What runs in CI -Every offline example runs in CI against the published PyPI package and must exit 0: `refuse_and_wipe`, the closed/open e2e pair, `multi_stage_byom`, `revocation_kill_switch`, `channel_binding`, `sovereign_self_custody`, `transparency_log`, `post_quantum`, `quote_verification`, `load_guard`, `snp_replay`, and `provenance_model_signing` (with the `[model-signing]` extra). Only `real_open_model.py` is excluded (it downloads a model). +`run_demos.py` runs every offline example and the unit tests, and must exit 0. + +```bash +python run_demos.py # run everything +python run_demos.py --list # show the plan, including what is skipped and why +``` + +Demos are **discovered, not listed**. Every top-level module that is not a test +gets run, so a new demo is covered the day it lands. A demo that genuinely +cannot run offline goes in `REQUIRES_NETWORK` with a reason, which keeps the +exclusion visible on every run rather than buried in a workflow file. Two are +excluded today: `real_open_model.py` downloads a model, and `real_lora_custody.py` +trains a LoRA adapter. + +The same script is called from two places, and that is the point. This +repository runs it against the published PyPI package, which catches a demo that +somebody broke. The SDK repository runs it against a wheel built from the branch +under review, which catches an SDK change that breaks the demos. + +That second one is why this exists. `weight-custody-manifest` 0.27.0 shipped two +correct security changes and both broke demos here: key release began refusing +manifests whose identity was not pinned out of band, and the memory-fingerprint +challenge began requiring a signed sweep. Six demos that passed on 0.26.0 failed +on 0.27.0, and nothing caught it, because the SDK had no way to run these and +this repository only ever tested against what was already published. ## Reference diff --git a/weight-custody-manifest/run_demos.py b/weight-custody-manifest/run_demos.py new file mode 100644 index 0000000..1663db1 --- /dev/null +++ b/weight-custody-manifest/run_demos.py @@ -0,0 +1,138 @@ +#!/usr/bin/env python3 +"""Run every offline WCM demo and the unit tests, and report what broke. + +This exists because of a specific failure. weight-custody-manifest 0.27.0 +shipped two correct security changes, and both broke demos in this directory: +key release began refusing manifests whose identity was not pinned out of band, +and the memory-fingerprint challenge began requiring a signed sweep. Six demos +that passed on 0.26.0 failed on 0.27.0, and nothing caught it, because the SDK +repository had no way to run these and this repository only tested against +whatever was already published. + +So the list of what to run lives here, next to the demos, and both repositories +call it: this repository's CI on every PR, and the SDK's CI against a wheel +built from the branch under review. A change that breaks these now fails on the +pull request that causes it rather than after a release is on PyPI. + +**Demos are discovered, not listed.** Every top-level module that is not a test +and not explicitly excluded gets run. A new demo is covered the day it lands, +which is the point: the gap was a change nobody happened to exercise. A demo +that genuinely cannot run offline has to be added to ``REQUIRES_NETWORK`` with a +reason, which is a visible decision rather than a silent omission. + +Usage:: + + python run_demos.py # demos, then pytest + python run_demos.py --list # show what would run, and what is skipped +""" + +from __future__ import annotations + +import argparse +import pathlib +import subprocess +import sys +import time + +HERE = pathlib.Path(__file__).resolve().parent + +#: Scripts that cannot run in a bare CI container, and why. Anything here is +#: reported as skipped rather than quietly dropped, so the exclusions stay +#: visible every run instead of living in a comment nobody re-reads. +REQUIRES_NETWORK = { + "real_open_model.py": "downloads a real model from the Hugging Face Hub", + "real_lora_custody.py": "trains a LoRA adapter; needs torch and a download", +} + +#: This file. +SELF = pathlib.Path(__file__).name + + +def demos() -> list[pathlib.Path]: + """Every runnable demo, sorted, excluding tests and this runner.""" + return sorted( + path + for path in HERE.glob("*.py") + if path.name != SELF + and not path.name.startswith("test_") + and path.name not in REQUIRES_NETWORK + ) + + +def run(command: list[str], label: str) -> tuple[bool, float, str]: + started = time.perf_counter() + completed = subprocess.run( + command, cwd=HERE, capture_output=True, text=True, encoding="utf-8", errors="replace" + ) + elapsed = time.perf_counter() - started + # Both streams: a demo that prints its refusal to stdout and traces to + # stderr is the normal shape here, and reading only one of them has sent + # people looking in the wrong place. + output = (completed.stdout or "") + (completed.stderr or "") + return completed.returncode == 0, elapsed, output + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--list", action="store_true", help="print the plan and exit") + parser.add_argument("--no-tests", action="store_true", help="skip the pytest run") + args = parser.parse_args(argv) + + scripts = demos() + + if args.list: + print(f"{len(scripts)} demo(s) would run:") + for path in scripts: + print(f" {path.name}") + print(f"\n{len(REQUIRES_NETWORK)} skipped:") + for name, why in sorted(REQUIRES_NETWORK.items()): + print(f" {name}: {why}") + return 0 + + try: + import wcm + + print(f"weight-custody-manifest {wcm.__version__}\n") + except ImportError: + print("weight-custody-manifest is not installed", file=sys.stderr) + return 1 + + failures: list[tuple[str, str]] = [] + + for path in scripts: + ok, elapsed, output = run([sys.executable, path.name], path.name) + print(f"{'ok ' if ok else 'FAIL'} {path.name:<32} {elapsed:5.1f}s") + if not ok: + failures.append((path.name, output)) + + for name, why in sorted(REQUIRES_NETWORK.items()): + print(f"skip {name:<32} {why}") + + if not args.no_tests: + ok, elapsed, output = run( + [sys.executable, "-m", "pytest", ".", "-q"], "pytest" + ) + print(f"{'ok ' if ok else 'FAIL'} {'unit tests':<32} {elapsed:5.1f}s") + if not ok: + failures.append(("unit tests", output)) + + if not failures: + print(f"\nall {len(scripts)} demos and the unit tests pass") + return 0 + + # The output is printed after the summary, not interleaved, so the shape of + # the failure is visible before the detail. Six demos failing the same way + # is a different problem from one failing on its own, and that is the first + # thing worth knowing. + print(f"\n{len(failures)} failure(s): {', '.join(name for name, _ in failures)}\n") + for name, output in failures: + print("=" * 72) + print(name) + print("=" * 72) + print(output.strip()[-4000:]) + print() + return 1 + + +if __name__ == "__main__": + sys.exit(main())