From dc3fd16239d05e5220a2637f967eac874515b6d9 Mon Sep 17 00:00:00 2001
From: xiaoyu <1259084489@qq.com>
Date: Sat, 3 Oct 2026 19:57:31 +0800
Subject: [PATCH 1/4] cuda: fix three ways the checker lost or accepted a bad
report
All three come from review on #25.
1. reference.entrypoint was used as a path without checking its type. A list
reached os.path.isabs, whose TypeError escaped check_manifest: the CLI printed
no structured report at all, not even under --json, and every later backend
went unchecked. Non-strings are now an Issue and checking continues.
2. A directory holding any *.backend.json was exempt from the undeclared-backend
check, but discovery only loads
/.backend.json. A manifest named
typo.backend.json therefore bypassed every check for that backend: exit 0,
checked: [], no errors and no warnings. A manifest that is present but not
under the expected name is now reported.
3. Architectures read from `arch=` variables and targets written literally in a
-gencode flag were combined with `or`, so the literals were discarded whenever
a variable existed. A script assigning `arch=${2:-89}` and then compiling
`-gencode arch=compute_90,code=sm_90` builds sm_90 only, yet a manifest
declaring [89] passed with no findings.
build_script.py now reports gencode_architectures: the targets that actually
reach nvcc, resolved against the variable environment in force at each line,
so a loop that reassigns `arch` still yields both targets. The checker treats
those as the authority and reports an assigned value that never reaches a
-gencode flag.
62 tests, seven of them new and one per defect. Verified against #19's real
build.sh, which still parses to variables [89] / gencode [89] and passes with no
findings.
---
src/backends/cuda/build_script.py | 227 +++++++
src/backends/cuda/check_contract.py | 417 ++++++++++++
src/backends/cuda/contract.md | 239 +++++++
src/backends/cuda/tests/__init__.py | 0
src/backends/cuda/tests/test_build_script.py | 138 ++++
.../cuda/tests/test_check_contract.py | 609 ++++++++++++++++++
6 files changed, 1630 insertions(+)
create mode 100644 src/backends/cuda/build_script.py
create mode 100755 src/backends/cuda/check_contract.py
create mode 100644 src/backends/cuda/contract.md
create mode 100644 src/backends/cuda/tests/__init__.py
create mode 100644 src/backends/cuda/tests/test_build_script.py
create mode 100644 src/backends/cuda/tests/test_check_contract.py
diff --git a/src/backends/cuda/build_script.py b/src/backends/cuda/build_script.py
new file mode 100644
index 0000000..dd5d9b5
--- /dev/null
+++ b/src/backends/cuda/build_script.py
@@ -0,0 +1,227 @@
+#!/usr/bin/env python3
+"""Read a backend build script as text and report what it declares.
+
+``contract.md`` requires ``build.sh`` to accept an output directory and a compute
+capability, and to build exactly the library the manifest names for exactly the
+architectures it names. Checking that means reading the script — never running
+it: CI must not execute arbitrary repository code to decide whether a manifest is
+honest.
+
+The parsing is shell-shaped but deliberately shallow. It evaluates the two forms
+a build script actually uses (``${NAME:-default}`` and ``$NAME``, including
+nesting) and ignores comments, without pretending to be a shell.
+"""
+
+from __future__ import annotations
+
+import os
+import re
+
+# `sm_90a` / `compute_89` / `arch=compute_90` in a build script. Only used to
+# catch a build script that hard-codes one architecture while the manifest
+# claims more; it is not a substitute for reading the script.
+# A `-gencode` flag and its value: `-gencode "arch=compute_89,code=sm_89"`, or with
+# `=`. This is the flag that decides what nvcc actually builds, which is not
+# necessarily what an `arch=` variable says.
+_GENCODE = re.compile(r"-gencode[=\s]+(\"[^\"]*\"|\S+)")
+
+_ARCH_IN_SCRIPT = re.compile(r"\b(?:sm|compute)_(\d+)[af]?\b")
+_BUILD_SCRIPT_ARCH_FIELD = re.compile(r"\b(?:CUDA_COMPUTE_CAP|ARCH|arch)\b")
+_SCRIPT_RESOLVED = re.compile(r"^\s*([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.+?)\s*$")
+
+_TOKEN = re.compile(r"\$\{[^{}]*\{[^{}]*\}[^{}]*\}|\$\{[^{}]*\}|\$[A-Za-z_]\w*|\$\d")
+
+def _strip_comment(line):
+ """Remove a shell comment, respecting quotes.
+
+ `# defaults to CUDA_COMPUTE_CAP, else 89.` must not be read as an assignment
+ or as a compute capability, so comments are removed before any scanning.
+ """
+ out = []
+ quote = ""
+ for index, char in enumerate(line):
+ if quote:
+ if char == quote:
+ quote = ""
+ out.append(char)
+ continue
+ if char in ("'", '"'):
+ quote = char
+ out.append(char)
+ continue
+ if char == "#" and (index == 0 or line[index - 1].isspace()):
+ break
+ out.append(char)
+ return "".join(out)
+
+
+def _balanced_parameter(value):
+ """Return (name, default) for a value that is exactly one ``${...}``.
+
+ Shell parameter expansions nest (``${2:-${CUDA_COMPUTE_CAP:-89}}``), so the
+ closing brace has to be matched by counting rather than by a regex, which
+ would stop at the inner brace.
+ """
+ if not value.startswith("${"):
+ return None
+ depth = 0
+ for index, char in enumerate(value):
+ if char == "{":
+ depth += 1
+ elif char == "}":
+ depth -= 1
+ if depth == 0:
+ if index != len(value) - 1:
+ return None # trailing text, not a lone parameter expansion
+ interior = value[2:index]
+ name, separator, default = interior.partition(":-")
+ return name, (default if separator else "")
+ return None
+
+
+def _resolve_token(token, script_env, depth=0):
+ """Return (value, known) for one shell token inside a larger string.
+
+ ``known`` is False when the token depends on something CI cannot know, such
+ as a positional parameter the release process supplies. An unknown token must
+ not collapse to an empty string: 'we cannot tell' and 'the value is empty'
+ have to stay distinguishable, or an unexpanded path would be read as a
+ filename.
+ """
+ if depth > 8:
+ return "", False
+ value = token.strip().strip('"').strip("'")
+ parameter = _balanced_parameter(value)
+ if parameter is not None:
+ name, default = parameter
+ if name.isdigit() or ":" in name:
+ if default:
+ return _resolve_token(default, script_env, depth + 1)
+ return "", False
+ assigned = script_env.get(name, "").strip()
+ if not assigned:
+ return _resolve_token(default, script_env, depth + 1) if default else ("", False)
+ return _resolve_token(assigned, script_env, depth + 1)
+ for name, assigned in script_env.items():
+ if value == "$" + name:
+ return _resolve_token(assigned, script_env, depth + 1)
+ if value.startswith("$"):
+ return "", False
+ return value, True
+
+
+# A shell expansion inside a larger word. Alternation order matters: the nested
+# form has to be tried before the flat form, and the flat form must stop at the
+# first `}`. `$` is only a parameter start when followed by a name, `{` or a
+# digit, so the literal prefix in `lib_${arch}.so` is not swallowed by `\$\d`.
+def _expand_string(text, script_env):
+ """Substitute every shell token in ``text``, keeping the literal tail.
+
+ ``"$out/libqwen3_5_cuda.so"`` -> ``libqwen3_5_cuda.so``: the prefix is
+ unknowable but the filename is not, and the filename is what the manifest is
+ checked against.
+ """
+ out = []
+ position = 0
+ for match in _TOKEN.finditer(text):
+ out.append(text[position:match.start()])
+ value, _known = _resolve_token(match.group(0), script_env)
+ out.append(value)
+ position = match.end()
+ out.append(text[position:])
+ return "".join(out)
+
+
+def _output_name(text, script_env):
+ """The filename a build script writes, from its last ``-o`` argument.
+
+ The last one wins, for the same reason the last ``arch=`` assignment does: a
+ script that builds more than one target ends with the one a plain invocation
+ produces. The argument may be quoted and may mix a literal prefix with an
+ expansion (``"$out/liblaya_cuda.so"``). Expansion is applied to the word
+ first and quotes are trimmed after, because the closing quote of
+ ``-o "$out/libx.so"`` is only trailing relative to the expanded word.
+ """
+ name = ""
+ for line in text.splitlines():
+ for match in re.finditer(r"(?/.backend.json`` manifest and
+checks the parts of ``contract.md`` that do not need hardware:
+
+ * the manifest schema,
+ * that every declared source file exists,
+ * that the build script's declared output and architectures match the manifest,
+ * that the ABI version is consistent across backends,
+ * that a ``validated`` backend declares a tolerance and a reference entrypoint,
+ * that a kernel requiring a newer compute capability than the build declares is
+ flagged.
+
+It does not compile CUDA and does not prove numerics. Tier 2 (``--gpu`` in CI)
+runs the reference entrypoint on a self-hosted GPU runner.
+
+Usage:
+ python3 src/backends/cuda/check_contract.py [--repo-root PATH] [--json]
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import sys
+
+from build_script import parse_build_script
+
+MANIFEST_SUFFIX = ".backend.json"
+BACKENDS_DIR = os.path.join("src", "backends", "cuda")
+STATUSES = ("planned", "experimental", "validated")
+# Every key check_manifest reads directly. The missing-key report and the guard
+# that stops after it are both derived from this tuple.
+REQUIRED_KEYS = ("name", "abi_version", "status", "sources", "build")
+CONTRACT_ABI_VERSION = 1
+
+
+
+class Issue(object):
+ """One contract violation or warning for a backend."""
+
+ def __init__(self, level, backend, message):
+ self.level = level # "error" or "warning"
+ self.backend = backend
+ self.message = message
+
+ def __str__(self):
+ return "[%s] %s: %s" % (self.level, self.backend, self.message)
+
+ def as_dict(self):
+ return {"level": self.level, "backend": self.backend, "message": self.message}
+
+
+
+def load_manifests(repo_root):
+ """Return (manifests, issues) for every ``*.backend.json`` under backends/cuda."""
+ issues = []
+ manifests = []
+ root = os.path.join(repo_root, BACKENDS_DIR)
+ if not os.path.isdir(root):
+ issues.append(Issue("error", "-", "%s does not exist" % BACKENDS_DIR))
+ return manifests, issues
+
+ # A backend may be a subdirectory named after itself, or the cuda directory
+ # itself when its files sit directly under it (`kernels/`, `tools/`). The
+ # layout is the model author's call; only the manifest's contents are fixed.
+ found = []
+ for entry in sorted(os.listdir(root)):
+ directory = os.path.join(root, entry)
+ if not os.path.isdir(directory) or entry.startswith("."):
+ continue
+ expected = os.path.join(directory, entry + MANIFEST_SUFFIX)
+ if os.path.isfile(expected):
+ found.append((entry, directory, expected))
+ for filename in sorted(os.listdir(root)):
+ if filename.endswith(MANIFEST_SUFFIX) and os.path.isfile(os.path.join(root, filename)):
+ found.append((filename[: -len(MANIFEST_SUFFIX)], root,
+ os.path.join(root, filename)))
+
+ # A directory holding `.cu` files with no manifest either way is a backend
+ # someone forgot to declare. Directories that belong to a declared backend
+ # (its `kernels/`, `tools/`) or hold a manifest of their own are not.
+ declared_roots = {directory for _entry, directory, _path in found}
+ for entry in sorted(os.listdir(root)):
+ directory = os.path.join(root, entry)
+ if not os.path.isdir(directory) or entry.startswith("."):
+ continue
+ if any(directory == root_of or directory.startswith(root_of + os.sep)
+ for root_of in declared_roots):
+ continue
+ contents = [f for f in os.listdir(directory) if not f.startswith(".")]
+ present = sorted(f for f in contents if f.endswith(MANIFEST_SUFFIX))
+ if present:
+ # A manifest is here but not under the name discovery looks for, so
+ # nothing loaded it. Continuing silently meant a typo in the filename
+ # bypassed every check for that backend: exit 0, `checked: []`, no
+ # errors and no warnings.
+ issues.append(Issue(
+ "error", entry,
+ "has %s but not %s, so it is never discovered; rename it or fix the "
+ "directory name" % (", ".join(present), entry + MANIFEST_SUFFIX)))
+ continue
+ kernel_like = [f for f in contents if f.endswith((".cu", ".cuh"))]
+ if kernel_like:
+ issues.append(Issue(
+ "error", entry,
+ "has kernel sources (%s) but no %s manifest or parent backend manifest"
+ % (", ".join(sorted(kernel_like)[:3]), entry + MANIFEST_SUFFIX)))
+
+ for entry, directory, expected in found:
+ try:
+ with open(expected, "r", encoding="utf-8") as handle:
+ manifest = json.load(handle)
+ except ValueError as error:
+ issues.append(Issue("error", entry, "%s is not valid JSON: %s"
+ % (os.path.basename(expected), error)))
+ continue
+ if not isinstance(manifest, dict):
+ issues.append(Issue("error", entry, "%s must contain a JSON object"
+ % os.path.basename(expected)))
+ continue
+ manifest["_directory"] = directory
+ manifest["_entry"] = entry
+ manifests.append(manifest)
+ return manifests, issues
+
+
+def check_manifest(manifest, issues, repo_root=None):
+ backend = manifest.get("_entry", "?")
+ directory = manifest.get("_directory", "")
+ if repo_root is None:
+ # Best effort when called directly: four levels up from
+ # /src/backends/cuda/. Callers that know the root should pass
+ # it, because the flat layout makes this guess wrong.
+ repo_root = os.path.dirname(os.path.dirname(os.path.dirname(
+ os.path.dirname(os.path.abspath(directory or ".")))))
+
+ # These were two hand-written lists that had to agree, and they drifted:
+ # `name` was reported as missing and then read anyway, so a manifest missing
+ # only `name` raised KeyError out of check_manifest and aborted the checks of
+ # every other backend, in both the text and --json paths. Deriving the guard
+ # from the same constant is what stops that recurring.
+ missing = [key for key in REQUIRED_KEYS if key not in manifest]
+ for key in missing:
+ issues.append(Issue("error", backend, "manifest is missing required key %r" % key))
+ if missing:
+ return
+
+ name = manifest["name"]
+ if not isinstance(name, str) or not name:
+ issues.append(Issue("error", backend, "name must be a non-empty string"))
+ elif name != backend:
+ issues.append(Issue("error", backend,
+ "name %r does not match directory name %r" % (name, backend)))
+
+ abi = manifest["abi_version"]
+ if not isinstance(abi, int) or abi < CONTRACT_ABI_VERSION:
+ issues.append(Issue("error", backend,
+ "abi_version must be an integer >= %d, got %r"
+ % (CONTRACT_ABI_VERSION, abi)))
+
+ status = manifest["status"]
+ if status not in STATUSES:
+ issues.append(Issue("error", backend,
+ "status must be one of %s, got %r" % (", ".join(STATUSES), status)))
+ return
+
+ sources = manifest["sources"]
+ if not isinstance(sources, list) or not all(isinstance(item, str) for item in sources):
+ issues.append(Issue("error", backend, "sources must be a list of strings"))
+ else:
+ for source in sources:
+ if os.path.isabs(source) or ".." in source.split("/"):
+ issues.append(Issue("error", backend,
+ "source %r must be relative to the backend directory" % source))
+ continue
+ if not os.path.isfile(os.path.join(directory, source)):
+ issues.append(Issue("error", backend, "declared source %r does not exist" % source))
+
+ build = manifest["build"]
+ if not isinstance(build, dict):
+ issues.append(Issue("error", backend, "build must be an object"))
+ return
+
+ for key in ("script", "output"):
+ if not isinstance(build.get(key), str) or not build.get(key):
+ issues.append(Issue("error", backend, "build.%s must be a non-empty string" % key))
+ architectures = build.get("architectures")
+ if not isinstance(architectures, list) or not architectures \
+ or not all(isinstance(item, int) and 50 <= item <= 200 for item in architectures):
+ issues.append(Issue("error", backend,
+ "build.architectures must be a non-empty list of compute "
+ "capabilities, e.g. [89, 90]"))
+ architectures = None
+ default_arch = build.get("default_arch")
+ if default_arch is not None and architectures is not None and default_arch not in architectures:
+ issues.append(Issue("error", backend,
+ "build.default_arch %r is not listed in build.architectures %r"
+ % (default_arch, architectures)))
+
+ # What the kernels require is currently stated only in comments and READMEs
+ # ("tensor-core kernels need sm_80 or newer"). Declaring it makes the claim
+ # checkable here rather than discoverable as a build failure on someone
+ # else's GPU.
+ min_capability = build.get("min_capability")
+ if min_capability is not None:
+ if not isinstance(min_capability, int):
+ issues.append(Issue("error", backend,
+ "build.min_capability must be an integer compute capability, "
+ "e.g. 80"))
+ elif architectures is not None:
+ too_low = [item for item in architectures if item < min_capability]
+ if too_low:
+ issues.append(Issue("error", backend,
+ "build.architectures includes %r, below build.min_capability "
+ "%d; the kernels would not build for that target"
+ % (too_low, min_capability)))
+
+ # A backend can serve more than one model. Kev and Cua-S1 share the same
+ # Qwen3.5 backbone, so listing consumers is what makes reuse visible instead
+ # of a private arrangement between two PRs.
+ models = manifest.get("models")
+ if models is not None:
+ if not isinstance(models, list) or not all(isinstance(item, str) for item in models):
+ issues.append(Issue("error", backend, "models must be a list of strings"))
+ else:
+ for model in models:
+ if not model.endswith("/") or os.path.isabs(model) or ".." in model.split("/"):
+ issues.append(Issue("error", backend,
+ "models entry %r must be a repository-relative directory "
+ "path ending in '/'" % model))
+ elif not os.path.isdir(os.path.join(repo_root, model)):
+ issues.append(Issue("warning", backend,
+ "models entry %r does not exist yet; the consumer engine "
+ "is not in the tree" % model))
+
+ if status == "planned":
+ return
+
+ # A non-planned backend must ship the build script it names.
+ script_path = ""
+ if isinstance(build.get("script"), str):
+ script_path = os.path.join(directory, build["script"])
+ if not os.path.isfile(script_path):
+ issues.append(Issue("error", backend,
+ "build.script %r does not exist" % build["script"]))
+ else:
+ _check_build_script(backend, script_path, build, architectures, issues)
+ # The compile job runs the script directly, so the execute bit must
+ # be set in git, not only in a local working copy: a fresh clone is
+ # what CI checks out.
+ if os.name == "posix" and not os.access(script_path, os.X_OK):
+ issues.append(Issue("error", backend,
+ "build.script %r is not executable in git; the compile job "
+ "runs it as ./%s. Fix with: git update-index --chmod=+x %s"
+ % (build["script"], build["script"], script_path)))
+
+ numerics = manifest.get("numerics")
+ reference = manifest.get("reference")
+ if status == "validated":
+ if not isinstance(numerics, dict) or not isinstance(numerics.get("tolerance"), dict) \
+ or not numerics["tolerance"]:
+ issues.append(Issue("error", backend,
+ "status is validated but numerics.tolerance is missing; a "
+ "parity claim needs a tolerance declared before comparison"))
+ if not isinstance(reference, dict) or not reference.get("entrypoint"):
+ issues.append(Issue("error", backend,
+ "status is validated but reference.entrypoint is missing"))
+ if isinstance(reference, dict) and reference.get("entrypoint"):
+ entrypoint = reference["entrypoint"]
+ # Type first. `os.path.isabs` and `split` raise TypeError on a list or a
+ # number, and that exception escaped check_manifest: the CLI printed no
+ # structured report at all, not even under --json, and every later
+ # backend went unchecked.
+ if not isinstance(entrypoint, str):
+ issues.append(Issue("error", backend,
+ "reference.entrypoint must be a string, got %s"
+ % type(entrypoint).__name__))
+ entrypoint = ""
+ elif os.path.isabs(entrypoint) or ".." in entrypoint.split("/"):
+ issues.append(Issue("error", backend,
+ "reference.entrypoint must be repository-relative"))
+ elif entrypoint:
+ manifest["_repo_relative_entrypoint"] = entrypoint
+ _check_reference_entrypoint(backend, entrypoint, issues, repo_root)
+ if isinstance(numerics, dict) and isinstance(numerics.get("tolerance"), dict):
+ for key, value in numerics["tolerance"].items():
+ if key == "note":
+ continue
+ if not isinstance(value, (int, float)):
+ issues.append(Issue("error", backend,
+ "numerics.tolerance.%s must be a number or a note, got %r"
+ % (key, value)))
+
+
+def _check_reference_entrypoint(backend, entrypoint, issues, repo_root):
+ """A declared reference that is not there cannot be run by Tier 2."""
+ if not os.path.isfile(os.path.join(repo_root, entrypoint)):
+ issues.append(Issue("warning", backend,
+ "reference.entrypoint %r does not exist yet; the Tier-2 GPU job "
+ "cannot run parity for this backend" % entrypoint))
+
+
+def _check_build_script(backend, script_path, build, architectures, issues):
+ try:
+ with open(script_path, "r", encoding="utf-8") as handle:
+ text = handle.read()
+ except OSError as error:
+ issues.append(Issue("error", backend, "cannot read %s: %s" % (build.get("script"), error)))
+ return
+
+ parsed = parse_build_script(text)
+
+ declared_output = build.get("output")
+ if parsed["output"] and declared_output and parsed["output"] != declared_output:
+ issues.append(Issue("error", backend,
+ "build.output %r does not match the %r the build script writes"
+ % (declared_output, parsed["output"])))
+
+ if architectures is None:
+ return
+
+ # What nvcc is actually told to build, from the -gencode flags. This is the
+ # authority: a variable that never reaches one of those flags does not make a
+ # target reachable. A script that assigns `arch=${2:-89}` and then writes
+ # `-gencode arch=compute_90,code=sm_90` builds sm_90 only, so a manifest
+ # declaring sm_89 is false even though an `arch=` variable says 89.
+ gencode = parsed.get("gencode_architectures") or []
+ by_variable = parsed["architectures"]
+ if gencode and by_variable:
+ unused = [item for item in by_variable if item not in gencode]
+ if unused:
+ issues.append(Issue("error", backend,
+ "build script assigns %r but only passes %r to nvcc; the "
+ "assigned value never reaches a -gencode flag, so targets "
+ "read from variables alone cannot be claimed"
+ % (by_variable, gencode)))
+ script_arch = gencode or by_variable or parsed["literal_architectures"]
+ if not script_arch:
+ issues.append(Issue("warning", backend,
+ "build script declares no compute capability; the manifest claims "
+ "%r but CI cannot confirm the script honours it" % (architectures,)))
+ return
+ unbuildable = [item for item in architectures if item not in script_arch]
+ if unbuildable:
+ issues.append(Issue("error", backend,
+ "build.architectures claims %r but the build script only reaches %r "
+ "(missing %r)" % (architectures, script_arch, unbuildable)))
+ extra = [item for item in script_arch if item not in architectures]
+ if extra:
+ issues.append(Issue("warning", backend,
+ "build script also targets %r, which build.architectures omits"
+ % (extra,)))
+
+
+def check_abi_consistency(manifests, issues):
+ """A loader cannot know two ABI versions at once, so they must agree."""
+ versions = {}
+ for manifest in manifests:
+ backend = manifest.get("_entry", "?")
+ version = manifest.get("abi_version")
+ if isinstance(version, int):
+ versions.setdefault(version, []).append(backend)
+ if len(versions) > 1:
+ detail = ", ".join("%d (%s)" % (version, ", ".join(sorted(names)))
+ for version, names in sorted(versions.items()))
+ issues.append(Issue("error", "cuda",
+ "backends declare different abi_version values: %s; a process that "
+ "loads two of them cannot check one version" % detail))
+
+
+def run(repo_root):
+ manifests, issues = load_manifests(repo_root)
+ for manifest in manifests:
+ check_manifest(manifest, issues, repo_root)
+ check_abi_consistency(manifests, issues)
+ return manifests, issues
+
+
+def main(argv=None):
+ parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
+ parser.add_argument("--repo-root", default=".",
+ help="repository root to check (default: current directory)")
+ parser.add_argument("--json", action="store_true", help="emit machine-readable results")
+ args = parser.parse_args(argv)
+
+ repo_root = os.path.abspath(args.repo_root)
+ manifests, issues = run(repo_root)
+ errors = [issue for issue in issues if issue.level == "error"]
+ warnings = [issue for issue in issues if issue.level == "warning"]
+
+ if args.json:
+ json.dump({
+ "checked": sorted(m.get("_entry", "?") for m in manifests),
+ "issues": [issue.as_dict() for issue in issues],
+ "errors": len(errors),
+ "warnings": len(warnings),
+ }, sys.stdout, indent=2)
+ sys.stdout.write("\n")
+ else:
+ for issue in issues:
+ print(issue)
+ for manifest in manifests:
+ print("checked %s: status=%s abi=%s"
+ % (manifest.get("_entry"), manifest.get("status"), manifest.get("abi_version")))
+ print("%d backend(s), %d error(s), %d warning(s)"
+ % (len(manifests), len(errors), len(warnings)))
+ if not manifests:
+ print("no backends declared; add src/backends/cuda//.backend.json")
+
+ return 1 if errors else 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/src/backends/cuda/contract.md b/src/backends/cuda/contract.md
new file mode 100644
index 0000000..b2b9823
--- /dev/null
+++ b/src/backends/cuda/contract.md
@@ -0,0 +1,239 @@
+# CUDA backend contract
+
+`src/backends/cuda/` is shared by several model engines. Each model owns its own
+operations, its own library and its own numerics; this document fixes only the
+parts that have to agree for two model libraries to be built, distributed and
+validated the same way.
+
+The rule from the [repository layout](../../../README.md) still holds: model
+orchestration, batching policy, state management and kernel selection stay with
+the model engine. A backend does not need identical internal structures to
+another backend, and no universal tensor abstraction is introduced here.
+
+Status: proposed. The checker is [`check_contract.py`](check_contract.py); wiring it
+into CI is a separate change, so nothing here is enforced yet.
+
+## Why this exists
+
+Three CUDA efforts are now in flight, each with its own build path:
+
+| Effort | Kernels | Produces | Built by |
+| --- | --- | --- | --- |
+| Laya native (#14) | TileLang, exported to CUDA | `liblaya_cuda.so` | `tools/export.py` + `tools/build.py` |
+| Cua-S1 native (#19) | CUDA C++ | `libqwen3_5_cuda.so` | `build.sh` |
+| Cua-S1 multimodal (#12) | Triton / cuTile | (Python) | — |
+
+That is three build systems, three library names and three C ABIs. None of it
+conflicts today, because nothing links against anything else yet. It conflicts
+as soon as one model reuses another's kernels — which is already planned: the
+Cua-S1 native worker and a Kev engine share a Qwen3.5 base, and `#9` asks for
+CUDA paths for CLM as well.
+
+`#6` says to extract shared code once two implementations exist. Two exist now.
+This document is that extraction, kept to interfaces only.
+
+## 1. Artifact discovery
+
+Every CUDA backend declares itself in one JSON file:
+
+```text
+src/backends/cuda//.backend.json
+```
+
+The checker discovers backends by that path, so adding a backend needs no CI
+change. Schema:
+
+```json
+{
+ "name": "qwen3_5",
+ "abi_version": 1,
+ "status": "validated",
+ "sources": ["common.cuh", "mma.cuh", "ops.h", "norm.cu", "elementwise.cu",
+ "attention.cu", "gdn_prefill.cu", "gemm.cu", "runtime.cu"],
+ "models": ["src/models/cua_s1/", "src/models/kev/"],
+ "build": {
+ "script": "build.sh",
+ "output": "libqwen3_5_cuda.so",
+ "default_arch": 89,
+ "architectures": [89],
+ "min_capability": 80
+ },
+ "numerics": {
+ "precision": "bfloat16",
+ "accumulation": "float32",
+ "tolerance": { "max_abs": 0.039, "note": "float32 reference, see #11" }
+ },
+ "reference": {
+ "entrypoint": "recipe/cua_s1/check_native.py",
+ "note": "pinned upstream FourBModel; kernel tolerances are declared per suite"
+ }
+}
+```
+
+Required keys are `name`, `abi_version`, `status`, `sources` and `build`.
+`status` is one of:
+
+- `planned` — directory only; the checker skips everything else.
+- `experimental` — builds, but no parity claim yet. `numerics.tolerance` may be
+ omitted.
+- `validated` — a parity claim is made. `numerics.tolerance` and
+ `reference.entrypoint` are required, and the checker fails without them.
+
+`reference.entrypoint` is a repository-relative script that runs on a GPU and
+exits non-zero when parity fails. It takes no contract-defined arguments: a
+self-hosted runner that has the weights and the GPU runs it directly.
+
+### One backend, several models
+
+`models` lists the model engines that consume this backend, as
+repository-relative directories. It is optional, and it exists because reuse is
+the point: `#19`'s kernels serve a Qwen3.5-4B backbone, and Kev from `#9` uses
+the same backbone with a different adapter and readout, so one backend directory
+serves both. Without this field that sharing is a private arrangement between
+two PRs and invisible to anyone reading either one. An entry naming a directory
+that is not in the tree is a warning, not an error, so a backend can be merged
+before its second consumer lands.
+
+### Build script interface
+
+`build.script` is invoked by the compile job as:
+
+```sh
+./
+```
+
+so it must be executable, accept an output directory as `$1`, and accept a
+compute capability as `$2` — defaulting to `build.default_arch` when `$2` is
+absent. `#19`'s `build.sh` already has this shape:
+
+```sh
+out=${1:?usage: build.sh