diff --git a/src/backends/cuda/build_script.py b/src/backends/cuda/build_script.py new file mode 100644 index 0000000..dd5d9b5 --- /dev/null +++ b/src/backends/cuda/build_script.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""Read a backend build script as text and report what it declares. + +``contract.md`` requires ``build.sh`` to accept an output directory and a compute +capability, and to build exactly the library the manifest names for exactly the +architectures it names. Checking that means reading the script — never running +it: CI must not execute arbitrary repository code to decide whether a manifest is +honest. + +The parsing is shell-shaped but deliberately shallow. It evaluates the two forms +a build script actually uses (``${NAME:-default}`` and ``$NAME``, including +nesting) and ignores comments, without pretending to be a shell. +""" + +from __future__ import annotations + +import os +import re + +# `sm_90a` / `compute_89` / `arch=compute_90` in a build script. Only used to +# catch a build script that hard-codes one architecture while the manifest +# claims more; it is not a substitute for reading the script. +# A `-gencode` flag and its value: `-gencode "arch=compute_89,code=sm_89"`, or with +# `=`. This is the flag that decides what nvcc actually builds, which is not +# necessarily what an `arch=` variable says. +_GENCODE = re.compile(r"-gencode[=\s]+(\"[^\"]*\"|\S+)") + +_ARCH_IN_SCRIPT = re.compile(r"\b(?:sm|compute)_(\d+)[af]?\b") +_BUILD_SCRIPT_ARCH_FIELD = re.compile(r"\b(?:CUDA_COMPUTE_CAP|ARCH|arch)\b") +_SCRIPT_RESOLVED = re.compile(r"^\s*([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.+?)\s*$") + +_TOKEN = re.compile(r"\$\{[^{}]*\{[^{}]*\}[^{}]*\}|\$\{[^{}]*\}|\$[A-Za-z_]\w*|\$\d") + +def _strip_comment(line): + """Remove a shell comment, respecting quotes. + + `# defaults to CUDA_COMPUTE_CAP, else 89.` must not be read as an assignment + or as a compute capability, so comments are removed before any scanning. + """ + out = [] + quote = "" + for index, char in enumerate(line): + if quote: + if char == quote: + quote = "" + out.append(char) + continue + if char in ("'", '"'): + quote = char + out.append(char) + continue + if char == "#" and (index == 0 or line[index - 1].isspace()): + break + out.append(char) + return "".join(out) + + +def _balanced_parameter(value): + """Return (name, default) for a value that is exactly one ``${...}``. + + Shell parameter expansions nest (``${2:-${CUDA_COMPUTE_CAP:-89}}``), so the + closing brace has to be matched by counting rather than by a regex, which + would stop at the inner brace. + """ + if not value.startswith("${"): + return None + depth = 0 + for index, char in enumerate(value): + if char == "{": + depth += 1 + elif char == "}": + depth -= 1 + if depth == 0: + if index != len(value) - 1: + return None # trailing text, not a lone parameter expansion + interior = value[2:index] + name, separator, default = interior.partition(":-") + return name, (default if separator else "") + return None + + +def _resolve_token(token, script_env, depth=0): + """Return (value, known) for one shell token inside a larger string. + + ``known`` is False when the token depends on something CI cannot know, such + as a positional parameter the release process supplies. An unknown token must + not collapse to an empty string: 'we cannot tell' and 'the value is empty' + have to stay distinguishable, or an unexpanded path would be read as a + filename. + """ + if depth > 8: + return "", False + value = token.strip().strip('"').strip("'") + parameter = _balanced_parameter(value) + if parameter is not None: + name, default = parameter + if name.isdigit() or ":" in name: + if default: + return _resolve_token(default, script_env, depth + 1) + return "", False + assigned = script_env.get(name, "").strip() + if not assigned: + return _resolve_token(default, script_env, depth + 1) if default else ("", False) + return _resolve_token(assigned, script_env, depth + 1) + for name, assigned in script_env.items(): + if value == "$" + name: + return _resolve_token(assigned, script_env, depth + 1) + if value.startswith("$"): + return "", False + return value, True + + +# A shell expansion inside a larger word. Alternation order matters: the nested +# form has to be tried before the flat form, and the flat form must stop at the +# first `}`. `$` is only a parameter start when followed by a name, `{` or a +# digit, so the literal prefix in `lib_${arch}.so` is not swallowed by `\$\d`. +def _expand_string(text, script_env): + """Substitute every shell token in ``text``, keeping the literal tail. + + ``"$out/libqwen3_5_cuda.so"`` -> ``libqwen3_5_cuda.so``: the prefix is + unknowable but the filename is not, and the filename is what the manifest is + checked against. + """ + out = [] + position = 0 + for match in _TOKEN.finditer(text): + out.append(text[position:match.start()]) + value, _known = _resolve_token(match.group(0), script_env) + out.append(value) + position = match.end() + out.append(text[position:]) + return "".join(out) + + +def _output_name(text, script_env): + """The filename a build script writes, from its last ``-o`` argument. + + The last one wins, for the same reason the last ``arch=`` assignment does: a + script that builds more than one target ends with the one a plain invocation + produces. The argument may be quoted and may mix a literal prefix with an + expansion (``"$out/liblaya_cuda.so"``). Expansion is applied to the word + first and quotes are trimmed after, because the closing quote of + ``-o "$out/libx.so"`` is only trailing relative to the expanded word. + """ + name = "" + for line in text.splitlines(): + for match in re.finditer(r"(?/.backend.json`` manifest and +checks the parts of ``contract.md`` that do not need hardware: + + * the manifest schema, + * that every declared source file exists, + * that the build script's declared output and architectures match the manifest, + * that the ABI version is consistent across backends, + * that a ``validated`` backend declares a tolerance and a reference entrypoint, + * that a kernel requiring a newer compute capability than the build declares is + flagged. + +It does not compile CUDA and does not prove numerics. Tier 2 (``--gpu`` in CI) +runs the reference entrypoint on a self-hosted GPU runner. + +Usage: + python3 src/backends/cuda/check_contract.py [--repo-root PATH] [--json] +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import sys + +from build_script import parse_build_script + +MANIFEST_SUFFIX = ".backend.json" +BACKENDS_DIR = os.path.join("src", "backends", "cuda") +STATUSES = ("planned", "experimental", "validated") +# Every key check_manifest reads directly. The missing-key report and the guard +# that stops after it are both derived from this tuple. +REQUIRED_KEYS = ("name", "abi_version", "status", "sources", "build") +CONTRACT_ABI_VERSION = 1 +# `_ABI_VERSION N` in a backend header. #19 uses `CS1_ABI_VERSION`, +# bumped whenever the C interface changes. +_ABI_MACRO = re.compile(r"^[ \t]*#[ \t]*define[ \t]+(\w*ABI_VERSION)[ \t]+(\d+)[ \t]*$", re.M) + + + +class Issue(object): + """One contract violation or warning for a backend.""" + + def __init__(self, level, backend, message): + self.level = level # "error" or "warning" + self.backend = backend + self.message = message + + def __str__(self): + return "[%s] %s: %s" % (self.level, self.backend, self.message) + + def as_dict(self): + return {"level": self.level, "backend": self.backend, "message": self.message} + + + +def load_manifests(repo_root): + """Return (manifests, issues) for every ``*.backend.json`` under backends/cuda.""" + issues = [] + manifests = [] + root = os.path.join(repo_root, BACKENDS_DIR) + if not os.path.isdir(root): + issues.append(Issue("error", "-", "%s does not exist" % BACKENDS_DIR)) + return manifests, issues + + # A backend may be a subdirectory named after itself, or the cuda directory + # itself when its files sit directly under it (`kernels/`, `tools/`). The + # layout is the model author's call; only the manifest's contents are fixed. + found = [] + for entry in sorted(os.listdir(root)): + directory = os.path.join(root, entry) + if not os.path.isdir(directory) or entry.startswith("."): + continue + expected = os.path.join(directory, entry + MANIFEST_SUFFIX) + if os.path.isfile(expected): + found.append((entry, directory, expected)) + for filename in sorted(os.listdir(root)): + if filename.endswith(MANIFEST_SUFFIX) and os.path.isfile(os.path.join(root, filename)): + found.append((filename[: -len(MANIFEST_SUFFIX)], root, + os.path.join(root, filename))) + + # A directory holding `.cu` files with no manifest either way is a backend + # someone forgot to declare. Directories that belong to a declared backend + # (its `kernels/`, `tools/`) or hold a manifest of their own are not. + declared_roots = {directory for _entry, directory, _path in found} + for entry in sorted(os.listdir(root)): + directory = os.path.join(root, entry) + if not os.path.isdir(directory) or entry.startswith("."): + continue + if any(directory == root_of or directory.startswith(root_of + os.sep) + for root_of in declared_roots): + continue + contents = [f for f in os.listdir(directory) if not f.startswith(".")] + present = sorted(f for f in contents if f.endswith(MANIFEST_SUFFIX)) + if present: + # A manifest is here but not under the name discovery looks for, so + # nothing loaded it. Continuing silently meant a typo in the filename + # bypassed every check for that backend: exit 0, `checked: []`, no + # errors and no warnings. + issues.append(Issue( + "error", entry, + "has %s but not %s, so it is never discovered; rename it or fix the " + "directory name" % (", ".join(present), entry + MANIFEST_SUFFIX))) + continue + kernel_like = [f for f in contents if f.endswith((".cu", ".cuh"))] + if kernel_like: + issues.append(Issue( + "error", entry, + "has kernel sources (%s) but no %s manifest or parent backend manifest" + % (", ".join(sorted(kernel_like)[:3]), entry + MANIFEST_SUFFIX))) + + for entry, directory, expected in found: + try: + with open(expected, "r", encoding="utf-8") as handle: + manifest = json.load(handle) + except ValueError as error: + issues.append(Issue("error", entry, "%s is not valid JSON: %s" + % (os.path.basename(expected), error))) + continue + if not isinstance(manifest, dict): + issues.append(Issue("error", entry, "%s must contain a JSON object" + % os.path.basename(expected))) + continue + manifest["_directory"] = directory + manifest["_entry"] = entry + manifests.append(manifest) + return manifests, issues + + +def check_manifest(manifest, issues, repo_root=None): + backend = manifest.get("_entry", "?") + directory = manifest.get("_directory", "") + if repo_root is None: + # Best effort when called directly: four levels up from + # /src/backends/cuda/. Callers that know the root should pass + # it, because the flat layout makes this guess wrong. + repo_root = os.path.dirname(os.path.dirname(os.path.dirname( + os.path.dirname(os.path.abspath(directory or "."))))) + + # These were two hand-written lists that had to agree, and they drifted: + # `name` was reported as missing and then read anyway, so a manifest missing + # only `name` raised KeyError out of check_manifest and aborted the checks of + # every other backend, in both the text and --json paths. Deriving the guard + # from the same constant is what stops that recurring. + missing = [key for key in REQUIRED_KEYS if key not in manifest] + for key in missing: + issues.append(Issue("error", backend, "manifest is missing required key %r" % key)) + if missing: + return + + name = manifest["name"] + if not isinstance(name, str) or not name: + issues.append(Issue("error", backend, "name must be a non-empty string")) + elif name != backend: + issues.append(Issue("error", backend, + "name %r does not match directory name %r" % (name, backend))) + + check_abi_version(manifest, issues) + + status = manifest["status"] + if status not in STATUSES: + issues.append(Issue("error", backend, + "status must be one of %s, got %r" % (", ".join(STATUSES), status))) + return + + sources = manifest["sources"] + if not isinstance(sources, list) or not all(isinstance(item, str) for item in sources): + issues.append(Issue("error", backend, "sources must be a list of strings")) + else: + for source in sources: + if os.path.isabs(source) or ".." in source.split("/"): + issues.append(Issue("error", backend, + "source %r must be relative to the backend directory" % source)) + continue + if not os.path.isfile(os.path.join(directory, source)): + issues.append(Issue("error", backend, "declared source %r does not exist" % source)) + + build = manifest["build"] + if not isinstance(build, dict): + issues.append(Issue("error", backend, "build must be an object")) + return + + for key in ("script", "output"): + if not isinstance(build.get(key), str) or not build.get(key): + issues.append(Issue("error", backend, "build.%s must be a non-empty string" % key)) + architectures = build.get("architectures") + if not isinstance(architectures, list) or not architectures \ + or not all(isinstance(item, int) and 50 <= item <= 200 for item in architectures): + issues.append(Issue("error", backend, + "build.architectures must be a non-empty list of compute " + "capabilities, e.g. [89, 90]")) + architectures = None + default_arch = build.get("default_arch") + if default_arch is not None and architectures is not None and default_arch not in architectures: + issues.append(Issue("error", backend, + "build.default_arch %r is not listed in build.architectures %r" + % (default_arch, architectures))) + + # What the kernels require is currently stated only in comments and READMEs + # ("tensor-core kernels need sm_80 or newer"). Declaring it makes the claim + # checkable here rather than discoverable as a build failure on someone + # else's GPU. + min_capability = build.get("min_capability") + if min_capability is not None: + if not isinstance(min_capability, int): + issues.append(Issue("error", backend, + "build.min_capability must be an integer compute capability, " + "e.g. 80")) + elif architectures is not None: + too_low = [item for item in architectures if item < min_capability] + if too_low: + issues.append(Issue("error", backend, + "build.architectures includes %r, below build.min_capability " + "%d; the kernels would not build for that target" + % (too_low, min_capability))) + + # A backend can serve more than one model. Kev and Cua-S1 share the same + # Qwen3.5 backbone, so listing consumers is what makes reuse visible instead + # of a private arrangement between two PRs. + models = manifest.get("models") + if models is not None: + if not isinstance(models, list) or not all(isinstance(item, str) for item in models): + issues.append(Issue("error", backend, "models must be a list of strings")) + else: + for model in models: + if not model.endswith("/") or os.path.isabs(model) or ".." in model.split("/"): + issues.append(Issue("error", backend, + "models entry %r must be a repository-relative directory " + "path ending in '/'" % model)) + elif not os.path.isdir(os.path.join(repo_root, model)): + issues.append(Issue("warning", backend, + "models entry %r does not exist yet; the consumer engine " + "is not in the tree" % model)) + + if status == "planned": + return + + # A non-planned backend must ship the build script it names. + script_path = "" + if isinstance(build.get("script"), str): + script_path = os.path.join(directory, build["script"]) + if not os.path.isfile(script_path): + issues.append(Issue("error", backend, + "build.script %r does not exist" % build["script"])) + else: + _check_build_script(backend, script_path, build, architectures, issues) + # The compile job runs the script directly, so the execute bit must + # be set in git, not only in a local working copy: a fresh clone is + # what CI checks out. + if os.name == "posix" and not os.access(script_path, os.X_OK): + issues.append(Issue("error", backend, + "build.script %r is not executable on the checked-out tree " + "(CI takes that bit from the committed one); the compile job " + "runs it as ./%s. Fix with: git update-index --chmod=+x %s" + % (build["script"], build["script"], script_path))) + + numerics = manifest.get("numerics") + reference = manifest.get("reference") + if status == "validated": + if not isinstance(numerics, dict) or not isinstance(numerics.get("tolerance"), dict) \ + or not numerics["tolerance"]: + issues.append(Issue("error", backend, + "status is validated but numerics.tolerance is missing; a " + "parity claim needs a tolerance declared before comparison")) + if not isinstance(reference, dict) or not reference.get("entrypoint"): + issues.append(Issue("error", backend, + "status is validated but reference.entrypoint is missing")) + if isinstance(reference, dict) and reference.get("entrypoint"): + entrypoint = reference["entrypoint"] + # Type first. `os.path.isabs` and `split` raise TypeError on a list or a + # number, and that exception escaped check_manifest: the CLI printed no + # structured report at all, not even under --json, and every later + # backend went unchecked. + if not isinstance(entrypoint, str): + issues.append(Issue("error", backend, + "reference.entrypoint must be a string, got %s" + % type(entrypoint).__name__)) + entrypoint = "" + elif os.path.isabs(entrypoint) or ".." in entrypoint.split("/"): + issues.append(Issue("error", backend, + "reference.entrypoint must be repository-relative")) + elif entrypoint: + manifest["_repo_relative_entrypoint"] = entrypoint + _check_reference_entrypoint(backend, entrypoint, issues, repo_root) + if isinstance(numerics, dict) and isinstance(numerics.get("tolerance"), dict): + for key, value in numerics["tolerance"].items(): + if key == "note": + continue + if not isinstance(value, (int, float)): + issues.append(Issue("error", backend, + "numerics.tolerance.%s must be a number or a note, got %r" + % (key, value))) + + +def _check_reference_entrypoint(backend, entrypoint, issues, repo_root): + """A declared reference that is not there cannot be run by Tier 2.""" + if not os.path.isfile(os.path.join(repo_root, entrypoint)): + issues.append(Issue("warning", backend, + "reference.entrypoint %r does not exist yet; the Tier-2 GPU job " + "cannot run parity for this backend" % entrypoint)) + + +def _check_build_script(backend, script_path, build, architectures, issues): + try: + with open(script_path, "r", encoding="utf-8") as handle: + text = handle.read() + except OSError as error: + issues.append(Issue("error", backend, "cannot read %s: %s" % (build.get("script"), error))) + return + + parsed = parse_build_script(text) + + declared_output = build.get("output") + if parsed["output"] and declared_output and parsed["output"] != declared_output: + issues.append(Issue("error", backend, + "build.output %r does not match the %r the build script writes" + % (declared_output, parsed["output"]))) + + if architectures is None: + return + + # What nvcc is actually told to build, from the -gencode flags. This is the + # authority: a variable that never reaches one of those flags does not make a + # target reachable. A script that assigns `arch=${2:-89}` and then writes + # `-gencode arch=compute_90,code=sm_90` builds sm_90 only, so a manifest + # declaring sm_89 is false even though an `arch=` variable says 89. + gencode = parsed.get("gencode_architectures") or [] + by_variable = parsed["architectures"] + if gencode and by_variable: + unused = [item for item in by_variable if item not in gencode] + if unused: + issues.append(Issue("error", backend, + "build script assigns %r but only passes %r to nvcc; the " + "assigned value never reaches a -gencode flag, so targets " + "read from variables alone cannot be claimed" + % (by_variable, gencode))) + script_arch = gencode or by_variable or parsed["literal_architectures"] + if not script_arch: + issues.append(Issue("warning", backend, + "build script declares no compute capability; the manifest claims " + "%r but CI cannot confirm the script honours it" % (architectures,))) + return + unbuildable = [item for item in architectures if item not in script_arch] + if unbuildable: + issues.append(Issue("error", backend, + "build.architectures claims %r but the build script only reaches %r " + "(missing %r)" % (architectures, script_arch, unbuildable))) + extra = [item for item in script_arch if item not in architectures] + if extra: + issues.append(Issue("warning", backend, + "build script also targets %r, which build.architectures omits" + % (extra,))) + + +def _declared_abi(manifest): + """Every ``#define *ABI_VERSION N`` in the backend's declared sources. + + Read as text, like the build script: the checker must not need a compiler, + and the macro is a fact the source already states. + """ + directory = manifest.get("_directory", "") + found = [] + for source in manifest.get("sources") or []: + if not isinstance(source, str): + continue + path = os.path.join(directory, source) + if not os.path.isfile(path): + continue # a missing source is already reported separately + try: + with open(path, "r", encoding="utf-8", errors="replace") as handle: + text = handle.read() + except OSError: + continue + for match in _ABI_MACRO.finditer(text): + found.append((os.path.basename(source), match.group(1), int(match.group(2)))) + return found + + +def check_abi_version(manifest, issues): + """``abi_version`` is the library's own, and its header is the authority. + + This is the one manifest field whose source of truth lives in the code, and + the checker could not see it before: nothing read the header, so the only + guard against a stale manifest was the example in contract.md. #19's + ``ops.h`` says ``CS1_ABI_VERSION 4``; a manifest claiming 1 would have been + accepted. + + Two libraries may legitimately differ, so there is deliberately no + cross-backend sameness requirement: what has to hold is that each manifest + agrees with its own header. + """ + backend = manifest.get("_entry", "?") + abi = manifest.get("abi_version") + if not isinstance(abi, int) or abi < CONTRACT_ABI_VERSION: + issues.append(Issue("error", backend, + "abi_version must be an integer >= %d, got %r" + % (CONTRACT_ABI_VERSION, abi))) + return + + found = _declared_abi(manifest) + if not found: + issues.append(Issue("warning", backend, + "no #define *ABI_VERSION in the declared sources, so abi_version " + "%d cannot be checked against the library" % abi)) + return + + versions = {version for _file, _macro, version in found} + if len(versions) > 1: + issues.append(Issue("error", backend, + "the declared sources define more than one ABI version: %s" + % ", ".join("%s %s=%d" % entry for entry in sorted(found)))) + return + + header_version = versions.pop() + if header_version != abi: + file_name, macro, _value = found[0] + issues.append(Issue("error", backend, + "abi_version is %d but %s defines %s %d; the manifest and the " + "library it describes have to agree" + % (abi, file_name, macro, header_version))) + + +def run(repo_root): + manifests, issues = load_manifests(repo_root) + for manifest in manifests: + check_manifest(manifest, issues, repo_root) + return manifests, issues + + +def main(argv=None): + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + parser.add_argument("--repo-root", default=".", + help="repository root to check (default: current directory)") + parser.add_argument("--json", action="store_true", help="emit machine-readable results") + args = parser.parse_args(argv) + + repo_root = os.path.abspath(args.repo_root) + manifests, issues = run(repo_root) + errors = [issue for issue in issues if issue.level == "error"] + warnings = [issue for issue in issues if issue.level == "warning"] + + if args.json: + json.dump({ + "checked": sorted(m.get("_entry", "?") for m in manifests), + "issues": [issue.as_dict() for issue in issues], + "errors": len(errors), + "warnings": len(warnings), + }, sys.stdout, indent=2) + sys.stdout.write("\n") + else: + for issue in issues: + print(issue) + for manifest in manifests: + print("checked %s: status=%s abi=%s" + % (manifest.get("_entry"), manifest.get("status"), manifest.get("abi_version"))) + print("%d backend(s), %d error(s), %d warning(s)" + % (len(manifests), len(errors), len(warnings))) + if not manifests: + print("no backends declared; add src/backends/cuda//.backend.json") + + return 1 if errors else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/src/backends/cuda/contract.md b/src/backends/cuda/contract.md new file mode 100644 index 0000000..bfc80ba --- /dev/null +++ b/src/backends/cuda/contract.md @@ -0,0 +1,249 @@ +# CUDA backend contract + +`src/backends/cuda/` is shared by several model engines. Each model owns its own +operations, its own library and its own numerics; this document fixes only the +parts that have to agree for two model libraries to be built, distributed and +validated the same way. + +The rule from the [repository layout](../../../README.md) still holds: model +orchestration, batching policy, state management and kernel selection stay with +the model engine. A backend does not need identical internal structures to +another backend, and no universal tensor abstraction is introduced here. + +Status: proposed. The checker is [`check_contract.py`](check_contract.py); wiring it +into CI is a separate change, so nothing here is enforced yet. + +## Why this exists + +Three CUDA efforts are now in flight, each with its own build path: + +| Effort | Kernels | Produces | Built by | +| --- | --- | --- | --- | +| Laya native (#14) | TileLang, exported to CUDA | `liblaya_cuda.so` | `tools/export.py` + `tools/build.py` | +| Cua-S1 native (#19) | CUDA C++ | `libqwen3_5_cuda.so` | `build.sh` | +| Cua-S1 multimodal (#12) | Triton / cuTile | (Python) | — | + +That is three build systems, three library names and three C ABIs. None of it +conflicts today, because nothing links against anything else yet. It conflicts +as soon as one model reuses another's kernels — which is already planned: the +Cua-S1 native worker and a Kev engine share a Qwen3.5 base, and `#9` asks for +CUDA paths for CLM as well. + +`#6` says to extract shared code once two implementations exist. Two exist now. +This document is that extraction, kept to interfaces only. + +## 1. Artifact discovery + +Every CUDA backend declares itself in one JSON file: + +```text +src/backends/cuda//.backend.json +``` + +The checker discovers backends by that path, so adding a backend needs no CI +change. Schema: + +```json +{ + "name": "qwen3_5", + "abi_version": 4, + "status": "validated", + "sources": ["common.cuh", "mma.cuh", "ops.h", "norm.cu", "elementwise.cu", + "attention.cu", "gdn_prefill.cu", "gemm.cu", "runtime.cu"], + "models": ["src/models/cua_s1/", "src/models/kev/"], + "build": { + "script": "build.sh", + "output": "libqwen3_5_cuda.so", + "default_arch": 89, + "architectures": [89], + "min_capability": 80 + }, + "numerics": { + "precision": "bfloat16", + "accumulation": "float32", + "tolerance": { "max_abs": 0.039, "note": "float32 reference, see #11" } + }, + "reference": { + "entrypoint": "recipe/cua_s1/check_native.py", + "note": "pinned upstream FourBModel; kernel tolerances are declared per suite" + } +} +``` + +Required keys are `name`, `abi_version`, `status`, `sources` and `build`. +`status` is one of: + +- `planned` — directory only; the checker skips everything else. +- `experimental` — builds, but no parity claim yet. `numerics.tolerance` may be + omitted. +- `validated` — a parity claim is made. `numerics.tolerance` and + `reference.entrypoint` are required, and the checker fails without them. + +`reference.entrypoint` is a repository-relative script that runs on a GPU and +exits non-zero when parity fails. It takes no contract-defined arguments: a +self-hosted runner that has the weights and the GPU runs it directly. + +### One backend, several models + +`models` lists the model engines that consume this backend, as +repository-relative directories. It is optional, and it exists because reuse is +the point: `#19`'s kernels serve a Qwen3.5-4B backbone, and Kev from `#9` uses +the same backbone with a different adapter and readout, so one backend directory +serves both. Without this field that sharing is a private arrangement between +two PRs and invisible to anyone reading either one. An entry naming a directory +that is not in the tree is a warning, not an error, so a backend can be merged +before its second consumer lands. + +### Build script interface + +`build.script` is invoked by the compile job as: + +```sh +./ +``` + +so it must be executable, accept an output directory as `$1`, and accept a +compute capability as `$2` — defaulting to `build.default_arch` when `$2` is +absent. `#19`'s `build.sh` already has this shape: + +```sh +out=${1:?usage: build.sh [compute capability]} +arch=${2:-${CUDA_COMPUTE_CAP:-89}} +``` + +This is a deliberate narrowing, and it is the first place the two in-flight +build paths diverge. `#19` builds from a shell script that emits one library. +`#14` builds through `tools/export.py` and `tools/build.py`, which are Python +programs driven by a TileLang export step and take a bundle directory, not an +output directory plus an architecture. A backend of that shape satisfies the +contract with a small `build.sh` wrapper that documents the real invocation +rather than by CI growing a second code path. Making that wrapper the required +entry point keeps CI one path and makes "how do I build this backend" answerable +from the manifest alone. + +## 2. C ABI + +A backend library exports a flat `extern "C"` interface and is loaded at run +time, so a Rust engine builds without a CUDA toolkit. `#19`'s `ops.h` is the +reference for the shape. The contract fixes four points: + +1. **One version symbol per library, and the manifest repeats it.** A backend + defines `_ABI_VERSION` in a header and exports + `uint32_t _abi_version(void)` returning it; a loader refuses a library + whose value it does not know. The manifest's `abi_version` is **that number**, + not a version of this document, and the checker reads the macro out of the + declared sources and requires the two to agree. `#19`'s `ops.h` says + `CS1_ABI_VERSION 4` today, so a `qwen3_5` manifest declares 4. + + Two backends may declare different values. They are independent libraries, and + the repository layout says CUDA and Metal implementations need not share + internal structure; forcing one number across models would invent a coupling + nothing needs. What must hold is that each manifest matches its own header. +2. **Errors are `int`, not exceptions.** Every entry point returns `0` on + success. CUDA runtime errors are returned as `cudaError_t` values; anything + the library defines itself starts at `1000`. Every library exports + `const char* _error_string(int)`. +3. **Work is queued on a caller-supplied stream.** Operations take a + `cudaStream_t` as their last argument and must not synchronize internally, so + that a caller can capture them into a CUDA Graph. Allocations, copies, stream + creation and graph capture are exported by the library too, so a caller needs + no direct CUDA linkage. +4. **GEMM algorithm choice is explicit and reproducible.** A tuned plan is + exported and imported rather than re-tuned at load. A plan that reduces + split-K in place must be refused, so a given plan always produces the same + result. + +Point 3 is what makes two model libraries composable: a model engine that +already owns a stream and a captured graph can call into either library. + +### Runtime symbol sharing + +`#19`'s runtime block (`malloc`, `free`, `stream_create`, `stream_sync`, +`upload`, `download`, `graph_begin`, `graph_end`, `graph_launch`, +`graph_destroy`, `device_info`, `set_device`) is the same work every backend +needs. Implementing it per library means two libraries loaded into one process +each carry a copy, and a graph captured through one cannot be launched through +the other. + +Extracting that set into one shared header is a follow-up, deliberately not part +of this change: there is only one backend today, so the shape of the shared +surface would be guesswork. It becomes worth doing when a second backend +actually needs to be loaded alongside the first, and the interface can be +written against two real callers instead of an imagined one. + +## 3. Numerics + +Bit-exactness is not claimed anywhere, so the contract records what varies +instead of leaving it to be discovered in a parity failure: + +- **Rounding points.** Where each kernel rounds to the storage dtype is part of + the kernel's documentation. `#19`'s `qwen3_5/README.md` is the model: it names + the kernels that round where the Transformers reference rounds, and the + attention and Gated DeltaNet paths that keep bfloat16 intermediates as + FlashAttention and flash-linear-attention do. +- **Accumulation.** Float32 unless documented otherwise. +- **Tolerance.** Declared before comparison, per suite, in + `numerics.tolerance`. A single end-to-end number is not enough: an end-to-end + check cannot localize a failure to one kernel. +- **Architecture.** `build.architectures` lists the compute capabilities the + library is built for. A kernels file that requires a newer capability than the + build script's default is a contract error — `#19`'s `mma.cuh` and + `gdn_prefill.cu` both say "sm_80 and later" in their first line, while `#14` + builds sm_90a only. +- **Kernel requirement.** `build.min_capability` states the oldest compute + capability the kernels actually compile for. It exists because that fact is + currently only in prose — `#19`'s `gdn_prefill.cu` and `mma.cuh` both say + "sm_80 and later" in a header comment, and its README says "Tensor-core kernels + need sm_80 or newer". One integer turns that into something CI checks: every + entry in `build.architectures` must be at least `build.min_capability`, so a + library cannot claim a target its own kernels reject. The field is optional; + a backend that omits it declares no verified floor, which is honest but not + checked. + +## 4. Validation + +Two tiers, because they have different requirements. + +**Tier 1 — contract check (no GPU).** `check_contract.py` reads the manifests and +the build scripts and checks: the schema, that every declared source exists, that +a build script declares the architectures the manifest claims and writes the +library the manifest names, that an ABI version is consistent across backends, and +that a `validated` backend declares both a tolerance and a reference entrypoint. +It runs on a stock runner and needs no CUDA toolkit. This is the tier implemented +in this change. + +It does not compile CUDA and does not prove numerics. + +**Tier 2 — compile (no GPU).** Building each backend with `nvcc` in a CUDA +container, so a kernel that stops compiling cannot land. The interface it relies +on is the one Tier 1 already checks: `./ +`, producing exactly the declared library. Not yet wired up; +it needs a container image and belongs with the rest of the CI wiring rather than +with the checker. + +Note what a build check does and does not prove. It shows the sources compile and +that a file of the declared name appears. It does not prove the binary honoured +`build.architectures`: `#19`'s `build.sh` advertises that it "also embeds PTX", +and nothing reads the artifact to confirm which targets are inside. Inspecting +that would need `cuobjdump`, which is a reasonable follow-up rather than +something this change claims to do. + +**Tier 3 — GPU parity (self-hosted).** Running each validated backend's +`reference.entrypoint` on a real GPU, and reporting kernel-level and end-to-end +parity separately. Contributors with a card can attach their own runner to their +own fork, which is how `#19` (RTX 6000 Ada, sm_89), `#12` (RTX 4090) and `#14` +(H800, sm_90a) can each be checked on the hardware they were measured on. Not yet +wired up. + +The split is the point. A CPU-only workflow that claims to validate CUDA is worse +than no workflow, which is the gap `#14`'s own validation report already records: +today's CI checks every Rust feature and compiles no CUDA at all. Tier 1 removes +the part of that gap which does not need hardware; Tiers 2 and 3 need hardware and +are therefore separate. + +## Reporting + +Kernel-level and end-to-end results stay separate, and both report the hardware, +driver, CUDA version, compute capability, source revision and tolerance that +applied. Load, warmup and warm latency are reported separately. A run that does +not establish an improvement reports that, rather than an acceleration claim. diff --git a/src/backends/cuda/tests/__init__.py b/src/backends/cuda/tests/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/src/backends/cuda/tests/test_build_script.py b/src/backends/cuda/tests/test_build_script.py new file mode 100644 index 0000000..1d04fd5 --- /dev/null +++ b/src/backends/cuda/tests/test_build_script.py @@ -0,0 +1,138 @@ +#!/usr/bin/env python3 +"""Tests for reading a build script as text. + +A parser validated only against invented samples proves nothing about the scripts +it has to read, so the main case here is the real ``qwen3_5/build.sh`` added by +#19, and the real ``build.sh`` the Laya backend wrapper exposes. + +Nothing is executed: these are the parsing rules CI uses to decide whether a +manifest is honest. +""" + +from __future__ import annotations + +import os +import sys +import unittest + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import build_script # noqa: E402 (path is set above) + +# The argument handling of src/backends/cuda/qwen3_5/build.sh in #19: `$1` is the +# output directory, `$2` defaults through CUDA_COMPUTE_CAP to 89. +QWEN3_5_BUILD_SCRIPT = """#!/usr/bin/env bash +# Build libqwen3_5_cuda.so from the kernels in this directory. +# +# src/backends/cuda/qwen3_5/build.sh [compute capability, e.g. 89] +set -euo pipefail +here=$(cd "$(dirname "$0")" && pwd) +out=${1:?usage: build.sh [compute capability]} +arch=${2:-${CUDA_COMPUTE_CAP:-89}} +mkdir -p "$out" +"$nvcc" -O3 -std=c++17 -gencode "arch=compute_${arch},code=[sm_${arch},compute_${arch}]" \\ + -shared -Xcompiler -fPIC -I"$here" "$here"/*.cu \\ + "${link[@]}" -o "$out/libqwen3_5_cuda.so" +echo "built $out/libqwen3_5_cuda.so for sm_${arch}" +""" + + +class StripCommentTest(unittest.TestCase): + def test_a_trailing_comment_is_removed(self): + self.assertEqual(build_script._strip_comment("arch=89 # default"), "arch=89 ") + + def test_a_hash_inside_quotes_is_kept(self): + self.assertEqual(build_script._strip_comment('x="a # b"'), 'x="a # b"') + + def test_a_comment_at_the_start_is_removed(self): + self.assertEqual(build_script._strip_comment("# sm_80 and later"), "") + + def test_a_hash_not_preceded_by_space_is_kept(self): + # Not a comment in shell: `a#b` is one word. + self.assertEqual(build_script._strip_comment("a#b"), "a#b") + + +class ParseBuildScriptTest(unittest.TestCase): + def test_reads_the_real_qwen3_5_script(self): + parsed = build_script.parse_build_script(QWEN3_5_BUILD_SCRIPT) + self.assertEqual(parsed["output"], "libqwen3_5_cuda.so") + self.assertIn(89, parsed["architectures"]) + # `compute_${arch}` is a template, not a literal, so no architecture + # number is invented from it. + self.assertEqual(parsed["literal_architectures"], []) + + def test_literal_gencode_is_read_when_no_variable_is_used(self): + parsed = build_script.parse_build_script( + "nvcc -gencode arch=compute_90,code=sm_90 -shared -o libx.so a.cu\n") + self.assertEqual(parsed["literal_architectures"], [90]) + self.assertEqual(parsed["output"], "libx.so") + + def test_ignores_architectures_in_comments(self): + parsed = build_script.parse_build_script( + "# works on sm_120\nout=x\nnvcc -o liby.so a.cu\n") + self.assertEqual(parsed["literal_architectures"], []) + self.assertEqual(parsed["output"], "liby.so") + + def test_a_quoted_output_path_with_a_variable_prefix(self): + # `-o "$out/libx.so"`: the directory is unknowable, the filename is not. + parsed = build_script.parse_build_script( + 'out=${1:?usage}\nnvcc -shared -o "$out/libqwen3_5_cuda.so" ./*.cu\n') + self.assertEqual(parsed["output"], "libqwen3_5_cuda.so") + + def test_nested_defaults_are_expanded(self): + parsed = build_script.parse_build_script( + "arch=${2:-${CUDA_COMPUTE_CAP:-89}}\nnvcc -gencode \"x=sm_${arch}\" -o l.so a.cu\n") + self.assertEqual(parsed["architectures"], [89]) + + def test_every_declaration_is_collected_not_only_the_last(self): + # A script that builds two targets assigns arch twice; both count, since + # the claim is about what the script can reach. + parsed = build_script.parse_build_script( + "arch=90\nnvcc -o a.so x.cu\narch=89\nnvcc -o a.so x.cu\n") + self.assertEqual(sorted(parsed["architectures"]), [89, 90]) + + def test_a_script_declaring_nothing_reports_nothing(self): + parsed = build_script.parse_build_script("#!/usr/bin/env bash\nnvcc -o l.so ./*.cu\n") + self.assertEqual(parsed["architectures"], []) + self.assertEqual(parsed["literal_architectures"], []) + self.assertEqual(parsed["output"], "l.so") + + def test_the_last_output_wins(self): + parsed = build_script.parse_build_script( + "nvcc -o first.so a.cu\nnvcc -o second.so b.cu\n") + self.assertEqual(parsed["output"], "second.so") + + def test_a_trailing_quote_does_not_end_up_in_the_name(self): + parsed = build_script.parse_build_script('nvcc -o "libx.so" a.cu\n') + self.assertEqual(parsed["output"], "libx.so") + + def test_gencode_targets_are_read_from_the_flags_not_the_variables(self): + parsed = build_script.parse_build_script( + "out=${1:?}\narch=${2:-89}\n" + 'nvcc -gencode "arch=compute_90,code=sm_90" -shared ' + '-o "$out/libx.so" ./*.cu\n') + self.assertEqual(parsed["architectures"], [89]) + self.assertEqual(parsed["gencode_architectures"], [90]) + + def test_each_gencode_flag_resolves_against_the_value_in_force(self): + # A loop that reassigns `arch` builds both targets; resolving every line + # against the final value would report 90 twice and lose 89. + parsed = build_script.parse_build_script( + "out=${1:?}\n" + 'arch=89\nnvcc -gencode "arch=compute_${arch},code=sm_${arch}" -o "$out/a.so" x.cu\n' + 'arch=90\nnvcc -gencode "arch=compute_${arch},code=sm_${arch}" -o "$out/a.so" x.cu\n') + self.assertEqual(parsed["gencode_architectures"], [89, 90]) + + def test_a_script_with_no_gencode_reports_no_gencode_targets(self): + parsed = build_script.parse_build_script("nvcc -shared -o l.so ./*.cu\n") + self.assertEqual(parsed["gencode_architectures"], []) + + def test_the_parser_never_executes_the_script(self): + # A script whose first command would be destructive must still be read. + parsed = build_script.parse_build_script( + "rm -rf /\nout=${1}\nnvcc -o \"$out/libz.so\" a.cu\n") + self.assertEqual(parsed["output"], "libz.so") + + +if __name__ == "__main__": + unittest.main(verbosity=2) diff --git a/src/backends/cuda/tests/test_check_contract.py b/src/backends/cuda/tests/test_check_contract.py new file mode 100644 index 0000000..553bd33 --- /dev/null +++ b/src/backends/cuda/tests/test_check_contract.py @@ -0,0 +1,636 @@ +#!/usr/bin/env python3 +"""Tests for the Tier-1 CUDA backend contract check. + +No GPU, no nvcc, no CUDA toolkit. Run from the repository root: + + python3 -m unittest discover -s src/backends/cuda/tests -v + python3 src/backends/cuda/tests/test_check_contract.py + +The build-script case uses the real text of the ``qwen3_5/build.sh`` added by +PR #19, because a parser validated only against invented samples proves nothing +about the script it has to read. +""" + +from __future__ import annotations + +import contextlib +import io +import json +import os +import shutil +import sys +import tempfile +import unittest + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +import check_contract # noqa: E402 (path is set above) + +# The argument handling of src/backends/cuda/qwen3_5/build.sh in PR #19: `$1` is +# the output directory and `$2` defaults through CUDA_COMPUTE_CAP to 89. +QWEN3_5_BUILD_SCRIPT = """#!/usr/bin/env bash +# Build libqwen3_5_cuda.so from the kernels in this directory. +set -euo pipefail +here=$(cd "$(dirname "$0")" && pwd) +out=${1:?usage: build.sh [compute capability]} +arch=${2:-${CUDA_COMPUTE_CAP:-89}} +mkdir -p "$out" +"$nvcc" -O3 -std=c++17 -gencode "arch=compute_${arch},code=[sm_${arch},compute_${arch}]" \\ + -shared -Xcompiler -fPIC -I"$here" "$here"/*.cu \\ + "${link[@]}" -o "$out/libqwen3_5_cuda.so" +echo "built $out/libqwen3_5_cuda.so for sm_${arch}" +""" + +KERNELS_CU = """ +#include "ops.h" +extern "C" int fake_kernel(const void* x, int n); +""" + + +def manifest(**overrides): + """A minimal schema-complete manifest, matching PR #19's qwen3_5 backend. + + ``build.architectures`` is ``[89]`` because that is the whole truth about + this build script: one invocation produces machine code for one compute + capability, and PTX for newer parts to compile at load time. + """ + value = { + "name": "qwen3_5", + "abi_version": 1, + "status": "validated", + "sources": ["ops.h", "kernels.cu"], + "build": { + "script": "build.sh", + "output": "libqwen3_5_cuda.so", + "default_arch": 89, + "architectures": [89], + }, + "numerics": { + "precision": "bfloat16", + "accumulation": "float32", + "tolerance": {"max_abs": 0.039}, + }, + "reference": {"entrypoint": "recipe/cua_s1/check_native.py"}, + } + value.update(overrides) + return value + + +class CheckManifestTest(unittest.TestCase): + def setUp(self): + self.root = tempfile.mkdtemp(prefix="omni-contract-") + self.directory = os.path.join(self.root, "src", "backends", "cuda", "qwen3_5") + os.makedirs(self.directory) + with open(os.path.join(self.directory, "kernels.cu"), "w", encoding="utf-8") as handle: + handle.write(KERNELS_CU) + with open(os.path.join(self.directory, "ops.h"), "w", encoding="utf-8") as handle: + handle.write("#pragma once\n#define TEST_ABI_VERSION 1\n") + with open(os.path.join(self.directory, "build.sh"), "w", encoding="utf-8") as handle: + handle.write(QWEN3_5_BUILD_SCRIPT) + os.chmod(os.path.join(self.directory, "build.sh"), 0o755) + # The validated manifest declares this reference, so the fixture ships it. + recipe = os.path.join(self.root, "recipe", "cua_s1") + os.makedirs(recipe) + with open(os.path.join(recipe, "check_native.py"), "w", encoding="utf-8") as handle: + handle.write("import sys\nsys.exit(0)\n") + + def tearDown(self): + shutil.rmtree(self.root, ignore_errors=True) + + def write_manifest(self, value): + path = os.path.join(self.directory, "qwen3_5.backend.json") + with open(path, "w", encoding="utf-8") as handle: + json.dump(value, handle) + + def check(self): + manifests, issues = check_contract.run(self.root) + return manifests, [issue for issue in issues if issue.level == "error"], \ + [issue for issue in issues if issue.level == "warning"] + + def messages(self, issues): + return " | ".join(issue.message for issue in issues) + + def test_a_complete_manifest_passes(self): + self.write_manifest(manifest()) + manifests, errors, warnings = self.check() + self.assertEqual(len(manifests), 1) + self.assertEqual(errors, [], self.messages(errors)) + self.assertEqual(warnings, [], self.messages(warnings)) + + def test_kernels_without_a_manifest_are_an_error(self): + manifests, errors, _ = self.check() + self.assertEqual(manifests, []) + self.assertEqual(len(errors), 1) + self.assertIn("no qwen3_5.backend.json manifest", errors[0].message) + + def test_every_required_key_is_reported_and_then_guarded(self): + """Removing any required key must produce a report, never an exception. + + The guard and the report were two hand-written lists, and `name` was in + one but not the other: a manifest missing only `name` raised KeyError out + of check_manifest, which also aborted the checks of every other backend + and broke --json. Iterating over the keys rather than testing one of them + is what covers the whole class. + """ + for key in check_contract.REQUIRED_KEYS: + value = manifest() + del value[key] + self.write_manifest(value) + try: + _, errors, _ = self.check() + except KeyError as error: + self.fail("removing %r raised %s instead of reporting it" % (key, error)) + self.assertIn("missing required key %r" % key, self.messages(errors)) + + def test_a_manifest_missing_name_does_not_stop_other_backends(self): + # The failure mode the maintainer reproduced: one bad manifest must not + # hide the results for the rest. + value = manifest() + del value["name"] + self.write_manifest(value) + other = os.path.join(self.root, "src", "backends", "cuda", "other") + os.makedirs(other) + with open(os.path.join(other, "k.cu"), "w", encoding="utf-8") as handle: + handle.write("// kernel\n") + with open(os.path.join(other, "other.backend.json"), "w", encoding="utf-8") as handle: + json.dump({"name": "other", "abi_version": 1, "status": "planned", + "sources": ["k.cu"], + "build": {"script": "build.sh", "output": "libother.so", + "default_arch": 89, "architectures": [89]}}, handle) + manifests, issues = check_contract.run(self.root) + # A manifest missing `name` is still discovered and still listed; it + # is the entry key that identifies it, which is why the checker + # reports against the directory name rather than the missing field. + self.assertEqual(sorted(m["_entry"] for m in manifests), ["other", "qwen3_5"]) + self.assertTrue(any("missing required key 'name'" in i.message for i in issues)) + + def test_missing_required_key_is_an_error(self): + value = manifest() + del value["sources"] + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("missing required key 'sources'", self.messages(errors)) + + def test_declared_source_that_does_not_exist_is_an_error(self): + self.write_manifest(manifest(sources=["ops.h", "gdn_prefill.cu"])) + _, errors, _ = self.check() + self.assertIn("declared source 'gdn_prefill.cu' does not exist", self.messages(errors)) + + def test_source_escaping_the_backend_directory_is_an_error(self): + self.write_manifest(manifest(sources=["ops.h", "../../../etc/passwd"])) + _, errors, _ = self.check() + self.assertIn("must be relative to the backend directory", self.messages(errors)) + + def test_output_name_must_match_the_build_script(self): + value = manifest() + value["build"]["output"] = "libwrong.so" + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("does not match the 'libqwen3_5_cuda.so' the build script writes", + self.messages(errors)) + + def test_architecture_the_script_cannot_reach_is_an_error(self): + # The build script defaults one architecture and offers no way to reach + # another, so a manifest claiming [89, 90] is over-claiming. + value = manifest() + value["build"]["architectures"] = [89, 90] + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("build.architectures claims [89, 90] but the build script only reaches [89]", + self.messages(errors)) + + def test_two_architectures_in_one_script_are_accepted(self): + # The architectures have to be visible in the script. A `for arch in ...` + # loop alone is not read as a declaration, so the script names them. + script = """#!/usr/bin/env bash +set -euo pipefail +out=${1:?usage: build.sh } +arch=89 +"$nvcc" -gencode "arch=compute_${arch},code=sm_${arch}" -shared \\ + -o "$out/libmulti_cuda_${arch}.so" ./*.cu +arch=90 +"$nvcc" -gencode "arch=compute_${arch},code=sm_${arch}" -shared \\ + -o "$out/libmulti_cuda_${arch}.so" ./*.cu +""" + with open(os.path.join(self.directory, "build.sh"), "w", encoding="utf-8") as handle: + handle.write(script) + value = manifest() + value["build"]["architectures"] = [89, 90] + # The script's last assignment is arch=90, so that is the library a run + # produces; the manifest has to name the same file. + value["build"]["output"] = "libmulti_cuda_90.so" + self.write_manifest(value) + _, errors, _ = self.check() + self.assertEqual(errors, [], self.messages(errors)) + + def test_warns_when_a_script_declares_no_compute_capability(self): + with open(os.path.join(self.directory, "build.sh"), "w", encoding="utf-8") as handle: + handle.write('#!/usr/bin/env bash\nnvcc -shared -o libqwen3_5_cuda.so ./*.cu\n') + self.write_manifest(manifest()) + _, errors, warnings = self.check() + self.assertEqual(errors, [], self.messages(errors)) + self.assertIn("declares no compute capability", self.messages(warnings)) + + def test_min_capability_covered_by_the_architectures_passes(self): + # The real #19 case: kernels documented as sm_80-and-later, built sm_89. + value = manifest() + value["build"]["min_capability"] = 80 + self.write_manifest(value) + _, errors, _ = self.check() + self.assertEqual(errors, [], self.messages(errors)) + + def test_an_architecture_below_the_kernel_requirement_is_an_error(self): + value = manifest() + value["build"]["min_capability"] = 90 + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("below build.min_capability 90", self.messages(errors)) + + def test_a_non_integer_min_capability_is_an_error(self): + value = manifest() + value["build"]["min_capability"] = "80" + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("must be an integer compute capability", self.messages(errors)) + + def test_listing_a_consumer_engine_that_exists_passes(self): + # Kev and Cua-S1 share the Qwen3.5 backbone, so one backend can serve + # both without either model owning a private copy of the kernels. + os.makedirs(os.path.join(self.root, "src", "models", "kev")) + value = manifest(models=["src/models/cua_s1/", "src/models/kev/"]) + self.write_manifest(value) + _, errors, warnings = self.check() + self.assertEqual(errors, [], self.messages(errors)) + self.assertIn("src/models/cua_s1/", self.messages(warnings)) + + def test_a_consumer_path_that_is_not_a_directory_is_an_error(self): + value = manifest(models=["src/models/kev"]) + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("must be a repository-relative directory path", self.messages(errors)) + + def test_a_non_list_models_field_is_an_error(self): + value = manifest(models="src/models/kev/") + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("models must be a list of strings", self.messages(errors)) + + def test_min_capability_is_optional(self): + # Omitting it must not fail: not every backend states a requirement yet. + self.write_manifest(manifest()) + _, errors, _ = self.check() + self.assertEqual(errors, [], self.messages(errors)) + + def test_default_arch_outside_the_list_is_an_error(self): + value = manifest() + value["build"]["default_arch"] = 75 + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("is not listed in build.architectures", self.messages(errors)) + + def test_validated_without_a_tolerance_is_an_error(self): + value = manifest() + del value["numerics"]["tolerance"] + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("status is validated but numerics.tolerance is missing", + self.messages(errors)) + + def test_validated_without_a_reference_entrypoint_is_an_error(self): + value = manifest() + del value["reference"] + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("status is validated but reference.entrypoint is missing", + self.messages(errors)) + + def test_experimental_needs_neither_tolerance_nor_entrypoint(self): + self.write_manifest(manifest(status="experimental", numerics={}, reference={})) + _, errors, warnings = self.check() + self.assertEqual(errors, [], self.messages(errors)) + self.assertEqual(warnings, [], self.messages(warnings)) + + def test_planned_backend_is_not_required_to_build(self): + # `planned` means the directory is declared but nothing is built yet, so + # a build script that does not exist is fine. The schema still applies. + self.write_manifest(manifest(status="planned", sources=["ops.h"], build={ + "script": "build.sh", "output": "libqwen3_5_cuda.so", + "default_arch": 89, "architectures": [89], + })) + _, errors, _ = self.check() + self.assertEqual(errors, [], self.messages(errors)) + + def test_a_non_executable_build_script_is_an_error(self): + # The compile job runs `./build.sh`, so the execute bit is part of the + # contract, not a local detail. + os.chmod(os.path.join(self.directory, "build.sh"), 0o644) + self.write_manifest(manifest()) + _, errors, _ = self.check() + self.assertIn("is not executable", self.messages(errors)) + + def test_a_non_string_reference_entrypoint_is_reported_not_raised(self): + # A list here reached os.path.isabs, whose TypeError escaped + # check_manifest: no report at all, not even --json, and every later + # backend went unchecked. + self.write_manifest(manifest(reference={"entrypoint": ["reference.py"]})) + _, errors, _ = self.check() + self.assertIn("reference.entrypoint must be a string, got list", + self.messages(errors)) + + def test_an_empty_reference_entrypoint_is_reported(self): + self.write_manifest(manifest(reference={"entrypoint": ""})) + _, errors, _ = self.check() + self.assertIn("status is validated but reference.entrypoint is missing", + self.messages(errors)) + + def test_an_assigned_architecture_never_passed_to_nvcc_is_an_error(self): + # The variable says 89; the compiler is told 90. Reading the variable + # alone accepted a manifest declaring sm_89 that the script cannot build. + script = ("#!/usr/bin/env bash\n" + "out=${1:?usage}\n" + "arch=${2:-89}\n" + 'nvcc -gencode "arch=compute_90,code=sm_90" ' + '-shared -o "$out/libqwen3_5_cuda.so" ./*.cu\n') + with open(os.path.join(self.directory, "build.sh"), "w", encoding="utf-8") as handle: + handle.write(script) + os.chmod(os.path.join(self.directory, "build.sh"), 0o755) + self.write_manifest(manifest()) + _, errors, _ = self.check() + self.assertIn("never reaches a -gencode flag", self.messages(errors)) + + def test_a_misnamed_manifest_is_an_error_not_an_exemption(self): + # Discovery looks for /.backend.json. A typo there used to be + # exempted silently: exit 0, checked: [], no errors. + self.write_manifest(manifest()) + os.rename(os.path.join(self.directory, "qwen3_5.backend.json"), + os.path.join(self.directory, "typo.backend.json")) + manifests, issues = check_contract.run(self.root) + self.assertEqual(manifests, []) + messages = " | ".join(issue.message for issue in issues) + self.assertIn("has typo.backend.json but not qwen3_5.backend.json", messages) + self.assertIn("never discovered", messages) + + def test_unknown_status_is_an_error(self): + self.write_manifest(manifest(status="done")) + _, errors, _ = self.check() + self.assertIn("status must be one of", self.messages(errors)) + + def test_name_must_match_the_directory(self): + self.write_manifest(manifest(name="laya")) + _, errors, _ = self.check() + self.assertIn("does not match directory name 'qwen3_5'", self.messages(errors)) + + def test_non_numeric_tolerance_is_an_error(self): + value = manifest() + value["numerics"]["tolerance"] = {"max_abs": "small"} + self.write_manifest(value) + _, errors, _ = self.check() + self.assertIn("numerics.tolerance.max_abs must be a number", self.messages(errors)) + + def test_reference_entrypoint_that_is_not_there_warns(self): + # The fixture ships the reference; a declared-but-missing one must warn, + # because Tier 2 would have nothing to run. + os.remove(os.path.join(self.root, "recipe", "cua_s1", "check_native.py")) + self.write_manifest(manifest()) + _, errors, warnings = self.check() + self.assertEqual(errors, [], self.messages(errors)) + self.assertIn("reference.entrypoint", self.messages(warnings)) + + def test_absolute_reference_entrypoint_is_an_error(self): + self.write_manifest(manifest(reference={"entrypoint": "/tmp/check.py"})) + _, errors, _ = self.check() + self.assertIn("must be repository-relative", self.messages(errors)) + + def test_invalid_json_is_reported_not_raised(self): + with open(os.path.join(self.directory, "qwen3_5.backend.json"), "w", + encoding="utf-8") as handle: + handle.write("{not json") + manifests, errors, _ = self.check() + self.assertEqual(manifests, []) + self.assertIn("is not valid JSON", self.messages(errors)) + + +class AbiVersionTest(unittest.TestCase): + """`abi_version` is the library's own, taken from its header. + + Read as text, so the checker needs no compiler. What has to hold is that a + manifest agrees with the header of the library it describes. Two libraries + may differ from each other, so there is deliberately no cross-backend + sameness rule: the old code required one, which would have forced unrelated + model engines onto a shared interface version. + """ + + def setUp(self): + self.root = tempfile.mkdtemp(prefix="omni-abi-") + self.directory = os.path.join(self.root, "src", "backends", "cuda") + os.makedirs(self.directory) + + def tearDown(self): + shutil.rmtree(self.root, ignore_errors=True) + + def add_backend(self, name, abi_version, macro=None, extra_source=None): + directory = os.path.join(self.directory, name) + os.makedirs(directory) + header = "#pragma once\n" + if macro is not None: + header += "#define %s_ABI_VERSION %d\n" % (name.upper(), macro) + with open(os.path.join(directory, "ops.h"), "w", encoding="utf-8") as handle: + handle.write(header) + sources = ["ops.h"] + if extra_source is not None: + with open(os.path.join(directory, "extra.h"), "w", encoding="utf-8") as handle: + handle.write("#define OTHER_ABI_VERSION %d\n" % extra_source) + sources.append("extra.h") + with open(os.path.join(directory, "build.sh"), "w", encoding="utf-8") as handle: + handle.write("#!/usr/bin/env bash\nout=${1:?}\narch=${2:-89}\n" + 'nvcc -o "$out/lib%s.so" ./*.cu\n' % name) + os.chmod(os.path.join(directory, "build.sh"), 0o755) + value = { + "name": name, + "abi_version": abi_version, + "status": "planned", + "sources": sources, + "build": {"script": "build.sh", "output": "lib%s.so" % name, + "default_arch": 89, "architectures": [89]}, + } + with open(os.path.join(directory, name + ".backend.json"), "w", + encoding="utf-8") as handle: + json.dump(value, handle) + + def check(self, root=None): + _, issues = check_contract.run(root or self.root) + errors = [i.message for i in issues if i.level == "error"] + warnings = [i.message for i in issues if i.level == "warning"] + return errors, warnings + + def test_a_manifest_agreeing_with_its_header_passes(self): + self.add_backend("qwen3_5", 4, macro=4) + errors, warnings = self.check() + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + + def test_a_manifest_disagreeing_with_its_header_is_an_error(self): + # The case nothing caught before: #19's ops.h says 4, and a manifest + # claiming 1 was accepted because no check read the header. + self.add_backend("qwen3_5", 1, macro=4) + errors, _ = self.check() + self.assertEqual(len(errors), 1) + self.assertIn("abi_version is 1 but ops.h defines QWEN3_5_ABI_VERSION 4", errors[0]) + + def test_sources_without_an_abi_macro_warn(self): + self.add_backend("qwen3_5", 1, macro=None) + errors, warnings = self.check() + self.assertEqual(errors, []) + self.assertIn("cannot be checked against the library", warnings[0]) + + def test_two_versions_in_one_backends_sources_are_an_error(self): + self.add_backend("qwen3_5", 4, macro=4, extra_source=7) + errors, _ = self.check() + self.assertEqual(len(errors), 1) + self.assertIn("more than one ABI version", errors[0]) + + def test_abi_version_below_the_contract_is_an_error(self): + self.add_backend("qwen3_5", 0, macro=0) + errors, _ = self.check() + self.assertIn("abi_version must be an integer >= 1", " | ".join(errors)) + + def test_two_backends_may_declare_different_versions(self): + # Deliberately allowed now. Requiring them to match would force unrelated + # model engines onto one interface version, which the repository layout + # explicitly does not ask for. + self.add_backend("qwen3_5", 4, macro=4) + self.add_backend("laya", 2, macro=2) + errors, warnings = self.check() + self.assertEqual(errors, []) + self.assertEqual(warnings, []) + + def test_a_planned_backend_is_still_checked(self): + self.add_backend("qwen3_5", 9, macro=4) + errors, _ = self.check() + self.assertIn("abi_version is 9", " | ".join(errors)) + + +class RepositoryStateTest(unittest.TestCase): + """The checker must pass on a repository that satisfies the contract. + + The fixture mirrors the qwen3_5 backend as PR #19 defines it, including a + build script with the same argument handling, so this exercises discovery, + script parsing, schema checks and the ABI rule together. + """ + + def setUp(self): + self.root = tempfile.mkdtemp(prefix="omni-repo-") + self.directory = os.path.join(self.root, "src", "backends", "cuda", "qwen3_5") + os.makedirs(self.directory) + with open(os.path.join(self.directory, "kernels.cu"), "w", encoding="utf-8") as handle: + handle.write(KERNELS_CU) + with open(os.path.join(self.directory, "ops.h"), "w", encoding="utf-8") as handle: + handle.write("#pragma once\n#define TEST_ABI_VERSION 1\n") + with open(os.path.join(self.directory, "build.sh"), "w", encoding="utf-8") as handle: + handle.write(QWEN3_5_BUILD_SCRIPT) + os.chmod(os.path.join(self.directory, "build.sh"), 0o755) + recipe = os.path.join(self.root, "recipe", "cua_s1") + os.makedirs(recipe) + with open(os.path.join(recipe, "check_native.py"), "w", encoding="utf-8") as handle: + handle.write("import sys\nsys.exit(0)\n") + with open(os.path.join(self.directory, "qwen3_5.backend.json"), "w", + encoding="utf-8") as handle: + json.dump(manifest(), handle) + + def tearDown(self): + shutil.rmtree(self.root, ignore_errors=True) + + def test_a_contract_satisfying_repository_has_no_findings(self): + manifests, issues = check_contract.run(self.root) + self.assertEqual([issue.message for issue in issues], []) + self.assertEqual([m["name"] for m in manifests], ["qwen3_5"]) + + def test_the_repository_under_test_is_discovered_by_the_script(self): + """Running the module as CI does must succeed on this fixture.""" + import contextlib + import io + + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + exit_code = check_contract.main(["--repo-root", self.root, "--json"]) + self.assertEqual(exit_code, 0) + self.assertIn('"errors": 0', buffer.getvalue()) + + +class FlatLayoutTest(unittest.TestCase): + """A backend whose files sit directly under src/backends/cuda/. + + Laya's kernels and tools are `src/backends/cuda/kernels/` and + `src/backends/cuda/tools/` rather than a `/` subdirectory, so the + manifest sits beside them as `laya.backend.json`. The layout is the model + author's call; only the manifest's contents are fixed. + """ + + def setUp(self): + self.root = tempfile.mkdtemp(prefix="omni-flat-") + self.cuda = os.path.join(self.root, "src", "backends", "cuda") + os.makedirs(os.path.join(self.cuda, "kernels")) + os.makedirs(os.path.join(self.cuda, "tools")) + for path in ("kernels/runtime.cu", "kernels/model_ops.cu"): + with open(os.path.join(self.cuda, path), "w", encoding="utf-8") as handle: + handle.write("// kernel\n") + with open(os.path.join(self.cuda, "ops.h"), "w", encoding="utf-8") as handle: + handle.write("#pragma once\n#define LAYA_ABI_VERSION 1\n") + with open(os.path.join(self.cuda, "build.sh"), "w", encoding="utf-8") as handle: + handle.write("#!/usr/bin/env bash\n" + "out=${1:?usage}\n" + "arch=${2:-90}\n" + 'nvcc -gencode "arch=compute_${arch},code=sm_${arch}" ' + '-shared -o "$out/liblaya_cuda.so" ./*.cu\n') + os.chmod(os.path.join(self.cuda, "build.sh"), 0o755) + os.makedirs(os.path.join(self.root, "recipe", "laya")) + with open(os.path.join(self.root, "recipe", "laya", "check.py"), "w", + encoding="utf-8") as handle: + handle.write("import sys\nsys.exit(0)\n") + manifest = { + "name": "laya", + "abi_version": 1, + "status": "experimental", + "sources": ["ops.h", "kernels/runtime.cu", "kernels/model_ops.cu"], + "build": { + "script": "build.sh", + "output": "liblaya_cuda.so", + "default_arch": 90, + "architectures": [90], + "min_capability": 90, + }, + } + with open(os.path.join(self.cuda, "laya.backend.json"), "w", encoding="utf-8") as handle: + json.dump(manifest, handle) + + def tearDown(self): + shutil.rmtree(self.root, ignore_errors=True) + + def test_a_flat_backend_is_discovered(self): + manifests, issues = check_contract.run(self.root) + self.assertEqual([m["name"] for m in manifests], ["laya"]) + self.assertEqual([i.message for i in issues], []) + + def test_its_sources_are_resolved_against_the_cuda_directory(self): + manifests, _ = check_contract.run(self.root) + self.assertEqual(manifests[0]["_directory"], + os.path.join(self.root, "src", "backends", "cuda")) + + def test_a_validated_flat_backend_finds_its_reference(self): + # The repo root must not be inferred by walking up from the backend + # directory: for the flat layout that lands one level too high. + path = os.path.join(self.cuda, "laya.backend.json") + with open(path, encoding="utf-8") as handle: + manifest = json.load(handle) + manifest["status"] = "validated" + manifest["numerics"] = {"tolerance": {"max_abs": 0.002}} + manifest["reference"] = {"entrypoint": "recipe/laya/check.py"} + with open(path, "w", encoding="utf-8") as handle: + json.dump(manifest, handle) + _, issues = check_contract.run(self.root) + self.assertEqual([i.message for i in issues], [], + "an existing reference must not be reported missing") + + +if __name__ == "__main__": + unittest.main(verbosity=2)