From 74c95e5817babe1f2906a7a4ed88f0e23c67d4da Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Sun, 27 Sep 2026 20:08:06 +0800 Subject: [PATCH 1/4] Add Cua-S1 multimodal CUDA worker and parity recipe --- .github/workflows/cua-s1-multimodal.yml | 30 + .gitignore | 7 + README.md | 8 +- recipe/cua_s1/README.md | 145 +++++ recipe/cua_s1/check_frontend.py | 62 ++ recipe/cua_s1/download_weights.py | 33 ++ recipe/cua_s1/evaluate_multimodal.py | 321 +++++++++++ recipe/cua_s1/experiments/README.md | 111 ++++ .../cua_s1/experiments/rtx4090/benchmark.json | 409 +++++++++++++ .../cua_s1/experiments/rtx4090/candidate.json | 536 ++++++++++++++++++ .../cua_s1/experiments/rtx4090/frontend.json | 79 +++ .../cua_s1/experiments/rtx4090/packages.txt | 62 ++ .../cua_s1/experiments/rtx4090/reference.json | 527 +++++++++++++++++ recipe/cua_s1/make_example.py | 71 +++ recipe/cua_s1/requirements-multimodal.txt | 11 + .../cua_s1/multimodal/THIRD_PARTY_NOTICES.md | 21 + src/models/cua_s1/multimodal/model.py | 164 ++++++ src/models/cua_s1/multimodal/protocol.py | 217 +++++++ src/models/cua_s1/multimodal/server.py | 128 +++++ .../cua_s1/multimodal/weights.lock.json | 102 ++++ tests/cua_s1/test_evaluation.py | 48 ++ tests/cua_s1/test_model.py | 112 ++++ tests/cua_s1/test_protocol.py | 156 +++++ tests/cua_s1/test_server.py | 85 +++ 24 files changed, 3442 insertions(+), 3 deletions(-) create mode 100644 .github/workflows/cua-s1-multimodal.yml create mode 100644 recipe/cua_s1/README.md create mode 100644 recipe/cua_s1/check_frontend.py create mode 100644 recipe/cua_s1/download_weights.py create mode 100644 recipe/cua_s1/evaluate_multimodal.py create mode 100644 recipe/cua_s1/experiments/README.md create mode 100644 recipe/cua_s1/experiments/rtx4090/benchmark.json create mode 100644 recipe/cua_s1/experiments/rtx4090/candidate.json create mode 100644 recipe/cua_s1/experiments/rtx4090/frontend.json create mode 100644 recipe/cua_s1/experiments/rtx4090/packages.txt create mode 100644 recipe/cua_s1/experiments/rtx4090/reference.json create mode 100644 recipe/cua_s1/make_example.py create mode 100644 recipe/cua_s1/requirements-multimodal.txt create mode 100644 src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md create mode 100644 src/models/cua_s1/multimodal/model.py create mode 100644 src/models/cua_s1/multimodal/protocol.py create mode 100644 src/models/cua_s1/multimodal/server.py create mode 100644 src/models/cua_s1/multimodal/weights.lock.json create mode 100644 tests/cua_s1/test_evaluation.py create mode 100644 tests/cua_s1/test_model.py create mode 100644 tests/cua_s1/test_protocol.py create mode 100644 tests/cua_s1/test_server.py diff --git a/.github/workflows/cua-s1-multimodal.yml b/.github/workflows/cua-s1-multimodal.yml new file mode 100644 index 0000000..9329244 --- /dev/null +++ b/.github/workflows/cua-s1-multimodal.yml @@ -0,0 +1,30 @@ +name: Cua-S1 multimodal CPU checks +on: + pull_request: + paths: + - 'src/models/cua_s1/multimodal/**' + - 'tests/cua_s1/**' + - 'recipe/cua_s1/**' + - '.github/workflows/cua-s1-multimodal.yml' + push: + branches: [main] + paths: + - 'src/models/cua_s1/multimodal/**' + - 'tests/cua_s1/**' + - 'recipe/cua_s1/**' + - '.github/workflows/cua-s1-multimodal.yml' +permissions: + contents: read +jobs: + cpu: + runs-on: ubuntu-latest + timeout-minutes: 10 + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install Pillow==11.3.0 pytest==9.1.1 ruff==0.16.8 + - run: PYTHONPATH=src python -m pytest tests/cua_s1 -q + - run: ruff check --isolated --select E4,E7,E9,F,I src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 + - run: ruff format --isolated --check src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 diff --git a/.gitignore b/.gitignore index ad67955..a2db644 100644 --- a/.gitignore +++ b/.gitignore @@ -19,3 +19,10 @@ target # and can be added to the global gitignore or merged into this file. For a more nuclear # option (not recommended) you can uncomment the following to ignore the entire idea folder. #.idea/ + +# Python workers and local model artifacts +__pycache__/ +.pytest_cache/ +.ruff_cache/ +.venv/ +weights/ diff --git a/README.md b/README.md index 6cc42ab..7fe790f 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ A community-maintained inference engine for prefill-only System1-Omni models, designed around a Rust frontend, model-owned execution, and high-performance CUDA and Metal backends. -The project is in its initial design stage. The architecture below describes the intended implementation; model engines and GPU backends are not implemented yet. +The project is in early development. A Cua-S1 multimodal worker has been validated on CUDA through Transformers and PEFT; the architecture below describes the intended native frontend and backend organization. ## Architecture @@ -27,20 +27,22 @@ Implementation code lives under `src/`; recipes and documentation stay at the re | --- | --- | | [`src/frontend/`](src/frontend/) | Rust serving code and the small engine interface. | | [`src/models/laya/`](src/models/laya/) | LAYA preprocessing, batching, state, execution, and output processing. | +| [`src/models/cua_s1/multimodal/`](src/models/cua_s1/multimodal/) | Cua-S1 screenshot preprocessing, multimodal LoRA execution, and choice probabilities. | | [`src/backends/cuda/`](src/backends/cuda/) | NVIDIA GPU operations and kernel integration. | | [`src/backends/metal/`](src/backends/metal/) | Apple GPU operations and kernel integration. | | [`recipe/`](recipe/) | Model setup instructions, launch commands, configuration examples, and example requests. | | [`docs/`](docs/) | Project documentation and architecture assets. | -These directories currently document ownership; implementations will be added incrementally. They do not prescribe process boundaries. Shared utilities will be extracted when concrete implementations need them. +These directories document ownership; implementations are being added incrementally. They do not prescribe process boundaries. Shared utilities will be extracted when concrete implementations need them. ## Supported models -No models are implemented yet. LAYA is the first planned model: +Validated coverage is listed by modality and execution path: | Model | Status | | --- | --- | | LAYA | Planned | +| [Cua-S1 4B 0.2](recipe/cua_s1/README.md) | Multimodal screenshot choices via Transformers/PEFT on RTX 4090 CUDA; [parity and measurements](recipe/cua_s1/experiments/README.md). Text serving, native CUDA kernels and Metal deferred. | CUDA and Metal coverage will be documented per model as implementations are added and validated. diff --git a/recipe/cua_s1/README.md b/recipe/cua_s1/README.md new file mode 100644 index 0000000..6f175ad --- /dev/null +++ b/recipe/cua_s1/README.md @@ -0,0 +1,145 @@ +# Cua-S1 0.2 multimodal CUDA worker + +This recipe adds screenshot decisions using the multimodal adapter discussed in +[#10](https://github.com/ThinkFlowLab/system1-omni/issues/10). It loads Transformers +and PEFT directly. Upstream `FourBModel` is used only as an independent parity +oracle. The worker owns image decoding, the processor/chat template, vision and +language LoRA loading, and the candidate-letter probability readout. + +The text mapping and pinned revisions follow +[PR #11](https://github.com/ThinkFlowLab/system1-omni/pull/11). The image `state` +format below is this PR's proposed extension. A separately launched multimodal +worker uses the same request model name; deployment routing selects its modality. + +## Setup + +Run from this repository's root on Linux with an NVIDIA GPU. The measured CUDA +wheel, driver, GPU memory and results are recorded in [experiments](experiments/README.md). +Python 3.12 is required by the pinned environment. + +```sh +python3.12 -m venv .venv +. .venv/bin/activate +pip install -r recipe/cua_s1/requirements-multimodal.txt +PYTHONPATH=src python recipe/cua_s1/download_weights.py --dest weights +PYTHONPATH=src python -m models.cua_s1.multimodal.server \ + --base weights/Qwen3.5-4B \ + --adapter weights/cua-s1-4b-0.2/multimodal +``` + +`weights.lock.json` pins and checks every loaded artifact's size and SHA-256. +Extra files are rejected, except Hugging Face's `.cache` metadata, so another +checkpoint cannot silently override verified shards. Downloads require roughly +9 GB plus cache/install space. Loading is offline after the download completes. +Weights are not included in this repository. + +The worker binds to `127.0.0.1:8000` only after loading and a successful warmup. +`GET /health` returns `{"status":"ready","modality":"multimodal"}`. One request +runs at a time; concurrent requests return `503`. This is a loopback model worker, +with the Rust frontend and an ingress responsible for public serving. + +To use the Rust frontend when [PR #2](https://github.com/ThinkFlowLab/system1-omni/pull/2) +is available in your checkout: + +```sh +cargo build --release --locked +OMNI_JEV_BIND=127.0.0.1:8080 \ +OMNI_JEV_BACKEND_URL=http://127.0.0.1:8000 ./target/release/omni-jev +``` + +## Request and response + +Generate a self-contained example with a synthetic settings screenshot: + +```sh +python recipe/cua_s1/make_example.py --output /tmp/cua-example.json +curl -sS http://127.0.0.1:8000/v1/systemone \ + -H 'Content-Type: application/json' --data-binary @/tmp/cua-example.json +``` + +Replace the port with `8080` to send the same request through the frontend. +The request shape is: + +```json +{ + "model": "cua-s1-4b-0.2", + "state": {"image": "data:image/png;base64,"}, + "questions": { + "next": { + "type": "choice", + "instructions": "Save the changes", + "criteria": {"save": "Save changes", "cancel": "Cancel"} + } + } +} +``` + +- `state` contains exactly one inline PNG or JPEG data URL. Images are decoded to + RGB. Local filenames, remote URLs, video, animation and mixed text/image state + are unsupported. +- There are 1–8 questions and 1–26 options per question. Option order assigns + letters A–Z. Labels are strings, objects, arrays or `null` (which uses the key). + Structured values use Python `json.dumps(..., ensure_ascii=False)` followed by + the upstream chooser's label escaping. Instructions accept strings, objects + or arrays; an omitted/empty instruction omits the goal block. +- Limits: 8 MiB body, 4 MiB decoded image, 2048 pixels per side, 1,048,576 pixels + total, 16,384 characters per question and 4096 processed tokens per question. + Every question is validated/preprocessed before any forward pass begins. +- `<|image_pad|>`, `<|video_pad|>`, `<|vision_start|>` and `<|vision_end|>` are + rejected in user text because the processor interprets them as media controls. + Other special-token spellings retain upstream tokenization behavior. +- Empty/invalid inputs, duplicate JSON keys, unsupported models and `score`/`noul` + questions return `422`. Oversized bodies return `413`; chunked uploads return + `411`. Send `Content-Length` and `Content-Type: application/json`. + +Each answer has `type`, `choice`, `probabilities` and `confidence`. The readout +uses the last position's candidate-letter logits, casts to fp32 and applies +softmax over those letters only. There is no decode. Ties select the earliest +option; confidence is `1 - H(p)/ln(n)`, or 1 for one option. Each question has a +separate forward pass over the same screenshot. Usage sums processed input tokens +and reports zero output tokens. + +Response identity: +`cua-ai/cua-s1-4b-0.2@16818868b0cc7813808aae4e87b417657046ab79:multimodal`. +The base is BF16; PEFT's rank-16, alpha-32 adapter remains unmerged with fp32 LoRA +branches, including 50 vision projection modules (178 total adapted modules). + +## Reproduce correctness and profiling + +```sh +git clone https://github.com/trycua/cua.git /tmp/cua-reference +git -C /tmp/cua-reference checkout 0e75660ce4c2edda519e0c795fa3ad98abf4e76f +for mode in reference candidate benchmark; do + PYTHONPATH=src python recipe/cua_s1/evaluate_multimodal.py \ + --weights weights --reference /tmp/cua-reference \ + --output /tmp/cua-evidence --mode "$mode" +done +``` + +The evaluator hashes the pinned `four_b.py` before importing it and checks report +provenance/environment before comparison. Reference and candidate are separate +processes to avoid keeping two models in GPU memory. Eight synthetic requests +(nine question forwards) cover two resolutions, PNG/JPEG, 1/26 candidates, +structured/Unicode labels, special-token text and multiple questions. It requires +identical processor tensor shapes/dtypes/hashes and identical fp32 candidate +probabilities; numerical tolerance is zero. These are integration/parity fixtures, +not an evaluation of GUI task success. + +Benchmark mode records two runs of 50 serial requests after five warmups on the +640×480 fixture, with synchronized end-to-end engine latency, p50/p95, serial +throughput, allocated/reserved GPU peaks and a separate operator profile. Load +and warmup are recorded separately; candidate load time includes artifact hash +verification. See the experiment report for measured scope and limitations. + +CPU-only validation: + +```sh +pip install Pillow==11.3.0 pytest==9.1.1 ruff==0.16.8 +PYTHONPATH=src python -m pytest tests/cua_s1 -q +ruff check --select E4,E7,E9,F,I src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 +ruff format --check src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 +``` + +Metal, native CUDA kernels, text-adapter serving, batching, caching and training +are outside this worker's scope. This implementation does not import or modify +another contributor's text engine. diff --git a/recipe/cua_s1/check_frontend.py b/recipe/cua_s1/check_frontend.py new file mode 100644 index 0000000..b2ae0fe --- /dev/null +++ b/recipe/cua_s1/check_frontend.py @@ -0,0 +1,62 @@ +"""Compare HTTP response status, content type and bytes through Rust and directly.""" + +import argparse +import hashlib +import json +import urllib.error +import urllib.request +from pathlib import Path + + +def exchange(base, route, body=None): + request = urllib.request.Request( + base.rstrip("/") + route, + data=body, + headers={"Content-Type": "application/json"}, + ) + try: + response = urllib.request.urlopen(request, timeout=60) + except urllib.error.HTTPError as exc: + response = exc + with response: + return response.status, response.headers.get("Content-Type"), response.read() + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--worker", default="http://127.0.0.1:8000") + parser.add_argument("--frontend", default="http://127.0.0.1:8080") + parser.add_argument("--fixtures", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + checks = [("health", "/health", None)] + checks += [ + (p.stem, "/v1/systemone", p.read_bytes()) + for p in sorted(args.fixtures.glob("*.json")) + ] + checks += [ + ("invalid", "/v1/systemone", b"{}"), + ("duplicate", "/v1/systemone", b'{"model":1,"model":2}'), + ] + report = [] + for name, route, body in checks: + direct = exchange(args.worker, route, body) + proxied = exchange(args.frontend, route, body) + assert direct == proxied, f"frontend changed response: {name}" + expected_status = 422 if name in {"invalid", "duplicate"} else 200 + assert direct[0] == expected_status, f"unexpected status: {name}: {direct[0]}" + report.append( + { + "name": name, + "status": direct[0], + "content_type": direct[1], + "body_sha256": hashlib.sha256(direct[2]).hexdigest(), + "identical": True, + } + ) + print(f"{name}: HTTP {direct[0]}, identical response", flush=True) + args.output.write_text(json.dumps(report, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/recipe/cua_s1/download_weights.py b/recipe/cua_s1/download_weights.py new file mode 100644 index 0000000..104b393 --- /dev/null +++ b/recipe/cua_s1/download_weights.py @@ -0,0 +1,33 @@ +"""Download the upstream-pinned artifacts and verify their checksums.""" + +import argparse +import json +from pathlib import Path + +from huggingface_hub import snapshot_download + +from models.cua_s1.multimodal.model import verify_weights + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--dest", type=Path, required=True) + args = parser.parse_args() + lock = ( + Path(__file__).resolve().parents[2] + / "src/models/cua_s1/multimodal/weights.lock.json" + ) + for artifact in json.loads(lock.read_text())["artifacts"]: + snapshot_download( + repo_id=artifact["repo_id"], + revision=artifact["revision"], + local_dir=args.dest / artifact["name"], + allow_patterns=list(artifact["files"]), + token=False, + ) + verify_weights(args.dest / "Qwen3.5-4B", args.dest / "cua-s1-4b-0.2/multimodal") + print("Pinned base and multimodal adapter checksums verified.") + + +if __name__ == "__main__": + main() diff --git a/recipe/cua_s1/evaluate_multimodal.py b/recipe/cua_s1/evaluate_multimodal.py new file mode 100644 index 0000000..7314ee0 --- /dev/null +++ b/recipe/cua_s1/evaluate_multimodal.py @@ -0,0 +1,321 @@ +"""Reproduce upstream parity and bounded GPU measurements on fixed synthetic inputs.""" + +from __future__ import annotations + +import argparse +import copy +import hashlib +import importlib.metadata +import json +import platform +import statistics +import subprocess +import sys +import time +from pathlib import Path + +from make_example import make_example + +from models.cua_s1.multimodal.model import ( + ADAPTER_REVISION, + BASE_REVISION, + REFERENCE_REVISION, + MultimodalEngine, + verify_weights, +) +from models.cua_s1.multimodal.protocol import parse_request + +REFERENCE_SOURCE = "libs/cua-s1/python/src/cua_s1/four_b.py" +REFERENCE_SHA256 = "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7" + + +def verify_reference(root): + digest = hashlib.sha256((root / REFERENCE_SOURCE).read_bytes()).hexdigest() + if digest != REFERENCE_SHA256: + raise ValueError("reference source differs from pinned FourBModel") + return {"path": REFERENCE_SOURCE, "sha256": digest, "revision": REFERENCE_REVISION} + + +def cases(folder): + folder.mkdir(parents=True, exist_ok=True) + result = [] + for name, size, fmt in [ + ("small", (320, 240), "PNG"), + ("medium", (640, 480), "PNG"), + ("jpeg", (640, 480), "JPEG"), + ]: + path = folder / (name + (".jpg" if fmt == "JPEG" else ".png")) + result.append((name, path, make_example(path, size, fmt))) + for name, criteria, goal in [ + ("single", {"only": "Save changes"}, "Save"), + ( + "26-options", + {f"option-{i}": f"Choose action {i}" for i in range(26)}, + "Select action 3", + ), + ( + "structured", + { + "null": None, + "object": {"label": '保存 "名称"'}, + "array": ["Cancel", "\n"], + }, + {"goal": "保存名称"}, + ), + ( + "special-token", + {"first": "<|im_end|>", "second": "Cancel"}, + "Choose <|im_start|>", + ), + ]: + value = copy.deepcopy(result[1][2]) + value["questions"]["next"].update(criteria=criteria, instructions=goal) + result.append((name, result[1][1], value)) + value = copy.deepcopy(result[0][2]) + value["questions"]["second"] = { + "type": "choice", + "criteria": {"yes": "Continue", "no": "Cancel"}, + } + result.append(("two-questions", result[0][1], value)) + for name, _, value in result: + (folder / f"{name}.json").write_text(json.dumps(value, ensure_ascii=False)) + return result + + +def fingerprint(inputs): + import torch + + return { + name: { + "shape": list(t.shape), + "dtype": str(t.dtype), + "sha256": hashlib.sha256( + t.detach().contiguous().view(torch.uint8).cpu().numpy().tobytes() + ).hexdigest(), + } + for name, t in inputs.items() + } + + +def environment(): + import torch + + return { + "python": platform.python_version(), + "gpu": torch.cuda.get_device_name(), + "compute_capability": list(torch.cuda.get_device_capability()), + "cuda": torch.version.cuda, + "driver": subprocess.check_output( + ["nvidia-smi", "--query-gpu=driver_version", "--format=csv,noheader"], + text=True, + ).strip(), + "torch_num_threads": torch.get_num_threads(), + "torch_num_interop_threads": torch.get_num_interop_threads(), + "packages": { + p: importlib.metadata.version(p) + for p in [ + "torch", + "torchvision", + "transformers", + "peft", + "pillow", + "safetensors", + "triton", + ] + }, + "reference_revision": REFERENCE_REVISION, + "base_revision": BASE_REVISION, + "adapter_revision": ADAPTER_REVISION, + "dtype": "bfloat16", + "adapter_merged": False, + } + + +def measure(call): + import torch + + torch.cuda.synchronize() + start = time.perf_counter() + result = call() + torch.cuda.synchronize() + return result, (time.perf_counter() - start) * 1000 + + +def quantiles(values): + ordered = sorted(values) + return { + "p50_ms": statistics.median(values), + "p95_ms": ordered[max(0, __import__("math").ceil(0.95 * len(values)) - 1)], + } + + +def main(): + import torch + + p = argparse.ArgumentParser(description=__doc__) + p.add_argument("--weights", type=Path, required=True) + p.add_argument( + "--reference", type=Path, required=True, help="checkout of pinned trycua/cua" + ) + p.add_argument("--output", type=Path, required=True) + p.add_argument( + "--mode", choices=["reference", "candidate", "benchmark"], required=True + ) + args = p.parse_args() + args.output.mkdir(parents=True, exist_ok=True) + base = str(args.weights / "Qwen3.5-4B") + adapter = str(args.weights / "cua-s1-4b-0.2/multimodal") + fixture_set = cases(args.output / "fixtures") + report = {"environment": environment(), "mode": args.mode, "cases": []} + report["reference_source"] = verify_reference(args.reference) + if args.mode == "reference": + _, report["artifact_verification_ms"] = measure( + lambda: verify_weights(Path(base), Path(adapter)) + ) + sys.path.insert(0, str(args.reference / "libs/cua-s1/python/src")) + from cua_s1.four_b import FourBModel, Option, assign_letters, build_prompt + + model = FourBModel( + base_model=base, lora_adapter_path=adapter, modality="multimodal" + ) + _, report["load_ms"] = measure(model.load) + first = True + for name, path, value in fixture_set: + request = parse_request(value) + for q in request.questions: + options = [ + Option(element_id=k, role="Decision", label=v, action="select") + for k, v in zip(q.keys, q.labels) + ] + messages = build_prompt( + assign_letters(options), + app="Cua Driver", + task_family="closed-candidate decision", + screenshot=path, + modality="multimodal", + goal=q.goal, + ) + text = model._processor.apply_chat_template( + messages, tokenize=False, add_generation_prompt=True + ) + inputs = model._processor( + text=[text], images=[request.image], return_tensors="pt" + ) + + def call(): + return model.forward( + options, + app="Cua Driver", + task_family="closed-candidate decision", + screenshot=path, + goal=q.goal, + ) + + if first: + _, report["warmup_ms"] = measure(call) + first = False + output, elapsed = measure(call) + report["cases"].append( + { + "name": name, + "question": q.name, + "inputs": fingerprint(inputs), + "probabilities": [x.probability for x in output], + "latency_ms": elapsed, + } + ) + print(f"reference {name}/{q.name}: {elapsed:.1f} ms", flush=True) + else: + engine, report["load_ms"] = measure(lambda: MultimodalEngine(base, adapter)) + _, report["warmup_ms"] = measure(engine.warmup) + report["adapter_modules"] = engine.adapter_modules + if args.mode == "candidate": + reference = json.loads((args.output / "reference.json").read_text()) + if ( + reference["reference_source"] != report["reference_source"] + or reference["environment"] != report["environment"] + ): + raise ValueError("reference provenance or environment mismatch") + expected = {(x["name"], x["question"]): x for x in reference["cases"]} + for name, _, value in fixture_set: + request = parse_request(value) + for q in request.questions: + inputs = engine.prepare(request.image, q) + probabilities, elapsed = measure(lambda: engine.score(inputs, q)) + ref = expected[name, q.name] + difference = max( + abs(a - b) for a, b in zip(ref["probabilities"], probabilities) + ) + assert fingerprint(inputs) == ref["inputs"], ( + f"preprocessing mismatch: {name}" + ) + assert probabilities == ref["probabilities"], ( + f"probability mismatch: {name}: {difference}" + ) + report["cases"].append( + { + "name": name, + "question": q.name, + "inputs": fingerprint(inputs), + "probabilities": probabilities, + "max_abs_difference": difference, + "latency_ms": elapsed, + } + ) + print(f"candidate {name}/{q.name}: exact parity", flush=True) + else: + request = parse_request(fixture_set[1][2]) + for _ in range(5): + engine.predict(request) + report["benchmark"] = { + "case": "medium", + "batch_size": 1, + "concurrency": 1, + "warmup_requests": 5, + "runs": [], + } + for run in range(2): + torch.cuda.reset_peak_memory_stats() + latencies = [ + measure(lambda: engine.predict(request))[1] for _ in range(50) + ] + report["benchmark"]["runs"].append( + { + "run": run + 1, + "latencies_ms": latencies, + **quantiles(latencies), + "requests_per_second_serial": 1000 / statistics.mean(latencies), + "peak_allocated_bytes": torch.cuda.max_memory_allocated(), + "peak_reserved_bytes": torch.cuda.max_memory_reserved(), + } + ) + print(f"benchmark run {run + 1}: {quantiles(latencies)}", flush=True) + with torch.profiler.profile( + activities=[ + torch.profiler.ProfilerActivity.CPU, + torch.profiler.ProfilerActivity.CUDA, + ], + record_shapes=True, + ) as prof: + engine.predict(request) + torch.cuda.synchronize() + report["profile"] = [ + { + "op": x.key, + "count": x.count, + "device_time_us": x.device_time_total, + "self_device_time_us": x.self_device_time_total, + "cpu_time_us": x.cpu_time_total, + "self_cpu_time_us": x.self_cpu_time_total, + } + for x in sorted( + prof.key_averages(), key=lambda x: x.device_time_total, reverse=True + )[:30] + ] + report["peak_allocated_bytes"] = torch.cuda.max_memory_allocated() + (args.output / f"{args.mode}.json").write_text(json.dumps(report, indent=2)) + print(f"Saved {args.mode}.json", flush=True) + + +if __name__ == "__main__": + main() diff --git a/recipe/cua_s1/experiments/README.md b/recipe/cua_s1/experiments/README.md new file mode 100644 index 0000000..d73403b --- /dev/null +++ b/recipe/cua_s1/experiments/README.md @@ -0,0 +1,111 @@ +# RTX 4090 multimodal validation — 2026-09-27 + +This is a Transformers/PEFT CUDA baseline, with the adapter unmerged and full +logits, for the screenshot worker in [the recipe](../README.md). It does not +implement a native CUDA backend or claim a speedup over upstream. + +## Environment and artifacts + +- One NVIDIA GeForce RTX 4090, 24,564 MiB, compute capability 8.9. +- Ubuntu 22.04 container; NVIDIA driver 595.71.05; PyTorch CUDA runtime 13.0. +- Python 3.12.13; torch 2.14.0, torchvision 0.29.0, Transformers 5.17.0, + PEFT 0.21.0, Pillow 11.3.0, safetensors 0.8.0, Triton 3.8.0. +- PyTorch reported 64 intra-op and 64 inter-op threads; no thread tuning applied. +- BF16 base, PEFT fp32 LoRA branches, no adapter merge, no quantization. +- Neither `flash-linear-attention` nor `causal-conv1d` is installed. The matching + upstream reference uses Transformers' PyTorch fallback implementations. +- All pinned weights passed size and SHA-256 verification. The lock contains + 9,342,907,469 base bytes and 186,638,393 adapter-repository bytes. Only the + `multimodal` adapter is loaded; all 178 target modules, including 50 visual + modules, are required at startup. + +Pinned revisions and the reference-source SHA-256 are present in every JSON +report. Full installed versions are in [packages.txt](rtx4090/packages.txt). +The numerical contract and dependency versions were fixed before comparisons. +No model weights or private screenshots are included. + +## Correctness + +| Check | Result | Evidence | +| --- | --- | --- | +| Processor tensor shapes, dtypes and SHA-256 values | Exact match on 9 question forwards across 8 requests | [reference.json](rtx4090/reference.json), [candidate.json](rtx4090/candidate.json) | +| Candidate fp32 probabilities | Exact match; maximum absolute difference **0** | Same reports | +| Multimodal LoRA attachment | 178 modules, visual modules present | [candidate.json](rtx4090/candidate.json) | +| Rust frontend vs direct worker | All 11 HTTP comparisons passed; status, content type and body bytes identical | [frontend.json](rtx4090/frontend.json) | +| CPU validation | 39 tests passed on local macOS and the Linux GPU host | `tests/cua_s1`, CPU CI workflow | + +Fixtures are generated by `evaluate_multimodal.py` using the pinned Pillow version. +They cover 320×240 and 640×480 screenshots, PNG/JPEG, 1 and 26 candidates, +structured/Unicode labels, special-token spellings and two questions sharing one +image. Processed prompts span 215–752 tokens. Hashes cover input IDs, attention +masks, image pixels and image-grid metadata, rather than just the selected option. +The fixtures establish integration parity, not GUI task accuracy or generalization. + +For the frontend test, PR #2's Rust frontend at +`0c91671ac7c8bd698b957b2c0df921de96e7e628` was built with `cargo build --release --locked` +on macOS. It forwarded to the Linux GPU worker over an SSH tunnel. This exercises +real inference and byte preservation; **network/HTTP latency is not benchmarked**. +Eight valid fixtures, health, malformed envelope and duplicate JSON keys were checked. + +Reproduce the HTTP check after generating the fixtures and starting both servers: + +```sh +python recipe/cua_s1/check_frontend.py \ + --worker http://127.0.0.1:8000 --frontend http://127.0.0.1:8080 \ + --fixtures /tmp/cua-evidence/fixtures --output /tmp/cua-evidence/frontend.json +``` + +## Warm engine performance + +[benchmark.json](rtx4090/benchmark.json) contains every sample and the operator +profile. One 640×480 PNG, three candidates, 456 processed tokens, batch size 1, +concurrency 1; five warmup requests followed by two runs of 50 requests. +`torch.cuda.synchronize()` brackets each measurement. p95 uses nearest rank. + +Timing covers `engine.predict`: chat-template application, processor work, host-to- +device transfer, forward pass, letter readout and answer construction. The request +has already been parsed and its image decoded. JSON parsing, image decoding, +HTTP, queueing, model load, artifact hashing and warmup are excluded. + +| Run | p50 | p95 | Serial throughput | Peak allocated | Peak reserved | +| --- | --- | --- | --- | --- | --- | +| 1 | 126.39 ms | 136.68 ms | 7.79 requests/s | 8.84 GiB | 9.07 GiB | +| 2 | 128.28 ms | 133.01 ms | 7.77 requests/s | 8.84 GiB | 9.07 GiB | + +The parity run's peak allocated memory across all nine forwards was 8.99 GiB. +These bounds describe the measured fixtures, not every request admitted by the +worker's 4096-token limit or a concurrent/batched deployment. + +Measured candidate construction, including weight hashing, was 19.94 s in the +parity process and 18.94 s in the benchmark process. Their initial 224×224 warmups +were 1.82 s and 1.63 s. Upstream load was 16.82 s plus 9.54 s of separately measured +artifact verification; its first 320×240 warmup was 3.67 s. These are individual +observations after importing PyTorch and probing the GPU, with different warmup +shapes and filesystem-cache histories; they are not a cold-start speed comparison. +Per-fixture timings in the parity reports are also single observations; the +reference includes image opening while candidate timing covers `score` only. + +## Profiling and the next optimization boundary + +A separate profiled inference, excluded from the latency samples, reported: + +| Operator | Calls | Self CUDA time | +| --- | --- | --- | +| `aten::mm` | 605 | 35.42 ms | +| `aten::copy_` | 1,642 | 9.55 ms | +| `aten::bmm` | 817 | 9.11 ms | +| `aten::addmm` | 98 | 5.89 ms | +| `aten::mul` | 1,103 | 5.76 ms | + +These are instrumented operator totals, not percentages of request wall time. +The raw profile includes both framework operators and CUDA kernels, which overlap; +do not add parent and child events or add kernel time to its enclosing operator. + +Matrix multiplication dominates the observed operator totals. A subsequent CUDA +optimization should first attribute those GEMMs and copies to vision, language +and LoRA branches using shapes/module ranges, then compare one bounded change +against this baseline. This profile alone does not justify replacing a specific +kernel or claiming a speedup. Gated DeltaNet/causal-convolution backend work should +be coordinated with the text-engine contributor because those language layers are +shared. This PR keeps the exact upstream execution path and contributes the +multimodal correctness and performance baseline needed for that work. diff --git a/recipe/cua_s1/experiments/rtx4090/benchmark.json b/recipe/cua_s1/experiments/rtx4090/benchmark.json new file mode 100644 index 0000000..7b69afc --- /dev/null +++ b/recipe/cua_s1/experiments/rtx4090/benchmark.json @@ -0,0 +1,409 @@ +{ + "environment": { + "python": "3.12.13", + "gpu": "NVIDIA GeForce RTX 4090", + "compute_capability": [ + 8, + 9 + ], + "cuda": "13.0", + "driver": "595.71.05", + "torch_num_threads": 64, + "torch_num_interop_threads": 64, + "packages": { + "torch": "2.14.0", + "torchvision": "0.29.0", + "transformers": "5.17.0", + "peft": "0.21.0", + "pillow": "11.3.0", + "safetensors": "0.8.0", + "triton": "3.8.0" + }, + "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", + "dtype": "bfloat16", + "adapter_merged": false + }, + "mode": "benchmark", + "cases": [], + "reference_source": { + "path": "libs/cua-s1/python/src/cua_s1/four_b.py", + "sha256": "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7", + "revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" + }, + "load_ms": 18941.51310157031, + "warmup_ms": 1633.873094804585, + "adapter_modules": 178, + "benchmark": { + "case": "medium", + "batch_size": 1, + "concurrency": 1, + "warmup_requests": 5, + "runs": [ + { + "run": 1, + "latencies_ms": [ + 125.8343830704689, + 127.36847903579473, + 125.82478113472462, + 125.89417397975922, + 146.41341753304005, + 127.77413055300713, + 127.22665630280972, + 125.83817914128304, + 126.10162515193224, + 127.15142965316772, + 126.42081826925278, + 126.38038955628872, + 126.02570466697216, + 128.5353172570467, + 129.56660240888596, + 129.32745181024075, + 126.28413271158934, + 126.18042714893818, + 133.06267280131578, + 126.3929232954979, + 131.68340362608433, + 126.76465138792992, + 125.68666227161884, + 126.71545054763556, + 125.96531771123409, + 125.74396748095751, + 130.0713624805212, + 132.96419847756624, + 131.16213120520115, + 126.18762347847223, + 125.84684416651726, + 126.00735854357481, + 125.77000726014376, + 126.28292944282293, + 125.9002760052681, + 130.30958734452724, + 136.85170747339725, + 134.48695000261068, + 136.67899277061224, + 134.1363089159131, + 133.8470932096243, + 126.33792497217655, + 126.30761601030827, + 126.43217574805021, + 125.64307544380426, + 125.50272978842258, + 125.40635839104652, + 125.05617458373308, + 125.54042227566242, + 127.74347327649593 + ], + "p50_ms": 126.3866564258933, + "p95_ms": 136.67899277061224, + "requests_per_second_serial": 7.792244462582349, + "peak_allocated_bytes": 9496202240, + "peak_reserved_bytes": 9741271040 + }, + { + "run": 2, + "latencies_ms": [ + 126.58042088150978, + 126.69678498059511, + 126.67341157793999, + 129.0692137554288, + 127.07117944955826, + 129.27887961268425, + 133.4229400381446, + 128.44505812972784, + 127.24453490227461, + 126.93716865032911, + 130.17353229224682, + 131.4986888319254, + 130.03864511847496, + 131.02801516652107, + 129.7577489167452, + 128.6515649408102, + 132.58606754243374, + 128.7766983732581, + 131.9235684350133, + 128.00912745296955, + 127.52761784940958, + 129.2492775246501, + 132.3700835928321, + 128.50904930382967, + 127.91914585977793, + 131.54291920363903, + 129.6907477080822, + 133.75771697610617, + 129.1659427806735, + 126.58135313540697, + 126.16921309381723, + 126.81071180850267, + 128.0105598270893, + 128.40583361685276, + 127.4806559085846, + 127.57191434502602, + 128.10698058456182, + 127.0896615460515, + 127.29146610945463, + 133.00949800759554, + 128.90154495835304, + 128.37096769362688, + 127.91174557060003, + 127.8772447258234, + 128.1898096203804, + 127.69030872732401, + 126.3129971921444, + 129.61805891245604, + 126.81897450238466, + 126.8136901780963 + ], + "p50_ms": 128.28038865700364, + "p95_ms": 133.00949800759554, + "requests_per_second_serial": 7.765628438386997, + "peak_allocated_bytes": 9496202240, + "peak_reserved_bytes": 9741271040 + } + ] + }, + "profile": [ + { + "op": "aten::matmul", + "count": 1422, + "device_time_us": 44527.040000000154, + "self_device_time_us": 0.0, + "cpu_time_us": 53924.26100000019, + "self_cpu_time_us": 9931.453000001111 + }, + { + "op": "aten::linear", + "count": 703, + "device_time_us": 41310.13700000033, + "self_device_time_us": 0.0, + "cpu_time_us": 24486.469999999994, + "self_cpu_time_us": 1650.2199999998538 + }, + { + "op": "aten::mm", + "count": 605, + "device_time_us": 35415.818000000334, + "self_device_time_us": 35415.818000000334, + "cpu_time_us": 13464.22999999985, + "self_cpu_time_us": 8533.79799999894 + }, + { + "op": "ampere_bf16_s1688gemm_bf16_64x128_sliced1x2_ldg8_f2f_tn", + "count": 96, + "device_time_us": 17010.658999999934, + "self_device_time_us": 17010.658999999934, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::copy_", + "count": 1642, + "device_time_us": 9549.038000000242, + "self_device_time_us": 9549.038000000242, + "cpu_time_us": 25915.56300000043, + "self_cpu_time_us": 9625.815000000488 + }, + { + "op": "aten::bmm", + "count": 817, + "device_time_us": 9111.22199999982, + "self_device_time_us": 9109.750999999822, + "cpu_time_us": 20146.271999999823, + "self_cpu_time_us": 16179.460999999617 + }, + { + "op": "ampere_bf16_s1688gemm_bf16_128x128_ldg8_f2f_stages_32x1_tn", + "count": 56, + "device_time_us": 6589.621000000137, + "self_device_time_us": 6589.621000000137, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::addmm", + "count": 98, + "device_time_us": 5894.318999999992, + "self_device_time_us": 5894.318999999992, + "cpu_time_us": 2576.6010000000297, + "self_cpu_time_us": 1777.2650000000776 + }, + { + "op": "aten::mul", + "count": 1103, + "device_time_us": 5763.554999999398, + "self_device_time_us": 5763.554999999398, + "cpu_time_us": 13280.975000000395, + "self_cpu_time_us": 8342.606000000327 + }, + { + "op": "aten::to", + "count": 1361, + "device_time_us": 5300.117000000242, + "self_device_time_us": 0.0, + "cpu_time_us": 28756.462000000047, + "self_cpu_time_us": 1621.334000000551 + }, + { + "op": "aten::_to_copy", + "count": 1080, + "device_time_us": 5300.117000000242, + "self_device_time_us": 0.0, + "cpu_time_us": 27135.127999999495, + "self_cpu_time_us": 4165.844999999095 + }, + { + "op": "ampere_bf16_s16816gemm_bf16_128x64_ldg8_f2f_tn", + "count": 1, + "device_time_us": 4485.06700000001, + "self_device_time_us": 4485.06700000001, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::add", + "count": 1009, + "device_time_us": 4335.962000000129, + "self_device_time_us": 4335.962000000129, + "cpu_time_us": 11608.1780000002, + "self_cpu_time_us": 7166.728000000563 + }, + { + "op": "ampere_bf16_s1688gemm_bf16_128x64_sliced1x2_ldg8_relu_f2f_tn", + "count": 50, + "device_time_us": 3960.761000000013, + "self_device_time_us": 3960.761000000013, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::linalg_solve_triangular", + "count": 48, + "device_time_us": 3340.373000000007, + "self_device_time_us": 2853.693999999952, + "cpu_time_us": 4383.135000000191, + "self_cpu_time_us": 1135.9290000003602 + }, + { + "op": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast > >(at::TensorIteratorBase&, at::native::BinaryFunctor > const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast > >(at::TensorIteratorBase&, at::native::BinaryFunctor > const&)::{lambda(int)#1})", + "count": 668, + "device_time_us": 3093.801999999436, + "self_device_time_us": 3093.801999999436, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::convolution", + "count": 25, + "device_time_us": 3044.873999999925, + "self_device_time_us": 0.0, + "cpu_time_us": 2021.7800000000425, + "self_cpu_time_us": 143.51700000008714 + }, + { + "op": "aten::_convolution", + "count": 25, + "device_time_us": 3044.873999999925, + "self_device_time_us": 0.0, + "cpu_time_us": 1878.2629999999554, + "self_cpu_time_us": 489.0389999998333 + }, + { + "op": "void batch_trsm_left_kernel(cublasTrsmBatchParams2, float, float const*, int)", + "count": 48, + "device_time_us": 2853.693999999952, + "self_device_time_us": 2853.693999999952, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "ampere_bf16_s1688gemm_bf16_64x64_sliced1x4_ldg8_f2f_tn", + "count": 32, + "device_time_us": 2847.620000000141, + "self_device_time_us": 2847.620000000141, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::clone", + "count": 141, + "device_time_us": 2841.203000000127, + "self_device_time_us": 0.0, + "cpu_time_us": 4514.54400000007, + "self_cpu_time_us": 654.2669999996788 + }, + { + "op": "ampere_sgemm_128x128_tn", + "count": 240, + "device_time_us": 2783.6780000002327, + "self_device_time_us": 2783.6780000002327, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "void at::native::unrolled_elementwise_kernel, std::array, 4, TrivialOffsetCalculator<2, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<2>, at::native::memory::StoreWithCast<1> >(int, at::native::CUDAFunctor_add, std::array, TrivialOffsetCalculator<2, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<2>, at::native::memory::StoreWithCast<1>)", + "count": 178, + "device_time_us": 2525.99999999996, + "self_device_time_us": 2525.99999999996, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "ampere_sgemm_128x128_nn", + "count": 192, + "device_time_us": 2428.904999999824, + "self_device_time_us": 2428.904999999824, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "ampere_sgemm_128x128_nt", + "count": 192, + "device_time_us": 2347.072999999873, + "self_device_time_us": 2347.072999999873, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase&, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#12}::operator()() const::{lambda(c10::BFloat16)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase&, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#12}::operator()() const::{lambda(c10::BFloat16)#1} const&)::{lambda(int)#1})", + "count": 112, + "device_time_us": 2314.617000000122, + "self_device_time_us": 2314.617000000122, + "cpu_time_us": 0, + "self_cpu_time_us": 0 + }, + { + "op": "aten::conv1d", + "count": 24, + "device_time_us": 2197.6299999999246, + "self_device_time_us": 0.0, + "cpu_time_us": 1939.474999999984, + "self_cpu_time_us": 93.11899999994057 + }, + { + "op": "aten::scaled_dot_product_attention", + "count": 32, + "device_time_us": 2140.842999999986, + "self_device_time_us": 0.0, + "cpu_time_us": 2570.106999999978, + "self_cpu_time_us": 379.762999999959 + }, + { + "op": "aten::_scaled_dot_product_flash_attention", + "count": 32, + "device_time_us": 2140.842999999986, + "self_device_time_us": 0.0, + "cpu_time_us": 2190.344000000019, + "self_cpu_time_us": 310.61200000015015 + }, + { + "op": "aten::_flash_attention_forward", + "count": 32, + "device_time_us": 2140.842999999986, + "self_device_time_us": 2140.842999999986, + "cpu_time_us": 1620.5300000000316, + "self_cpu_time_us": 539.4930000000277 + } + ], + "peak_allocated_bytes": 9496202240 +} \ No newline at end of file diff --git a/recipe/cua_s1/experiments/rtx4090/candidate.json b/recipe/cua_s1/experiments/rtx4090/candidate.json new file mode 100644 index 0000000..f8e4a92 --- /dev/null +++ b/recipe/cua_s1/experiments/rtx4090/candidate.json @@ -0,0 +1,536 @@ +{ + "environment": { + "python": "3.12.13", + "gpu": "NVIDIA GeForce RTX 4090", + "compute_capability": [ + 8, + 9 + ], + "cuda": "13.0", + "driver": "595.71.05", + "torch_num_threads": 64, + "torch_num_interop_threads": 64, + "packages": { + "torch": "2.14.0", + "torchvision": "0.29.0", + "transformers": "5.17.0", + "peft": "0.21.0", + "pillow": "11.3.0", + "safetensors": "0.8.0", + "triton": "3.8.0" + }, + "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", + "dtype": "bfloat16", + "adapter_merged": false + }, + "mode": "candidate", + "cases": [ + { + "name": "small", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" + }, + "attention_mask": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" + }, + "pixel_values": { + "shape": [ + 320, + 1536 + ], + "dtype": "torch.float32", + "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" + } + }, + "probabilities": [ + 0.9958575367927551, + 0.0005507932510226965, + 0.003591623157262802 + ], + "max_abs_difference": 0.0, + "latency_ms": 127.26952508091927 + }, + { + "name": "medium", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" + }, + "attention_mask": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.9958575367927551, + 0.0005507932510226965, + 0.003591623157262802 + ], + "max_abs_difference": 0.0, + "latency_ms": 126.0456619784236 + }, + { + "name": "jpeg", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" + }, + "attention_mask": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "5a78a7f25b40888ad5c134d02826d691c27707384603242b99897cc623817bd8" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.9936226010322571, + 0.0011634123511612415, + 0.005214052740484476 + ], + "max_abs_difference": 0.0, + "latency_ms": 121.78163509815931 + }, + { + "name": "single", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 431 + ], + "dtype": "torch.int64", + "sha256": "c4d7b1102e94984504efa938052a0bba3639ba85f9917de3e8af5426ceefd16b" + }, + "attention_mask": { + "shape": [ + 1, + 431 + ], + "dtype": "torch.int64", + "sha256": "3419eb840bd7f9d9e4b1fa6aaea783bae7a4752068701794687337702c2d8968" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 431 + ], + "dtype": "torch.int64", + "sha256": "077d91126e88fd3e8ba97a2d8079c28a4d5da388e56c89ca0e584fd5ad4b3641" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 1.0 + ], + "max_abs_difference": 0.0, + "latency_ms": 117.80334264039993 + }, + { + "name": "26-options", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 752 + ], + "dtype": "torch.int64", + "sha256": "726f7d304140077e32490233aba734494431d93b7e2885bcb72d8b1c730e022b" + }, + "attention_mask": { + "shape": [ + 1, + 752 + ], + "dtype": "torch.int64", + "sha256": "389336aff30134c9af5da75eb0a9694eef903cde261cd8741f8aafae5ff94dc7" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 752 + ], + "dtype": "torch.int64", + "sha256": "e4cb23e2c28d9f1909d0269d3b145ba19f7efb4176b5832347ba58fed9cfa084" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.07546718418598175, + 0.00034948240499943495, + 0.263406366109848, + 0.557631254196167, + 0.009013270027935505, + 0.01908109337091446, + 0.007019542157649994, + 0.016839005053043365, + 0.009013270027935505, + 0.0025823453906923532, + 0.004824455827474594, + 0.003315796609967947, + 0.002011132426559925, + 0.0022789116483181715, + 0.00619472423568368, + 0.0010764816543087363, + 0.0037572900764644146, + 0.0006529190577566624, + 0.002011132426559925, + 0.0012198134791105986, + 0.00044874430750496686, + 0.0005084939184598625, + 0.003315796609967947, + 0.0013822298496961594, + 0.0017748181708157063, + 0.004824455827474594 + ], + "max_abs_difference": 0.0, + "latency_ms": 141.8488211929798 + }, + { + "name": "structured", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 469 + ], + "dtype": "torch.int64", + "sha256": "51435b8464d5ef7dde621ff5dd7679036480601f39496e693687a2b81d178553" + }, + "attention_mask": { + "shape": [ + 1, + 469 + ], + "dtype": "torch.int64", + "sha256": "75a539e8d5305e6bc6b6d4f53e3c5e06be4f04fcaa8bdb5995794b7cf09fe80f" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 469 + ], + "dtype": "torch.int64", + "sha256": "269da87e532542e6e885dfd9ac112e706ee7f033044955ba5020f6b844c86a0c" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.07443060725927353, + 0.9067503809928894, + 0.01881900243461132 + ], + "max_abs_difference": 0.0, + "latency_ms": 121.30219116806984 + }, + { + "name": "special-token", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 441 + ], + "dtype": "torch.int64", + "sha256": "6abf0a9290f3d2a19d5063d24e5c831eae8697c604e7206e7fcf83c4f1b013fe" + }, + "attention_mask": { + "shape": [ + 1, + 441 + ], + "dtype": "torch.int64", + "sha256": "07d34f72fa2eb8b18b38092dfb3bb5364429d5008739c6290d61fd68d500a2b0" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 441 + ], + "dtype": "torch.int64", + "sha256": "7ffc7201ce92882bf791a82175d36c9f7910e5bdba09d335cac9b8d6640b6caa" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.531209409236908, + 0.4687906503677368 + ], + "max_abs_difference": 0.0, + "latency_ms": 117.51110758632421 + }, + { + "name": "two-questions", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" + }, + "attention_mask": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" + }, + "pixel_values": { + "shape": [ + 320, + 1536 + ], + "dtype": "torch.float32", + "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" + } + }, + "probabilities": [ + 0.9958575367927551, + 0.0005507932510226965, + 0.003591623157262802 + ], + "max_abs_difference": 0.0, + "latency_ms": 103.49029209464788 + }, + { + "name": "two-questions", + "question": "second", + "inputs": { + "input_ids": { + "shape": [ + 1, + 215 + ], + "dtype": "torch.int64", + "sha256": "1bc665c3f5aa0134e368d1a9edf0e652ec604cf43e969d59b14f696a4cf5c9c3" + }, + "attention_mask": { + "shape": [ + 1, + 215 + ], + "dtype": "torch.int64", + "sha256": "bbdd8ad88f622ea9585a6b471f3ef6810a0c3374e43dbcb747e3077de4e6e674" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 215 + ], + "dtype": "torch.int64", + "sha256": "6dc930bb14d93a028e6d2db398b391c4fe5af99184ec6625d38d2f7820e182f3" + }, + "pixel_values": { + "shape": [ + 320, + 1536 + ], + "dtype": "torch.float32", + "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" + } + }, + "probabilities": [ + 0.22270014882087708, + 0.7772998809814453 + ], + "max_abs_difference": 0.0, + "latency_ms": 104.30925991386175 + } + ], + "reference_source": { + "path": "libs/cua-s1/python/src/cua_s1/four_b.py", + "sha256": "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7", + "revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" + }, + "load_ms": 19939.03631903231, + "warmup_ms": 1815.6465897336602, + "adapter_modules": 178, + "peak_allocated_bytes": 9657058304 +} \ No newline at end of file diff --git a/recipe/cua_s1/experiments/rtx4090/frontend.json b/recipe/cua_s1/experiments/rtx4090/frontend.json new file mode 100644 index 0000000..39cee2e --- /dev/null +++ b/recipe/cua_s1/experiments/rtx4090/frontend.json @@ -0,0 +1,79 @@ +[ + { + "name": "health", + "status": 200, + "content_type": "application/json", + "body_sha256": "f54feed005f2b4d1d6f2398a306d13aeaaa0c90aea20b5cca2a4de4ecab5bdd7", + "identical": true + }, + { + "name": "26-options", + "status": 200, + "content_type": "application/json", + "body_sha256": "64889fe4b0e85dac1c361d29a4367ed27fe63ceb75d94166ee867ba11ceb83d3", + "identical": true + }, + { + "name": "jpeg", + "status": 200, + "content_type": "application/json", + "body_sha256": "aeedac2faea6c8d9cc6e8636751833f5ee03b88dc8092f8a9d93e4320eb3f887", + "identical": true + }, + { + "name": "medium", + "status": 200, + "content_type": "application/json", + "body_sha256": "1d6041c8144e070b2d766cc1c85c52764addcb4745da88e96f94c5058678e62c", + "identical": true + }, + { + "name": "single", + "status": 200, + "content_type": "application/json", + "body_sha256": "6e9194fc25bc7ec8741dabd52ee4dfb3412b5c18eac22c2935b3ba96989c7a8c", + "identical": true + }, + { + "name": "small", + "status": 200, + "content_type": "application/json", + "body_sha256": "51fa5f1cf910140c958cc2c6ca3738699527dd47cee58809060ac6017400602e", + "identical": true + }, + { + "name": "special-token", + "status": 200, + "content_type": "application/json", + "body_sha256": "38646659005cdc82b81a5bac4ab8120f7f1e1b16bdb1c610fa72e5794ed5a439", + "identical": true + }, + { + "name": "structured", + "status": 200, + "content_type": "application/json", + "body_sha256": "21fe1ab16324d937eb8f9c6c692ccb1d1ad41d9d01cb078440dee0fde8990920", + "identical": true + }, + { + "name": "two-questions", + "status": 200, + "content_type": "application/json", + "body_sha256": "3f8dd307c92bf5dbf7e1ff1c928d3961265516e587ce162f905ded363f6764c1", + "identical": true + }, + { + "name": "invalid", + "status": 422, + "content_type": "application/json", + "body_sha256": "866753c516b9090204f79318fa8d289f42ef1650821ef4b974bc915ec2836832", + "identical": true + }, + { + "name": "duplicate", + "status": 422, + "content_type": "application/json", + "body_sha256": "1175f6c373a386204fa01db801fac92f0b772d5f2ef09a43e788ac72ae156a27", + "identical": true + } +] \ No newline at end of file diff --git a/recipe/cua_s1/experiments/rtx4090/packages.txt b/recipe/cua_s1/experiments/rtx4090/packages.txt new file mode 100644 index 0000000..646c8bb --- /dev/null +++ b/recipe/cua_s1/experiments/rtx4090/packages.txt @@ -0,0 +1,62 @@ +accelerate==1.15.0 +annotated-doc==0.0.5 +anyio==4.15.1 +certifi==2026.7.22 +click==8.5.0 +cuda-bindings==13.4.2 +cuda-pathfinder==1.8.2 +cuda-toolkit==13.0.3.0 +filelock==4.0.1 +fsspec==2026.9.0 +h11==0.16.0 +hf-xet==1.6.0 +httpcore==1.0.9 +httpx==0.28.1 +huggingface-hub==1.32.0 +idna==3.20 +iniconfig==2.3.0 +jinja2==3.1.6 +markdown-it-py==4.2.0 +markupsafe==3.0.3 +mdurl==0.1.2 +mpmath==1.3.0 +networkx==3.6.1 +numpy==2.5.3 +nvidia-cublas==13.1.1.3 +nvidia-cuda-cupti==13.0.85 +nvidia-cuda-nvrtc==13.0.88 +nvidia-cuda-runtime==13.0.96 +nvidia-cudnn-cu13==9.24.0.43 +nvidia-cufft==12.0.0.61 +nvidia-cufile==1.15.1.6 +nvidia-curand==10.4.0.35 +nvidia-cusolver==12.0.4.66 +nvidia-cusparse==12.6.3.3 +nvidia-cusparselt-cu13==0.8.1 +nvidia-nccl-cu13==2.30.7 +nvidia-nvjitlink==13.4.92 +nvidia-nvshmem-cu13==3.4.5 +nvidia-nvtx==13.0.85 +packaging==26.3 +peft==0.21.0 +pillow==11.3.0 +pluggy==1.6.0 +psutil==7.2.2 +pygments==2.21.0 +pytest==9.1.1 +pyyaml==6.0.3 +regex==2026.9.10 +rich==15.0.0 +ruff==0.16.8 +safetensors==0.8.0 +setuptools==84.0.0 +shellingham==1.5.4 +sympy==1.14.0 +tokenizers==0.23.2 +torch==2.14.0 +torchvision==0.29.0 +tqdm==4.70.1 +transformers==5.17.0 +triton==3.8.0 +typer==0.27.2 +typing-extensions==4.16.0 diff --git a/recipe/cua_s1/experiments/rtx4090/reference.json b/recipe/cua_s1/experiments/rtx4090/reference.json new file mode 100644 index 0000000..eaba46f --- /dev/null +++ b/recipe/cua_s1/experiments/rtx4090/reference.json @@ -0,0 +1,527 @@ +{ + "environment": { + "python": "3.12.13", + "gpu": "NVIDIA GeForce RTX 4090", + "compute_capability": [ + 8, + 9 + ], + "cuda": "13.0", + "driver": "595.71.05", + "torch_num_threads": 64, + "torch_num_interop_threads": 64, + "packages": { + "torch": "2.14.0", + "torchvision": "0.29.0", + "transformers": "5.17.0", + "peft": "0.21.0", + "pillow": "11.3.0", + "safetensors": "0.8.0", + "triton": "3.8.0" + }, + "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", + "dtype": "bfloat16", + "adapter_merged": false + }, + "mode": "reference", + "cases": [ + { + "name": "small", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" + }, + "attention_mask": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" + }, + "pixel_values": { + "shape": [ + 320, + 1536 + ], + "dtype": "torch.float32", + "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" + } + }, + "probabilities": [ + 0.9958575367927551, + 0.0005507932510226965, + 0.003591623157262802 + ], + "latency_ms": 147.01103698462248 + }, + { + "name": "medium", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" + }, + "attention_mask": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.9958575367927551, + 0.0005507932510226965, + 0.003591623157262802 + ], + "latency_ms": 149.6963743120432 + }, + { + "name": "jpeg", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" + }, + "attention_mask": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 456 + ], + "dtype": "torch.int64", + "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "5a78a7f25b40888ad5c134d02826d691c27707384603242b99897cc623817bd8" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.9936226010322571, + 0.0011634123511612415, + 0.005214052740484476 + ], + "latency_ms": 137.3040061444044 + }, + { + "name": "single", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 431 + ], + "dtype": "torch.int64", + "sha256": "c4d7b1102e94984504efa938052a0bba3639ba85f9917de3e8af5426ceefd16b" + }, + "attention_mask": { + "shape": [ + 1, + 431 + ], + "dtype": "torch.int64", + "sha256": "3419eb840bd7f9d9e4b1fa6aaea783bae7a4752068701794687337702c2d8968" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 431 + ], + "dtype": "torch.int64", + "sha256": "077d91126e88fd3e8ba97a2d8079c28a4d5da388e56c89ca0e584fd5ad4b3641" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 1.0 + ], + "latency_ms": 135.01203525811434 + }, + { + "name": "26-options", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 752 + ], + "dtype": "torch.int64", + "sha256": "726f7d304140077e32490233aba734494431d93b7e2885bcb72d8b1c730e022b" + }, + "attention_mask": { + "shape": [ + 1, + 752 + ], + "dtype": "torch.int64", + "sha256": "389336aff30134c9af5da75eb0a9694eef903cde261cd8741f8aafae5ff94dc7" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 752 + ], + "dtype": "torch.int64", + "sha256": "e4cb23e2c28d9f1909d0269d3b145ba19f7efb4176b5832347ba58fed9cfa084" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.07546718418598175, + 0.00034948240499943495, + 0.263406366109848, + 0.557631254196167, + 0.009013270027935505, + 0.01908109337091446, + 0.007019542157649994, + 0.016839005053043365, + 0.009013270027935505, + 0.0025823453906923532, + 0.004824455827474594, + 0.003315796609967947, + 0.002011132426559925, + 0.0022789116483181715, + 0.00619472423568368, + 0.0010764816543087363, + 0.0037572900764644146, + 0.0006529190577566624, + 0.002011132426559925, + 0.0012198134791105986, + 0.00044874430750496686, + 0.0005084939184598625, + 0.003315796609967947, + 0.0013822298496961594, + 0.0017748181708157063, + 0.004824455827474594 + ], + "latency_ms": 314.77690767496824 + }, + { + "name": "structured", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 469 + ], + "dtype": "torch.int64", + "sha256": "51435b8464d5ef7dde621ff5dd7679036480601f39496e693687a2b81d178553" + }, + "attention_mask": { + "shape": [ + 1, + 469 + ], + "dtype": "torch.int64", + "sha256": "75a539e8d5305e6bc6b6d4f53e3c5e06be4f04fcaa8bdb5995794b7cf09fe80f" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 469 + ], + "dtype": "torch.int64", + "sha256": "269da87e532542e6e885dfd9ac112e706ee7f033044955ba5020f6b844c86a0c" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.07443060725927353, + 0.9067503809928894, + 0.01881900243461132 + ], + "latency_ms": 141.84184558689594 + }, + { + "name": "special-token", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 441 + ], + "dtype": "torch.int64", + "sha256": "6abf0a9290f3d2a19d5063d24e5c831eae8697c604e7206e7fcf83c4f1b013fe" + }, + "attention_mask": { + "shape": [ + 1, + 441 + ], + "dtype": "torch.int64", + "sha256": "07d34f72fa2eb8b18b38092dfb3bb5364429d5008739c6290d61fd68d500a2b0" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 441 + ], + "dtype": "torch.int64", + "sha256": "7ffc7201ce92882bf791a82175d36c9f7910e5bdba09d335cac9b8d6640b6caa" + }, + "pixel_values": { + "shape": [ + 1200, + 1536 + ], + "dtype": "torch.float32", + "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" + } + }, + "probabilities": [ + 0.531209409236908, + 0.4687906503677368 + ], + "latency_ms": 140.13662841171026 + }, + { + "name": "two-questions", + "question": "next", + "inputs": { + "input_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" + }, + "attention_mask": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 236 + ], + "dtype": "torch.int64", + "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" + }, + "pixel_values": { + "shape": [ + 320, + 1536 + ], + "dtype": "torch.float32", + "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" + } + }, + "probabilities": [ + 0.9958575367927551, + 0.0005507932510226965, + 0.003591623157262802 + ], + "latency_ms": 114.18061796575785 + }, + { + "name": "two-questions", + "question": "second", + "inputs": { + "input_ids": { + "shape": [ + 1, + 215 + ], + "dtype": "torch.int64", + "sha256": "1bc665c3f5aa0134e368d1a9edf0e652ec604cf43e969d59b14f696a4cf5c9c3" + }, + "attention_mask": { + "shape": [ + 1, + 215 + ], + "dtype": "torch.int64", + "sha256": "bbdd8ad88f622ea9585a6b471f3ef6810a0c3374e43dbcb747e3077de4e6e674" + }, + "mm_token_type_ids": { + "shape": [ + 1, + 215 + ], + "dtype": "torch.int64", + "sha256": "6dc930bb14d93a028e6d2db398b391c4fe5af99184ec6625d38d2f7820e182f3" + }, + "pixel_values": { + "shape": [ + 320, + 1536 + ], + "dtype": "torch.float32", + "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" + }, + "image_grid_thw": { + "shape": [ + 1, + 3 + ], + "dtype": "torch.int64", + "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" + } + }, + "probabilities": [ + 0.22270014882087708, + 0.7772998809814453 + ], + "latency_ms": 117.08385031670332 + } + ], + "reference_source": { + "path": "libs/cua-s1/python/src/cua_s1/four_b.py", + "sha256": "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7", + "revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" + }, + "artifact_verification_ms": 9541.830752044916, + "load_ms": 16820.161445997655, + "warmup_ms": 3665.197447873652, + "peak_allocated_bytes": 9655206912 +} \ No newline at end of file diff --git a/recipe/cua_s1/make_example.py b/recipe/cua_s1/make_example.py new file mode 100644 index 0000000..8adbc53 --- /dev/null +++ b/recipe/cua_s1/make_example.py @@ -0,0 +1,71 @@ +"""Create redistributable synthetic GUI fixtures and inline-image requests.""" + +from __future__ import annotations + +import argparse +import base64 +import json +from pathlib import Path + +from PIL import Image, ImageDraw + + +def make_example(path: Path, size=(640, 480), fmt="PNG") -> dict: + image = Image.new("RGB", size, "#f4f6f8") + draw = ImageDraw.Draw(image) + w, h = size + draw.rectangle( + (w // 10, h // 8, w * 9 // 10, h * 7 // 8), + fill="white", + outline="#8899aa", + width=2, + ) + draw.text( + (w // 7, h // 5), "Account settings", fill="black", font_size=max(12, w // 25) + ) + draw.text( + (w // 7, h // 3), + "Display name: Alice", + fill="black", + font_size=max(10, w // 32), + ) + draw.rectangle((w // 7, h // 2, w * 4 // 7, h * 2 // 3), fill="#1460b4") + draw.text( + (w // 6, h * 13 // 24), "Save changes", fill="white", font_size=max(10, w // 32) + ) + draw.text( + (w * 5 // 8, h * 13 // 24), "Cancel", fill="black", font_size=max(10, w // 32) + ) + image.save(path, format=fmt) + mime = "jpeg" if fmt == "JPEG" else "png" + return { + "model": "cua-s1-4b-0.2", + "state": { + "image": f"data:image/{mime};base64," + + base64.b64encode(path.read_bytes()).decode() + }, + "questions": { + "next": { + "type": "choice", + "instructions": "Save the changed display name.", + "criteria": { + "save": "Click Save changes", + "cancel": "Click Cancel", + "wait": "Wait", + }, + } + }, + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, default=Path("example.json")) + args = parser.parse_args() + args.output.parent.mkdir(parents=True, exist_ok=True) + value = make_example(args.output.with_suffix(".png")) + args.output.write_text(json.dumps(value, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/recipe/cua_s1/requirements-multimodal.txt b/recipe/cua_s1/requirements-multimodal.txt new file mode 100644 index 0000000..afb7422 --- /dev/null +++ b/recipe/cua_s1/requirements-multimodal.txt @@ -0,0 +1,11 @@ +# Reference package versions from trycua/cua 0e75660, four-b uv.lock. +# Lock the CUDA wheel/driver in the experiment report for the target GPU. +torch==2.14.0 +torchvision==0.29.0 +transformers==5.17.0 +peft==0.21.0 +accelerate==1.15.0 +Pillow==11.3.0 +safetensors==0.8.0 +huggingface-hub==1.32.0 +tokenizers==0.23.2 diff --git a/src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md b/src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md new file mode 100644 index 0000000..b8b198c --- /dev/null +++ b/src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2025 Cua AI, Inc. + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/src/models/cua_s1/multimodal/model.py b/src/models/cua_s1/multimodal/model.py new file mode 100644 index 0000000..db55070 --- /dev/null +++ b/src/models/cua_s1/multimodal/model.py @@ -0,0 +1,164 @@ +"""Direct Transformers/PEFT execution. No production dependency on cua_s1.""" + +from __future__ import annotations + +import hashlib +import json +from pathlib import Path + +from .protocol import InvalidRequest, Question, Request, answer, build_messages + +REFERENCE_REVISION = "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" +BASE_REVISION = "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a" +ADAPTER_REVISION = "16818868b0cc7813808aae4e87b417657046ab79" +IDENTITY = f"cua-ai/cua-s1-4b-0.2@{ADAPTER_REVISION}:multimodal" +MAX_TOKENS = 4096 + + +def letter_ids(tokenizer, count: int) -> list[int]: + ids = [] + for index in range(count): + encoded = tokenizer.encode(chr(65 + index), add_special_tokens=False) + if len(encoded) != 1: + raise ValueError("each candidate letter must be a single token") + ids.append(encoded[0]) + return ids + + +def validate_adapter_config(config: dict): + targets = { + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj", + "linear_fc1", + "linear_fc2", + } + if ( + config.get("peft_type") != "LORA" + or config.get("r") != 16 + or config.get("lora_alpha") != 32 + or set(config.get("target_modules", [])) != targets + or config.get("base_model_name_or_path") != "Qwen/Qwen3.5-4B" + ): + raise ValueError("expected the pinned 0.2 multimodal LoRA adapter") + + +def verify_weights(base: Path, adapter: Path): + """Check local artifacts before assigning the pinned identity to responses.""" + lock = json.loads(Path(__file__).with_name("weights.lock.json").read_text()) + allowed = {base: set(), adapter: set()} + for artifact in lock["artifacts"]: + for name, expected in artifact["files"].items(): + if artifact["role"] == "adapter": + if not name.startswith("multimodal/"): + continue + path = adapter / name.removeprefix("multimodal/") + else: + path = base / name + root = adapter if artifact["role"] == "adapter" else base + allowed[root].add(path.relative_to(root).as_posix()) + if not path.is_file() or path.stat().st_size != expected["size"]: + raise ValueError(f"missing or wrong-size pinned artifact: {path.name}") + with path.open("rb") as handle: + digest = hashlib.file_digest(handle, "sha256").hexdigest() + if digest != expected["sha256"]: + raise ValueError(f"checksum mismatch: {path.name}") + for root, names in allowed.items(): + for path in root.rglob("*"): + relative = path.relative_to(root) + if path.is_file() and relative.parts[0] != ".cache": + if relative.as_posix() not in names: + raise ValueError( + f"unlisted artifact may override pinned files: {relative}" + ) + + +class MultimodalEngine: + def __init__( + self, base: str, adapter: str, device: str = "cuda", dtype: str = "bfloat16" + ): + import torch + from peft import PeftModel + from peft.tuners.lora import LoraLayer + from transformers import ( + AutoModelForImageTextToText, + AutoProcessor, + AutoTokenizer, + ) + + base_path, adapter_path = Path(base), Path(adapter) + verify_weights(base_path, adapter_path) + validate_adapter_config( + json.loads((adapter_path / "adapter_config.json").read_text()) + ) + self.tokenizer = AutoTokenizer.from_pretrained(base, local_files_only=True) + self.processor = AutoProcessor.from_pretrained(base, local_files_only=True) + model = AutoModelForImageTextToText.from_pretrained( + base, + torch_dtype=getattr(torch, dtype), + device_map=device, + local_files_only=True, + ) + self.model = PeftModel.from_pretrained(model, adapter, local_files_only=True) + modules = [ + name + for name, module in self.model.named_modules() + if isinstance(module, LoraLayer) + ] + if len(modules) != 178 or not any(".visual." in name for name in modules): + raise RuntimeError( + "multimodal adapter did not attach to all 178 expected modules" + ) + self.adapter_modules = len(modules) + self.model.eval() + self.dtype = dtype + + def prepare(self, image, question: Question): + messages = build_messages(question) + text = self.processor.apply_chat_template( + messages, tokenize=False, add_generation_prompt=True + ) + inputs = self.processor(text=[text], images=[image], return_tensors="pt") + if inputs["input_ids"].shape[-1] > MAX_TOKENS: + raise InvalidRequest(f"processed prompt exceeds {MAX_TOKENS} tokens") + return inputs + + def score(self, inputs, question: Question) -> list[float]: + import torch + + ids = letter_ids(self.tokenizer, len(question.keys)) + inputs = inputs.to(self.model.device) + with torch.no_grad(): + output = self.model(**inputs) + logits = output.logits[0, -1, :] + return torch.softmax( + logits[torch.tensor(ids, device=logits.device)].float(), dim=-1 + ).tolist() + + def predict(self, request: Request) -> dict: + # Validate all processed lengths before executing any question. + prepared = [self.prepare(request.image, q) for q in request.questions] + answers = { + q.name: answer(q, self.score(inputs, q)) + for q, inputs in zip(request.questions, prepared) + } + return { + "model": IDENTITY, + "answers": answers, + "usage": { + "input_tokens": sum(x["input_ids"].shape[-1] for x in prepared), + "output_tokens": 0, + }, + } + + def warmup(self): + from PIL import Image + + q = Question( + "warmup", ("continue", "cancel"), ("Continue", "Cancel"), "Continue" + ) + self.predict(Request(Image.new("RGB", (224, 224), "white"), (q,))) diff --git a/src/models/cua_s1/multimodal/protocol.py b/src/models/cua_s1/multimodal/protocol.py new file mode 100644 index 0000000..0a0157a --- /dev/null +++ b/src/models/cua_s1/multimodal/protocol.py @@ -0,0 +1,217 @@ +"""Bounded screenshot-only extension of the Cua-S1 choice contract in PR #11.""" + +from __future__ import annotations + +import base64 +import binascii +import io +import json +import math +from dataclasses import dataclass + +from PIL import Image, UnidentifiedImageError + +MODEL = "cua-s1-4b-0.2" +MAX_BODY = 8 * 1024 * 1024 +MAX_IMAGE_BYTES = 4 * 1024 * 1024 +MAX_PIXELS = 1024 * 1024 +MAX_SIDE = 2048 +MAX_QUESTIONS = 8 +MAX_TEXT = 16384 + +# Prompt and letter layout follow trycua/cua at 0e75660ce4c2edda519e0c795fa3ad98abf4e76f. +# See THIRD_PARTY_NOTICES.md for the upstream MIT notice. +SYSTEM_PROMPT = ( + "You are a one-pass computer-use decision model. You are shown the " + "current state of a screen and a fixed, closed list of candidate " + "(element, action) options, each given a single letter. Choose exactly " + "one option: the single best next action to take. Answer with ONLY that " + "option's letter -- no words, no punctuation, no explanation." +) + + +class InvalidRequest(ValueError): + """Input cannot be evaluated under the supported contract.""" + + +@dataclass(frozen=True) +class Question: + name: str + keys: tuple[str, ...] + labels: tuple[str, ...] + goal: str + + +@dataclass(frozen=True) +class Request: + image: Image.Image + questions: tuple[Question, ...] + + +def _object(pairs): + result = {} + for key, value in pairs: + if key in result: + raise InvalidRequest("duplicate JSON keys are not supported") + result[key] = value + return result + + +def _nonfinite(value): + raise InvalidRequest("non-finite JSON numbers are not supported") + + +def decode_request(raw: bytes) -> dict: + if len(raw) > MAX_BODY: + raise InvalidRequest("request body exceeds 8 MiB") + try: + value = json.loads(raw, object_pairs_hook=_object, parse_constant=_nonfinite) + except (ValueError, UnicodeError, RecursionError) as exc: + raise InvalidRequest("invalid JSON or duplicate keys") from exc + if not isinstance(value, dict): + raise InvalidRequest("request must be a JSON object") + return value + + +def _text(value, field): + if not isinstance(value, (str, dict, list)): + raise InvalidRequest(f"{field} must be a string, object or array") + try: + result = ( + value + if isinstance(value, str) + else json.dumps(value, ensure_ascii=False, allow_nan=False) + ) + except (ValueError, TypeError, RecursionError) as exc: + raise InvalidRequest(f"invalid {field}") from exc + if len(result) > MAX_TEXT: + raise InvalidRequest(f"{field} exceeds {MAX_TEXT} characters") + if any( + token in result + for token in ( + "<|image_pad|>", + "<|video_pad|>", + "<|vision_start|>", + "<|vision_end|>", + ) + ): + raise InvalidRequest(f"{field} contains an unsupported media control token") + return result + + +def _image(state): + if not isinstance(state, dict) or set(state) != {"image"}: + raise InvalidRequest("state must contain exactly one image data URL") + url = state["image"] + if not isinstance(url, str): + raise InvalidRequest("state.image must be a PNG/JPEG base64 data URL") + prefix, separator, encoded = url.partition(",") + expected = {"data:image/png;base64": "PNG", "data:image/jpeg;base64": "JPEG"} + if not separator or prefix not in expected: + raise InvalidRequest("only inline PNG/JPEG images are supported") + if len(encoded) > 4 * ((MAX_IMAGE_BYTES + 2) // 3): + raise InvalidRequest("encoded image exceeds 4 MiB") + try: + raw = base64.b64decode(encoded, validate=True) + if len(raw) > MAX_IMAGE_BYTES: + raise InvalidRequest("image exceeds 4 MiB") + with Image.open(io.BytesIO(raw)) as source: + if source.format != expected[prefix]: + raise InvalidRequest("image format does not match its MIME type") + w, h = source.size + if ( + max(w, h) > MAX_SIDE + or w * h > MAX_PIXELS + or getattr(source, "n_frames", 1) != 1 + ): + raise InvalidRequest( + "image must be single-frame, at most 2048 per side and 1048576 pixels" + ) + source.load() + return source.convert("RGB") + except ( + binascii.Error, + UnidentifiedImageError, + OSError, + Image.DecompressionBombError, + ValueError, + ) as exc: + if isinstance(exc, InvalidRequest): + raise + raise InvalidRequest("invalid image data") from exc + + +def parse_request(value: dict) -> Request: + if not isinstance(value, dict) or set(value) != {"model", "state", "questions"}: + raise InvalidRequest("request must contain model, state and questions only") + if value["model"] != MODEL: + raise InvalidRequest(f"model must be {MODEL}") + questions = value["questions"] + if not isinstance(questions, dict) or not 1 <= len(questions) <= MAX_QUESTIONS: + raise InvalidRequest("questions must contain 1 to 8 questions") + parsed = [] + for name, q in questions.items(): + if not isinstance(name, str) or not name or len(name) > 256: + raise InvalidRequest("question names must contain 1 to 256 characters") + if not isinstance(q, dict) or q.get("type") != "choice": + raise InvalidRequest("only choice questions are supported") + if set(q) - {"type", "instructions", "criteria"}: + raise InvalidRequest("unsupported question fields") + criteria = q.get("criteria") + if not isinstance(criteria, dict) or not 1 <= len(criteria) <= 26: + raise InvalidRequest("choice requires 1 to 26 options") + labels = [] + for key, label in criteria.items(): + if not isinstance(key, str) or not key or len(key) > 256: + raise InvalidRequest("option keys must contain 1 to 256 characters") + text = _text(key if label is None else label, "criteria") + labels.append(json.dumps(text, ensure_ascii=False)[1:-1]) + goal = _text(q.get("instructions", ""), "instructions") + if len(goal) + sum(map(len, labels)) > MAX_TEXT: + raise InvalidRequest("combined question text exceeds 16384 characters") + parsed.append(Question(name, tuple(criteria), tuple(labels), goal)) + return Request(_image(value["state"]), tuple(parsed)) + + +def build_messages(q: Question, image_marker: str = "inline.png") -> list[dict]: + lines = "\n".join( + f'{chr(65 + i)}. Decision "{label}" -> select' + for i, label in enumerate(q.labels) + ) + text = ( + (f"Goal: {q.goal}\n\n" if q.goal else "") + + "App: Cua Driver\nTask family: closed-candidate decision\n\n" + + "The current screenshot is attached.\n\n" + + f"Options:\n{lines}\n\nAnswer with a single letter." + ) + return [ + {"role": "system", "content": SYSTEM_PROMPT}, + { + "role": "user", + "content": [ + {"type": "image", "image": image_marker}, + {"type": "text", "text": text}, + ], + }, + ] + + +def answer(q: Question, probabilities: list[float]) -> dict: + if len(probabilities) != len(q.keys) or any( + not math.isfinite(p) or not 0 <= p <= 1 for p in probabilities + ): + raise ValueError("model returned invalid probabilities") + if not math.isclose(sum(probabilities), 1.0, abs_tol=1e-5): + raise ValueError("model probabilities do not sum to one") + n = len(probabilities) + confidence = ( + 1.0 + if n == 1 + else 1 + sum(p * math.log(p) for p in probabilities if p) / math.log(n) + ) + return { + "type": "choice", + "choice": q.keys[max(range(n), key=probabilities.__getitem__)], + "probabilities": dict(zip(q.keys, probabilities)), + "confidence": max(0.0, min(1.0, confidence)), + } diff --git a/src/models/cua_s1/multimodal/server.py b/src/models/cua_s1/multimodal/server.py new file mode 100644 index 0000000..a399dac --- /dev/null +++ b/src/models/cua_s1/multimodal/server.py @@ -0,0 +1,128 @@ +"""Small loopback HTTP worker; the Rust frontend remains the public serving layer.""" + +from __future__ import annotations + +import argparse +import json +import logging +import socket +import threading +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +from .protocol import MAX_BODY, InvalidRequest, decode_request, parse_request + +LOG = logging.getLogger(__name__) + + +class WorkerServer(ThreadingHTTPServer): + daemon_threads = True + + def __init__(self, address, engine): + self.engine = engine + self.inference_lock = threading.Lock() + super().__init__(address, Handler) + + +class Handler(BaseHTTPRequestHandler): + def setup(self): + super().setup() + self.connection.settimeout(15) + + def log_message(self, format, *args): + # Do not log paths, input images, instructions or arbitrary request headers. + pass + + def send_json(self, status, value): + raw = json.dumps(value, ensure_ascii=False, allow_nan=False).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(raw))) + self.end_headers() + try: + self.wfile.write(raw) + except (BrokenPipeError, ConnectionResetError): + pass + + def do_GET(self): + if self.path == "/health": + self.send_json(200, {"status": "ready", "modality": "multimodal"}) + else: + self.send_json(404, {"error": "unknown route"}) + + def do_POST(self): + if self.path != "/v1/systemone": + self.send_json(404, {"error": "unknown route"}) + return + if self.headers.get("Transfer-Encoding"): + self.send_json( + 411, + { + "error": "Content-Length is required; chunked requests are unsupported" + }, + ) + return + lengths = self.headers.get_all("Content-Length", []) + if len(lengths) != 1: + self.send_json(411, {"error": "one Content-Length is required"}) + return + try: + length = int(lengths[0]) + except ValueError: + self.send_json(400, {"error": "invalid Content-Length"}) + return + if length < 0 or length > MAX_BODY: + self.send_json(413, {"error": "request exceeds body limit"}) + return + if self.headers.get_content_type() != "application/json": + self.send_json(415, {"error": "Content-Type must be application/json"}) + return + if not self.server.inference_lock.acquire(blocking=False): + self.send_json(503, {"error": "worker busy"}) + return + try: + raw = self.rfile.read(length) + if len(raw) != length: + self.send_json(400, {"error": "incomplete body"}) + return + parsed = parse_request(decode_request(raw)) + result = self.server.engine.predict(parsed) + self.send_json(200, result) + except InvalidRequest as exc: + self.send_json(422, {"error": str(exc)}) + except (TimeoutError, socket.timeout): + self.send_json(408, {"error": "request body timed out"}) + except Exception as exc: + LOG.error("inference failed: %s", type(exc).__name__) + self.send_json(500, {"error": "inference failed"}) + finally: + self.server.inference_lock.release() + + +def main(): + from .model import MultimodalEngine + + p = argparse.ArgumentParser(description=__doc__) + p.add_argument( + "--base", required=True, help="verified local base checkpoint directory" + ) + p.add_argument( + "--adapter", required=True, help="verified local multimodal adapter directory" + ) + p.add_argument("--port", type=int, default=8000) + args = p.parse_args() + logging.basicConfig(level=logging.INFO) + engine = MultimodalEngine(args.base, args.adapter) + engine.warmup() + # Bind only after model loading and a representative inference succeed. + server = WorkerServer(("127.0.0.1", args.port), engine) + LOG.info("multimodal worker ready on 127.0.0.1:%s", args.port) + try: + server.serve_forever() + except KeyboardInterrupt: + pass + finally: + server.server_close() + + +if __name__ == "__main__": + main() diff --git a/src/models/cua_s1/multimodal/weights.lock.json b/src/models/cua_s1/multimodal/weights.lock.json new file mode 100644 index 0000000..dfb9748 --- /dev/null +++ b/src/models/cua_s1/multimodal/weights.lock.json @@ -0,0 +1,102 @@ +{ + "schema": "cua-s1/weights-lock/v1", + "description": "Pinned public Hugging Face artifacts for the weights-backed Cua-S1-4B smoke. Every listed file is downloaded at the pinned revision and verified by size and SHA-256; keep in sync with libs/cua-s1/README.md.", + "artifacts": [ + { + "name": "Qwen3.5-4B", + "repo_id": "Qwen/Qwen3.5-4B", + "revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "role": "base", + "files": { + ".gitattributes": { + "size": 1570, + "sha256": "34448b82c17d60fec9b65b1f093c115ddbaadc04beb1b0140b6bfed2e012a930" + }, + "LICENSE": { + "size": 11544, + "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" + }, + "README.md": { + "size": 77661, + "sha256": "1406be1b6b8fd8a6545870da516912804756593628a1d0fb0a7965211e82a7bb" + }, + "chat_template.jinja": { + "size": 7756, + "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" + }, + "config.json": { + "size": 3161, + "sha256": "ddc63e1c717afa86c865bb5e01313d89d72bb53b97ad4a8a03ba8510c0621670" + }, + "merges.txt": { + "size": 3353259, + "sha256": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d" + }, + "model.safetensors-00001-of-00002.safetensors": { + "size": 5329398688, + "sha256": "26a93f066e1916adb13453dae5a0c707c0fbc71299ed98779571a907b8e74c61" + }, + "model.safetensors-00002-of-00002.safetensors": { + "size": 3990429408, + "sha256": "cb544bd9bfae93dc59b0f22b292f5933573854a7f9b97835c67060d7d910e188" + }, + "model.safetensors.index.json": { + "size": 76196, + "sha256": "cf3f798ee02ba45f9622aa8892a47369ab667d0afbf154ee7c2212de42e6302d" + }, + "preprocessor_config.json": { + "size": 390, + "sha256": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516" + }, + "tokenizer.json": { + "size": 12807982, + "sha256": "5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42" + }, + "tokenizer_config.json": { + "size": 16710, + "sha256": "316230d6a809701f4db5ea8f8fc862bc3a6f3229c937c174e674ff3ca0a64ac8" + }, + "video_preprocessor_config.json": { + "size": 385, + "sha256": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13" + }, + "vocab.json": { + "size": 6722759, + "sha256": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003" + } + } + }, + { + "name": "cua-s1-4b-0.2", + "repo_id": "cua-ai/cua-s1-4b-0.2", + "revision": "16818868b0cc7813808aae4e87b417657046ab79", + "role": "adapter", + "files": { + ".gitattributes": { + "size": 1519, + "sha256": "11ad7efa24975ee4b0c3c3a38ed18737f0658a5f75a0a96787b576a78a023361" + }, + "README.md": { + "size": 1134, + "sha256": "0e89cb4aea205e08c13f8af7fbfdfd0854ba0f6c57602071238771d0e1f8c7cd" + }, + "multimodal/adapter_config.json": { + "size": 1129, + "sha256": "f80edac43dd7605317ae8189613c75d13a029fa5d8a4870c04f1ba7097a8fab5" + }, + "multimodal/adapter_model.safetensors": { + "size": 101665112, + "sha256": "38ecd5a9191a436c1739db29104517f0aff4c70d0736d9cd810d3d49cfd652ea" + }, + "text/adapter_config.json": { + "size": 1091, + "sha256": "c246fce1fe1d44160ae5f246f9881dfeabb4fcd67fe4a777cef10a1dbcf16540" + }, + "text/adapter_model.safetensors": { + "size": 84968408, + "sha256": "9b59c5aed96171a50b26526613766bbf44347a5c7af70f81efe6bcc6e9dbfb0e" + } + } + } + ] +} diff --git a/tests/cua_s1/test_evaluation.py b/tests/cua_s1/test_evaluation.py new file mode 100644 index 0000000..b35ec7c --- /dev/null +++ b/tests/cua_s1/test_evaluation.py @@ -0,0 +1,48 @@ +"""Provenance checks must fail before a changed oracle can certify parity.""" + +import importlib.util +from pathlib import Path + +import pytest + + +def test_modified_reference_is_rejected(tmp_path, monkeypatch): + recipe = Path(__file__).resolve().parents[2] / "recipe/cua_s1" + monkeypatch.syspath_prepend(str(recipe)) + spec = importlib.util.spec_from_file_location( + "evaluation", recipe / "evaluate_multimodal.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + source = tmp_path / module.REFERENCE_SOURCE + source.parent.mkdir(parents=True) + source.write_text("modified oracle") + with pytest.raises(ValueError, match="reference source differs"): + module.verify_reference(tmp_path) + + +def test_environment_survives_json_roundtrip(monkeypatch): + import json + import sys + from types import SimpleNamespace + + recipe = Path(__file__).resolve().parents[2] / "recipe/cua_s1" + monkeypatch.syspath_prepend(str(recipe)) + spec = importlib.util.spec_from_file_location( + "evaluation", recipe / "evaluate_multimodal.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + torch = SimpleNamespace( + cuda=SimpleNamespace( + get_device_name=lambda: "GPU", get_device_capability=lambda: (8, 9) + ), + version=SimpleNamespace(cuda="13.0"), + get_num_threads=lambda: 16, + get_num_interop_threads=lambda: 16, + ) + monkeypatch.setitem(sys.modules, "torch", torch) + monkeypatch.setattr(module.importlib.metadata, "version", lambda name: "pinned") + monkeypatch.setattr(module.subprocess, "check_output", lambda *a, **kw: "driver\n") + environment = module.environment() + assert json.loads(json.dumps(environment)) == environment diff --git a/tests/cua_s1/test_model.py b/tests/cua_s1/test_model.py new file mode 100644 index 0000000..228e3ac --- /dev/null +++ b/tests/cua_s1/test_model.py @@ -0,0 +1,112 @@ +import pytest + +from models.cua_s1.multimodal.model import letter_ids, validate_adapter_config + + +class Tokenizer: + def encode(self, text, add_special_tokens): + assert add_special_tokens is False + return [ord(text) - 33] + + +def test_letter_readout_is_in_candidate_order(): + assert letter_ids(Tokenizer(), 3) == [32, 33, 34] + + +def test_multitoken_letters_are_rejected(): + class Bad: + def encode(self, *args, **kwargs): + return [1, 2] + + with pytest.raises(ValueError, match="single token"): + letter_ids(Bad(), 2) + + +def test_text_adapter_is_rejected_before_model_load(): + with pytest.raises(ValueError, match="multimodal"): + validate_adapter_config( + { + "peft_type": "LORA", + "r": 16, + "lora_alpha": 32, + "target_modules": ["q_proj", "k_proj"], + } + ) + + +def test_multimodal_adapter_contract(): + validate_adapter_config( + { + "peft_type": "LORA", + "r": 16, + "lora_alpha": 32, + "base_model_name_or_path": "Qwen/Qwen3.5-4B", + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj", + "linear_fc1", + "linear_fc2", + ], + } + ) + + +def test_unlisted_model_files_cannot_override_verified_shards(tmp_path, monkeypatch): + import hashlib + import json + + from models.cua_s1.multimodal import model + + base, adapter = tmp_path / "base", tmp_path / "adapter" + base.mkdir() + adapter.mkdir() + (base / "config.json").write_bytes(b"{}") + manifest = tmp_path / "weights.lock.json" + manifest.write_text( + json.dumps( + { + "artifacts": [ + { + "role": "base", + "files": { + "config.json": { + "size": 2, + "sha256": hashlib.sha256(b"{}").hexdigest(), + } + }, + } + ] + } + ) + ) + monkeypatch.setattr(model, "__file__", str(tmp_path / "model.py")) + model.verify_weights(base, adapter) + (base / "model.safetensors").write_bytes(b"override") + with pytest.raises(ValueError, match="unlisted"): + model.verify_weights(base, adapter) + + +def test_all_question_lengths_are_checked_before_inference(): + from models.cua_s1.multimodal.model import MultimodalEngine + from models.cua_s1.multimodal.protocol import InvalidRequest, Question, Request + + engine = object.__new__(MultimodalEngine) + first = Question("first", ("a",), ("A",), "") + second = Question("second", ("b",), ("B",), "") + forwarded = [] + + def prepare(image, question): + if question.name == "second": + raise InvalidRequest("processed prompt exceeds 4096 tokens") + return {} + + engine.prepare = prepare + engine.score = lambda inputs, q: forwarded.append(q) + with pytest.raises(InvalidRequest, match="4096"): + engine.predict(Request(None, (first, second))) + assert forwarded == [] diff --git a/tests/cua_s1/test_protocol.py b/tests/cua_s1/test_protocol.py new file mode 100644 index 0000000..0165eae --- /dev/null +++ b/tests/cua_s1/test_protocol.py @@ -0,0 +1,156 @@ +import base64 +import io +import json + +import pytest +from PIL import Image + +from models.cua_s1.multimodal.protocol import ( + InvalidRequest, + answer, + build_messages, + decode_request, + parse_request, +) + + +def image_url(fmt="PNG", size=(32, 32)): + out = io.BytesIO() + Image.new("RGB", size, "white").save(out, format=fmt) + mime = "jpeg" if fmt == "JPEG" else "png" + return f"data:image/{mime};base64," + base64.b64encode(out.getvalue()).decode() + + +def request(): + return { + "model": "cua-s1-4b-0.2", + "state": {"image": image_url()}, + "questions": { + "next": { + "type": "choice", + "instructions": "Submit the form", + "criteria": {"submit": "Submit", "cancel": "Cancel"}, + } + }, + } + + +def test_image_and_order_are_preserved(): + r = parse_request(request()) + assert r.image.mode == "RGB" and r.image.size == (32, 32) + assert r.questions[0].keys == ("submit", "cancel") + assert r.questions[0].labels == ("Submit", "Cancel") + + +def test_prompt_keeps_image_block_and_upstream_text(): + q = parse_request(request()).questions[0] + msg = build_messages(q) + assert msg[1]["content"][0]["type"] == "image" + assert msg[1]["content"][1]["text"] == ( + "Goal: Submit the form\n\nApp: Cua Driver\nTask family: closed-candidate decision\n\n" + "The current screenshot is attached.\n\nOptions:\n" + 'A. Decision "Submit" -> select\nB. Decision "Cancel" -> select\n\n' + "Answer with a single letter." + ) + + +def test_structured_values_and_escaping(): + r = request() + r["questions"]["next"]["instructions"] = {"目标": "提交"} + r["questions"]["next"]["criteria"] = {"fallback": None, "obj": {"x": '"\n'}} + q = parse_request(r).questions[0] + assert q.goal == '{"目标": "提交"}' + assert q.labels == ( + "fallback", + json.dumps(json.dumps({"x": '"\n'}, ensure_ascii=False), ensure_ascii=False)[ + 1:-1 + ], + ) + + +@pytest.mark.parametrize( + "state", + [ + {}, + {"image": "/etc/passwd"}, + {"image": "https://example.com/a.png"}, + {"image": "data:image/png;base64,!!"}, + {"image": image_url(), "text": "ignored"}, + ], +) +def test_invalid_images_and_unknown_state_fields(state): + r = request() + r["state"] = state + with pytest.raises(InvalidRequest): + parse_request(r) + + +def test_mime_mismatch_and_oversized_dimensions(): + for url in [ + image_url().replace("image/png", "image/jpeg"), + image_url(size=(2049, 1)), + ]: + r = request() + r["state"]["image"] = url + with pytest.raises(InvalidRequest): + parse_request(r) + + +@pytest.mark.parametrize( + "criteria", [{}, {str(i): "x" for i in range(27)}, {"a": 1}, {"a": True}] +) +def test_invalid_candidates(criteria): + r = request() + r["questions"]["next"]["criteria"] = criteria + with pytest.raises(InvalidRequest): + parse_request(r) + + +@pytest.mark.parametrize("kind", ["score", "noul"]) +def test_unsupported_question_rejects_entire_request(kind): + r = request() + r["questions"]["bad"] = {"type": kind, "instructions": "x"} + with pytest.raises(InvalidRequest): + parse_request(r) + + +def test_duplicate_keys_and_nonfinite_json(): + for raw in [ + b'{"model":1,"model":2}', + b'{"x":NaN}', + b'{"x":Infinity}', + b"[]", + b"not json", + ]: + with pytest.raises(InvalidRequest): + decode_request(raw) + + +def test_entropy_confidence_and_earliest_tie(): + q = parse_request(request()).questions[0] + result = answer(q, [0.5, 0.5]) + assert result["choice"] == "submit" + assert result["confidence"] == 0.0 + assert result["probabilities"] == {"submit": 0.5, "cancel": 0.5} + r = request() + r["questions"]["next"]["criteria"] = {"only": "Only"} + assert answer(parse_request(r).questions[0], [1.0])["confidence"] == 1.0 + + +def test_jpeg_supported(): + r = request() + r["state"]["image"] = image_url("JPEG") + assert parse_request(r).image.mode == "RGB" + + +@pytest.mark.parametrize( + "token", ["<|image_pad|>", "<|video_pad|>", "<|vision_start|>", "<|vision_end|>"] +) +@pytest.mark.parametrize("field", ["instructions", "criteria"]) +def test_media_control_tokens_are_rejected(token, field): + value = request() + value["questions"]["next"][field] = ( + token if field == "instructions" else {"a": {"text": token}} + ) + with pytest.raises(InvalidRequest, match="control token"): + parse_request(value) diff --git a/tests/cua_s1/test_server.py b/tests/cua_s1/test_server.py new file mode 100644 index 0000000..89eddaa --- /dev/null +++ b/tests/cua_s1/test_server.py @@ -0,0 +1,85 @@ +import json +import threading +import urllib.error +import urllib.request + +import pytest +from test_protocol import request + +from models.cua_s1.multimodal.protocol import MAX_BODY +from models.cua_s1.multimodal.server import WorkerServer + + +class Engine: + def predict(self, parsed): + return { + "model": "test:multimodal", + "answers": {q.name: {"type": "choice"} for q in parsed.questions}, + } + + +@pytest.fixture +def worker(): + server = WorkerServer(("127.0.0.1", 0), Engine()) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + yield server, f"http://127.0.0.1:{server.server_port}" + server.shutdown() + server.server_close() + thread.join() + + +def call(url, body=None, **headers): + data = None if body is None else json.dumps(body).encode() + req = urllib.request.Request( + url, data=data, headers={"Content-Type": "application/json", **headers} + ) + try: + with urllib.request.urlopen(req, timeout=5) as response: + return response.status, json.load(response) + except urllib.error.HTTPError as response: + return response.code, json.load(response) + + +def test_health_and_prediction(worker): + _, url = worker + assert call(url + "/health")[0] == 200 + status, body = call(url + "/v1/systemone", request()) + assert status == 200 and "next" in body["answers"] + + +def test_invalid_question_never_reaches_model(worker): + _, url = worker + r = request() + r["questions"]["next"]["type"] = "noul" + assert call(url + "/v1/systemone", r)[0] == 422 + + +def test_busy_worker_rejects_instead_of_queueing_gpu_work(worker): + server, url = worker + server.inference_lock.acquire() + try: + assert call(url + "/health")[0] == 200 + assert call(url + "/v1/systemone", request())[0] == 503 + finally: + server.inference_lock.release() + + +def test_body_limit_checked_before_reading(worker): + _, url = worker + assert ( + call(url + "/v1/systemone", {}, **{"Content-Length": str(MAX_BODY + 1)})[0] + == 413 + ) + + +def test_model_failure_is_not_reported_as_success(worker): + server, url = worker + + def fail(_): + raise RuntimeError("private file or input must not leak") + + server.engine.predict = fail + status, body = call(url + "/v1/systemone", request()) + assert status == 500 and "private" not in json.dumps(body) + assert not server.inference_lock.locked() From 63edd9bdc104c22363237e3970e36ec7d9871d3e Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Wed, 30 Sep 2026 00:17:46 +0800 Subject: [PATCH 2/4] chore: remove local experiment artifacts from PR Remove archived experiment outputs and assistant planning notes. Keep runtime code, reproducible tools, and model verification metadata unchanged. Link historical reports to an immutable commit and ignore future local artifacts. Remove only tests tied to the deleted historical evidence bundles. --- .gitignore | 4 + README.md | 2 +- recipe/cua_s1/README.md | 2 +- recipe/cua_s1/experiments/README.md | 136 ----- .../cua_s1/experiments/rtx4090/benchmark.json | 409 ------------- .../cua_s1/experiments/rtx4090/candidate.json | 536 ------------------ .../cua_s1/experiments/rtx4090/frontend.json | 79 --- .../cua_s1/experiments/rtx4090/packages.txt | 62 -- .../cua_s1/experiments/rtx4090/reference.json | 527 ----------------- 9 files changed, 6 insertions(+), 1751 deletions(-) delete mode 100644 recipe/cua_s1/experiments/README.md delete mode 100644 recipe/cua_s1/experiments/rtx4090/benchmark.json delete mode 100644 recipe/cua_s1/experiments/rtx4090/candidate.json delete mode 100644 recipe/cua_s1/experiments/rtx4090/frontend.json delete mode 100644 recipe/cua_s1/experiments/rtx4090/packages.txt delete mode 100644 recipe/cua_s1/experiments/rtx4090/reference.json diff --git a/.gitignore b/.gitignore index 07a95a7..b42bbc8 100644 --- a/.gitignore +++ b/.gitignore @@ -29,3 +29,7 @@ weights/ # macOS metadata .DS_Store + +# Local experiment outputs and assistant planning notes +/recipe/cua_s1/experiments/ +/docs/superpowers/ diff --git a/README.md b/README.md index 3bd7b64..2cca0d9 100644 --- a/README.md +++ b/README.md @@ -57,7 +57,7 @@ Validated coverage is listed by modality and execution path: | Model | Status | | --- | --- | | LAYA | [External worker](recipe/laya/README.md); model engine planned | -| [Cua-S1 4B 0.2](recipe/cua_s1/README.md) | Multimodal screenshot choices via Transformers/PEFT on RTX 4090 CUDA; [parity and measurements](recipe/cua_s1/experiments/README.md). Text serving, native CUDA kernels and Metal deferred. | +| [Cua-S1 4B 0.2](recipe/cua_s1/README.md) | Multimodal screenshot choices via Transformers/PEFT on RTX 4090 CUDA; [parity and measurements](https://github.com/Levius-Fubuki/system1-omni/blob/683d470669f19d5e9530cd8233727a0d4f1f8d29/recipe/cua_s1/experiments/README.md). Text serving, native CUDA kernels and Metal deferred. | CUDA and Metal coverage will be documented per model as implementations are added and validated. diff --git a/recipe/cua_s1/README.md b/recipe/cua_s1/README.md index 506d0c0..741a29e 100644 --- a/recipe/cua_s1/README.md +++ b/recipe/cua_s1/README.md @@ -14,7 +14,7 @@ worker uses the same request model name; deployment routing selects its modality ## Setup Run from this repository's root on Linux with an NVIDIA GPU. The measured CUDA -wheel, driver, GPU memory and results are recorded in [experiments](experiments/README.md). +wheel, driver, GPU memory and results are recorded in [experiments](https://github.com/Levius-Fubuki/system1-omni/blob/683d470669f19d5e9530cd8233727a0d4f1f8d29/recipe/cua_s1/experiments/README.md). Python 3.12 is required by the pinned environment. ```sh diff --git a/recipe/cua_s1/experiments/README.md b/recipe/cua_s1/experiments/README.md deleted file mode 100644 index 1fc834b..0000000 --- a/recipe/cua_s1/experiments/README.md +++ /dev/null @@ -1,136 +0,0 @@ -# RTX 4090 multimodal validation — 2026-09-27 - -This is a Transformers/PEFT CUDA baseline, with the adapter unmerged and full -logits, for the screenshot worker in [the recipe](../README.md). It does not -implement a native CUDA backend or claim a speedup over upstream. - -The checked-in GPU/HTTP results were measured at commit `74c95e5`. A subsequent -protocol-only revision aligned errors with the text-worker contract: `detail` -replaces `error`, malformed JSON/duplicate keys return 400, and `instructions` is -required (explicit `null` or an empty string omits the goal). The original -`frontend.json` therefore retains the **historical** duplicate-key status and -error-body hashes; it is not evidence for the revised error contract. Regression -tests cover the revised behavior without restarting the GPU instance. The -protocol revision passed 62 Python tests, 13 Rust tests, and 12 direct-versus- -frontend comparisons using the real worker HTTP handler with a CPU stub engine. -Those 12 checks validate transport and rejection behavior, not model inference. - -The generator now supplies `instructions: ""` for the second question of the -two-question fixture. This preserves its previous empty goal and prompt. Model -loading, image processing, prompt construction and scoring are unchanged; GPU -parity and performance have not been remeasured for the protocol revision. - -The 2026-09-29 review follow-up rejects image aspect ratios above 200:1 with -HTTP 422 before processing, matching pinned Transformers 5.17.0. Tests retain -acceptance at exactly 200:1 in either orientation. Lock cleanup checks now use -bounded acquisition, with a fixture that deliberately pauses the handler after -writing the response. All 70 Python tests passed with both pinned Pillow 11.3.0 -and Pillow 12.3.0; the 14 affected cleanup cases passed three additional runs in -the pinned environment. Rust's 13 tests, formatting and Clippy also passed. -These are CPU checks; the archived GPU measurements have not been rerun. - -## Environment and artifacts - -- One NVIDIA GeForce RTX 4090, 24,564 MiB, compute capability 8.9. -- Ubuntu 22.04 container; NVIDIA driver 595.71.05; PyTorch CUDA runtime 13.0. -- Python 3.12.13; torch 2.14.0, torchvision 0.29.0, Transformers 5.17.0, - PEFT 0.21.0, Pillow 11.3.0, safetensors 0.8.0, Triton 3.8.0. -- PyTorch reported 64 intra-op and 64 inter-op threads; no thread tuning applied. -- BF16 base, PEFT fp32 LoRA branches, no adapter merge, no quantization. -- Neither `flash-linear-attention` nor `causal-conv1d` is installed. The matching - upstream reference uses Transformers' PyTorch fallback implementations. -- All pinned weights passed size and SHA-256 verification. The lock contains - 9,342,907,469 base bytes and 186,638,393 adapter-repository bytes. Only the - `multimodal` adapter is loaded; all 178 target modules, including 50 visual - modules, are required at startup. - -Pinned revisions and the reference-source SHA-256 are present in every JSON -report. Full installed versions are in [packages.txt](rtx4090/packages.txt). -The numerical contract and dependency versions were fixed before comparisons. -No model weights or private screenshots are included. - -## Correctness - -| Check | Result | Evidence | -| --- | --- | --- | -| Processor tensor shapes, dtypes and SHA-256 values | Exact match on 9 question forwards across 8 requests | [reference.json](rtx4090/reference.json), [candidate.json](rtx4090/candidate.json) | -| Candidate fp32 probabilities | Exact match; maximum absolute difference **0** | Same reports | -| Multimodal LoRA attachment | 178 modules, visual modules present | [candidate.json](rtx4090/candidate.json) | -| Rust frontend vs direct worker | All 11 HTTP comparisons passed; status, content type and body bytes identical | [frontend.json](rtx4090/frontend.json) | -| CPU validation | 39 tests passed on local macOS and the Linux GPU host | `tests/cua_s1`, CPU CI workflow | - -Fixtures are generated by `evaluate_multimodal.py` using the pinned Pillow version. -They cover 320×240 and 640×480 screenshots, PNG/JPEG, 1 and 26 candidates, -structured/Unicode labels, special-token spellings and two questions sharing one -image. Processed prompts span 215–752 tokens. Hashes cover input IDs, attention -masks, image pixels and image-grid metadata, rather than just the selected option. -The fixtures establish integration parity, not GUI task accuracy or generalization. - -For the frontend test, PR #2's Rust frontend at -`0c91671ac7c8bd698b957b2c0df921de96e7e628` was built with `cargo build --release --locked` -on macOS. It forwarded to the Linux GPU worker over an SSH tunnel. This exercises -real inference and byte preservation; **network/HTTP latency is not benchmarked**. -Eight valid fixtures, health, malformed envelope and duplicate JSON keys were checked. - -Reproduce the HTTP check after generating the fixtures and starting both servers: - -```sh -python recipe/cua_s1/check_frontend.py \ - --worker http://127.0.0.1:8000 --frontend http://127.0.0.1:8080 \ - --fixtures /tmp/cua-evidence/fixtures --output /tmp/cua-evidence/frontend.json -``` - -## Warm engine performance - -[benchmark.json](rtx4090/benchmark.json) contains every sample and the operator -profile. One 640×480 PNG, three candidates, 456 processed tokens, batch size 1, -concurrency 1; five warmup requests followed by two runs of 50 requests. -`torch.cuda.synchronize()` brackets each measurement. p95 uses nearest rank. - -Timing covers `engine.predict`: chat-template application, processor work, host-to- -device transfer, forward pass, letter readout and answer construction. The request -has already been parsed and its image decoded. JSON parsing, image decoding, -HTTP, queueing, model load, artifact hashing and warmup are excluded. - -| Run | p50 | p95 | Serial throughput | Peak allocated | Peak reserved | -| --- | --- | --- | --- | --- | --- | -| 1 | 126.39 ms | 136.68 ms | 7.79 requests/s | 8.84 GiB | 9.07 GiB | -| 2 | 128.28 ms | 133.01 ms | 7.77 requests/s | 8.84 GiB | 9.07 GiB | - -The parity run's peak allocated memory across all nine forwards was 8.99 GiB. -These bounds describe the measured fixtures, not every request admitted by the -worker's 4096-token limit or a concurrent/batched deployment. - -Measured candidate construction, including weight hashing, was 19.94 s in the -parity process and 18.94 s in the benchmark process. Their initial 224×224 warmups -were 1.82 s and 1.63 s. Upstream load was 16.82 s plus 9.54 s of separately measured -artifact verification; its first 320×240 warmup was 3.67 s. These are individual -observations after importing PyTorch and probing the GPU, with different warmup -shapes and filesystem-cache histories; they are not a cold-start speed comparison. -Per-fixture timings in the parity reports are also single observations; the -reference includes image opening while candidate timing covers `score` only. - -## Profiling and the next optimization boundary - -A separate profiled inference, excluded from the latency samples, reported: - -| Operator | Calls | Self CUDA time | -| --- | --- | --- | -| `aten::mm` | 605 | 35.42 ms | -| `aten::copy_` | 1,642 | 9.55 ms | -| `aten::bmm` | 817 | 9.11 ms | -| `aten::addmm` | 98 | 5.89 ms | -| `aten::mul` | 1,103 | 5.76 ms | - -These are instrumented operator totals, not percentages of request wall time. -The raw profile includes both framework operators and CUDA kernels, which overlap; -do not add parent and child events or add kernel time to its enclosing operator. - -Matrix multiplication dominates the observed operator totals. A subsequent CUDA -optimization should first attribute those GEMMs and copies to vision, language -and LoRA branches using shapes/module ranges, then compare one bounded change -against this baseline. This profile alone does not justify replacing a specific -kernel or claiming a speedup. Gated DeltaNet/causal-convolution backend work should -be coordinated with the text-engine contributor because those language layers are -shared. This PR keeps the exact upstream execution path and contributes the -multimodal correctness and performance baseline needed for that work. diff --git a/recipe/cua_s1/experiments/rtx4090/benchmark.json b/recipe/cua_s1/experiments/rtx4090/benchmark.json deleted file mode 100644 index 7b69afc..0000000 --- a/recipe/cua_s1/experiments/rtx4090/benchmark.json +++ /dev/null @@ -1,409 +0,0 @@ -{ - "environment": { - "python": "3.12.13", - "gpu": "NVIDIA GeForce RTX 4090", - "compute_capability": [ - 8, - 9 - ], - "cuda": "13.0", - "driver": "595.71.05", - "torch_num_threads": 64, - "torch_num_interop_threads": 64, - "packages": { - "torch": "2.14.0", - "torchvision": "0.29.0", - "transformers": "5.17.0", - "peft": "0.21.0", - "pillow": "11.3.0", - "safetensors": "0.8.0", - "triton": "3.8.0" - }, - "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", - "dtype": "bfloat16", - "adapter_merged": false - }, - "mode": "benchmark", - "cases": [], - "reference_source": { - "path": "libs/cua-s1/python/src/cua_s1/four_b.py", - "sha256": "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7", - "revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" - }, - "load_ms": 18941.51310157031, - "warmup_ms": 1633.873094804585, - "adapter_modules": 178, - "benchmark": { - "case": "medium", - "batch_size": 1, - "concurrency": 1, - "warmup_requests": 5, - "runs": [ - { - "run": 1, - "latencies_ms": [ - 125.8343830704689, - 127.36847903579473, - 125.82478113472462, - 125.89417397975922, - 146.41341753304005, - 127.77413055300713, - 127.22665630280972, - 125.83817914128304, - 126.10162515193224, - 127.15142965316772, - 126.42081826925278, - 126.38038955628872, - 126.02570466697216, - 128.5353172570467, - 129.56660240888596, - 129.32745181024075, - 126.28413271158934, - 126.18042714893818, - 133.06267280131578, - 126.3929232954979, - 131.68340362608433, - 126.76465138792992, - 125.68666227161884, - 126.71545054763556, - 125.96531771123409, - 125.74396748095751, - 130.0713624805212, - 132.96419847756624, - 131.16213120520115, - 126.18762347847223, - 125.84684416651726, - 126.00735854357481, - 125.77000726014376, - 126.28292944282293, - 125.9002760052681, - 130.30958734452724, - 136.85170747339725, - 134.48695000261068, - 136.67899277061224, - 134.1363089159131, - 133.8470932096243, - 126.33792497217655, - 126.30761601030827, - 126.43217574805021, - 125.64307544380426, - 125.50272978842258, - 125.40635839104652, - 125.05617458373308, - 125.54042227566242, - 127.74347327649593 - ], - "p50_ms": 126.3866564258933, - "p95_ms": 136.67899277061224, - "requests_per_second_serial": 7.792244462582349, - "peak_allocated_bytes": 9496202240, - "peak_reserved_bytes": 9741271040 - }, - { - "run": 2, - "latencies_ms": [ - 126.58042088150978, - 126.69678498059511, - 126.67341157793999, - 129.0692137554288, - 127.07117944955826, - 129.27887961268425, - 133.4229400381446, - 128.44505812972784, - 127.24453490227461, - 126.93716865032911, - 130.17353229224682, - 131.4986888319254, - 130.03864511847496, - 131.02801516652107, - 129.7577489167452, - 128.6515649408102, - 132.58606754243374, - 128.7766983732581, - 131.9235684350133, - 128.00912745296955, - 127.52761784940958, - 129.2492775246501, - 132.3700835928321, - 128.50904930382967, - 127.91914585977793, - 131.54291920363903, - 129.6907477080822, - 133.75771697610617, - 129.1659427806735, - 126.58135313540697, - 126.16921309381723, - 126.81071180850267, - 128.0105598270893, - 128.40583361685276, - 127.4806559085846, - 127.57191434502602, - 128.10698058456182, - 127.0896615460515, - 127.29146610945463, - 133.00949800759554, - 128.90154495835304, - 128.37096769362688, - 127.91174557060003, - 127.8772447258234, - 128.1898096203804, - 127.69030872732401, - 126.3129971921444, - 129.61805891245604, - 126.81897450238466, - 126.8136901780963 - ], - "p50_ms": 128.28038865700364, - "p95_ms": 133.00949800759554, - "requests_per_second_serial": 7.765628438386997, - "peak_allocated_bytes": 9496202240, - "peak_reserved_bytes": 9741271040 - } - ] - }, - "profile": [ - { - "op": "aten::matmul", - "count": 1422, - "device_time_us": 44527.040000000154, - "self_device_time_us": 0.0, - "cpu_time_us": 53924.26100000019, - "self_cpu_time_us": 9931.453000001111 - }, - { - "op": "aten::linear", - "count": 703, - "device_time_us": 41310.13700000033, - "self_device_time_us": 0.0, - "cpu_time_us": 24486.469999999994, - "self_cpu_time_us": 1650.2199999998538 - }, - { - "op": "aten::mm", - "count": 605, - "device_time_us": 35415.818000000334, - "self_device_time_us": 35415.818000000334, - "cpu_time_us": 13464.22999999985, - "self_cpu_time_us": 8533.79799999894 - }, - { - "op": "ampere_bf16_s1688gemm_bf16_64x128_sliced1x2_ldg8_f2f_tn", - "count": 96, - "device_time_us": 17010.658999999934, - "self_device_time_us": 17010.658999999934, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::copy_", - "count": 1642, - "device_time_us": 9549.038000000242, - "self_device_time_us": 9549.038000000242, - "cpu_time_us": 25915.56300000043, - "self_cpu_time_us": 9625.815000000488 - }, - { - "op": "aten::bmm", - "count": 817, - "device_time_us": 9111.22199999982, - "self_device_time_us": 9109.750999999822, - "cpu_time_us": 20146.271999999823, - "self_cpu_time_us": 16179.460999999617 - }, - { - "op": "ampere_bf16_s1688gemm_bf16_128x128_ldg8_f2f_stages_32x1_tn", - "count": 56, - "device_time_us": 6589.621000000137, - "self_device_time_us": 6589.621000000137, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::addmm", - "count": 98, - "device_time_us": 5894.318999999992, - "self_device_time_us": 5894.318999999992, - "cpu_time_us": 2576.6010000000297, - "self_cpu_time_us": 1777.2650000000776 - }, - { - "op": "aten::mul", - "count": 1103, - "device_time_us": 5763.554999999398, - "self_device_time_us": 5763.554999999398, - "cpu_time_us": 13280.975000000395, - "self_cpu_time_us": 8342.606000000327 - }, - { - "op": "aten::to", - "count": 1361, - "device_time_us": 5300.117000000242, - "self_device_time_us": 0.0, - "cpu_time_us": 28756.462000000047, - "self_cpu_time_us": 1621.334000000551 - }, - { - "op": "aten::_to_copy", - "count": 1080, - "device_time_us": 5300.117000000242, - "self_device_time_us": 0.0, - "cpu_time_us": 27135.127999999495, - "self_cpu_time_us": 4165.844999999095 - }, - { - "op": "ampere_bf16_s16816gemm_bf16_128x64_ldg8_f2f_tn", - "count": 1, - "device_time_us": 4485.06700000001, - "self_device_time_us": 4485.06700000001, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::add", - "count": 1009, - "device_time_us": 4335.962000000129, - "self_device_time_us": 4335.962000000129, - "cpu_time_us": 11608.1780000002, - "self_cpu_time_us": 7166.728000000563 - }, - { - "op": "ampere_bf16_s1688gemm_bf16_128x64_sliced1x2_ldg8_relu_f2f_tn", - "count": 50, - "device_time_us": 3960.761000000013, - "self_device_time_us": 3960.761000000013, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::linalg_solve_triangular", - "count": 48, - "device_time_us": 3340.373000000007, - "self_device_time_us": 2853.693999999952, - "cpu_time_us": 4383.135000000191, - "self_cpu_time_us": 1135.9290000003602 - }, - { - "op": "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast > >(at::TensorIteratorBase&, at::native::BinaryFunctor > const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast > >(at::TensorIteratorBase&, at::native::BinaryFunctor > const&)::{lambda(int)#1})", - "count": 668, - "device_time_us": 3093.801999999436, - "self_device_time_us": 3093.801999999436, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::convolution", - "count": 25, - "device_time_us": 3044.873999999925, - "self_device_time_us": 0.0, - "cpu_time_us": 2021.7800000000425, - "self_cpu_time_us": 143.51700000008714 - }, - { - "op": "aten::_convolution", - "count": 25, - "device_time_us": 3044.873999999925, - "self_device_time_us": 0.0, - "cpu_time_us": 1878.2629999999554, - "self_cpu_time_us": 489.0389999998333 - }, - { - "op": "void batch_trsm_left_kernel(cublasTrsmBatchParams2, float, float const*, int)", - "count": 48, - "device_time_us": 2853.693999999952, - "self_device_time_us": 2853.693999999952, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "ampere_bf16_s1688gemm_bf16_64x64_sliced1x4_ldg8_f2f_tn", - "count": 32, - "device_time_us": 2847.620000000141, - "self_device_time_us": 2847.620000000141, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::clone", - "count": 141, - "device_time_us": 2841.203000000127, - "self_device_time_us": 0.0, - "cpu_time_us": 4514.54400000007, - "self_cpu_time_us": 654.2669999996788 - }, - { - "op": "ampere_sgemm_128x128_tn", - "count": 240, - "device_time_us": 2783.6780000002327, - "self_device_time_us": 2783.6780000002327, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "void at::native::unrolled_elementwise_kernel, std::array, 4, TrivialOffsetCalculator<2, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<2>, at::native::memory::StoreWithCast<1> >(int, at::native::CUDAFunctor_add, std::array, TrivialOffsetCalculator<2, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<2>, at::native::memory::StoreWithCast<1>)", - "count": 178, - "device_time_us": 2525.99999999996, - "self_device_time_us": 2525.99999999996, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "ampere_sgemm_128x128_nn", - "count": 192, - "device_time_us": 2428.904999999824, - "self_device_time_us": 2428.904999999824, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "ampere_sgemm_128x128_nt", - "count": 192, - "device_time_us": 2347.072999999873, - "self_device_time_us": 2347.072999999873, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase&, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#12}::operator()() const::{lambda(c10::BFloat16)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast(at::TensorIteratorBase&, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#12}::operator()() const::{lambda(c10::BFloat16)#1} const&)::{lambda(int)#1})", - "count": 112, - "device_time_us": 2314.617000000122, - "self_device_time_us": 2314.617000000122, - "cpu_time_us": 0, - "self_cpu_time_us": 0 - }, - { - "op": "aten::conv1d", - "count": 24, - "device_time_us": 2197.6299999999246, - "self_device_time_us": 0.0, - "cpu_time_us": 1939.474999999984, - "self_cpu_time_us": 93.11899999994057 - }, - { - "op": "aten::scaled_dot_product_attention", - "count": 32, - "device_time_us": 2140.842999999986, - "self_device_time_us": 0.0, - "cpu_time_us": 2570.106999999978, - "self_cpu_time_us": 379.762999999959 - }, - { - "op": "aten::_scaled_dot_product_flash_attention", - "count": 32, - "device_time_us": 2140.842999999986, - "self_device_time_us": 0.0, - "cpu_time_us": 2190.344000000019, - "self_cpu_time_us": 310.61200000015015 - }, - { - "op": "aten::_flash_attention_forward", - "count": 32, - "device_time_us": 2140.842999999986, - "self_device_time_us": 2140.842999999986, - "cpu_time_us": 1620.5300000000316, - "self_cpu_time_us": 539.4930000000277 - } - ], - "peak_allocated_bytes": 9496202240 -} \ No newline at end of file diff --git a/recipe/cua_s1/experiments/rtx4090/candidate.json b/recipe/cua_s1/experiments/rtx4090/candidate.json deleted file mode 100644 index f8e4a92..0000000 --- a/recipe/cua_s1/experiments/rtx4090/candidate.json +++ /dev/null @@ -1,536 +0,0 @@ -{ - "environment": { - "python": "3.12.13", - "gpu": "NVIDIA GeForce RTX 4090", - "compute_capability": [ - 8, - 9 - ], - "cuda": "13.0", - "driver": "595.71.05", - "torch_num_threads": 64, - "torch_num_interop_threads": 64, - "packages": { - "torch": "2.14.0", - "torchvision": "0.29.0", - "transformers": "5.17.0", - "peft": "0.21.0", - "pillow": "11.3.0", - "safetensors": "0.8.0", - "triton": "3.8.0" - }, - "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", - "dtype": "bfloat16", - "adapter_merged": false - }, - "mode": "candidate", - "cases": [ - { - "name": "small", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" - }, - "attention_mask": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" - }, - "pixel_values": { - "shape": [ - 320, - 1536 - ], - "dtype": "torch.float32", - "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" - } - }, - "probabilities": [ - 0.9958575367927551, - 0.0005507932510226965, - 0.003591623157262802 - ], - "max_abs_difference": 0.0, - "latency_ms": 127.26952508091927 - }, - { - "name": "medium", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" - }, - "attention_mask": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.9958575367927551, - 0.0005507932510226965, - 0.003591623157262802 - ], - "max_abs_difference": 0.0, - "latency_ms": 126.0456619784236 - }, - { - "name": "jpeg", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" - }, - "attention_mask": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "5a78a7f25b40888ad5c134d02826d691c27707384603242b99897cc623817bd8" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.9936226010322571, - 0.0011634123511612415, - 0.005214052740484476 - ], - "max_abs_difference": 0.0, - "latency_ms": 121.78163509815931 - }, - { - "name": "single", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 431 - ], - "dtype": "torch.int64", - "sha256": "c4d7b1102e94984504efa938052a0bba3639ba85f9917de3e8af5426ceefd16b" - }, - "attention_mask": { - "shape": [ - 1, - 431 - ], - "dtype": "torch.int64", - "sha256": "3419eb840bd7f9d9e4b1fa6aaea783bae7a4752068701794687337702c2d8968" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 431 - ], - "dtype": "torch.int64", - "sha256": "077d91126e88fd3e8ba97a2d8079c28a4d5da388e56c89ca0e584fd5ad4b3641" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 1.0 - ], - "max_abs_difference": 0.0, - "latency_ms": 117.80334264039993 - }, - { - "name": "26-options", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 752 - ], - "dtype": "torch.int64", - "sha256": "726f7d304140077e32490233aba734494431d93b7e2885bcb72d8b1c730e022b" - }, - "attention_mask": { - "shape": [ - 1, - 752 - ], - "dtype": "torch.int64", - "sha256": "389336aff30134c9af5da75eb0a9694eef903cde261cd8741f8aafae5ff94dc7" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 752 - ], - "dtype": "torch.int64", - "sha256": "e4cb23e2c28d9f1909d0269d3b145ba19f7efb4176b5832347ba58fed9cfa084" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.07546718418598175, - 0.00034948240499943495, - 0.263406366109848, - 0.557631254196167, - 0.009013270027935505, - 0.01908109337091446, - 0.007019542157649994, - 0.016839005053043365, - 0.009013270027935505, - 0.0025823453906923532, - 0.004824455827474594, - 0.003315796609967947, - 0.002011132426559925, - 0.0022789116483181715, - 0.00619472423568368, - 0.0010764816543087363, - 0.0037572900764644146, - 0.0006529190577566624, - 0.002011132426559925, - 0.0012198134791105986, - 0.00044874430750496686, - 0.0005084939184598625, - 0.003315796609967947, - 0.0013822298496961594, - 0.0017748181708157063, - 0.004824455827474594 - ], - "max_abs_difference": 0.0, - "latency_ms": 141.8488211929798 - }, - { - "name": "structured", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 469 - ], - "dtype": "torch.int64", - "sha256": "51435b8464d5ef7dde621ff5dd7679036480601f39496e693687a2b81d178553" - }, - "attention_mask": { - "shape": [ - 1, - 469 - ], - "dtype": "torch.int64", - "sha256": "75a539e8d5305e6bc6b6d4f53e3c5e06be4f04fcaa8bdb5995794b7cf09fe80f" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 469 - ], - "dtype": "torch.int64", - "sha256": "269da87e532542e6e885dfd9ac112e706ee7f033044955ba5020f6b844c86a0c" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.07443060725927353, - 0.9067503809928894, - 0.01881900243461132 - ], - "max_abs_difference": 0.0, - "latency_ms": 121.30219116806984 - }, - { - "name": "special-token", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 441 - ], - "dtype": "torch.int64", - "sha256": "6abf0a9290f3d2a19d5063d24e5c831eae8697c604e7206e7fcf83c4f1b013fe" - }, - "attention_mask": { - "shape": [ - 1, - 441 - ], - "dtype": "torch.int64", - "sha256": "07d34f72fa2eb8b18b38092dfb3bb5364429d5008739c6290d61fd68d500a2b0" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 441 - ], - "dtype": "torch.int64", - "sha256": "7ffc7201ce92882bf791a82175d36c9f7910e5bdba09d335cac9b8d6640b6caa" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.531209409236908, - 0.4687906503677368 - ], - "max_abs_difference": 0.0, - "latency_ms": 117.51110758632421 - }, - { - "name": "two-questions", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" - }, - "attention_mask": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" - }, - "pixel_values": { - "shape": [ - 320, - 1536 - ], - "dtype": "torch.float32", - "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" - } - }, - "probabilities": [ - 0.9958575367927551, - 0.0005507932510226965, - 0.003591623157262802 - ], - "max_abs_difference": 0.0, - "latency_ms": 103.49029209464788 - }, - { - "name": "two-questions", - "question": "second", - "inputs": { - "input_ids": { - "shape": [ - 1, - 215 - ], - "dtype": "torch.int64", - "sha256": "1bc665c3f5aa0134e368d1a9edf0e652ec604cf43e969d59b14f696a4cf5c9c3" - }, - "attention_mask": { - "shape": [ - 1, - 215 - ], - "dtype": "torch.int64", - "sha256": "bbdd8ad88f622ea9585a6b471f3ef6810a0c3374e43dbcb747e3077de4e6e674" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 215 - ], - "dtype": "torch.int64", - "sha256": "6dc930bb14d93a028e6d2db398b391c4fe5af99184ec6625d38d2f7820e182f3" - }, - "pixel_values": { - "shape": [ - 320, - 1536 - ], - "dtype": "torch.float32", - "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" - } - }, - "probabilities": [ - 0.22270014882087708, - 0.7772998809814453 - ], - "max_abs_difference": 0.0, - "latency_ms": 104.30925991386175 - } - ], - "reference_source": { - "path": "libs/cua-s1/python/src/cua_s1/four_b.py", - "sha256": "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7", - "revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" - }, - "load_ms": 19939.03631903231, - "warmup_ms": 1815.6465897336602, - "adapter_modules": 178, - "peak_allocated_bytes": 9657058304 -} \ No newline at end of file diff --git a/recipe/cua_s1/experiments/rtx4090/frontend.json b/recipe/cua_s1/experiments/rtx4090/frontend.json deleted file mode 100644 index 39cee2e..0000000 --- a/recipe/cua_s1/experiments/rtx4090/frontend.json +++ /dev/null @@ -1,79 +0,0 @@ -[ - { - "name": "health", - "status": 200, - "content_type": "application/json", - "body_sha256": "f54feed005f2b4d1d6f2398a306d13aeaaa0c90aea20b5cca2a4de4ecab5bdd7", - "identical": true - }, - { - "name": "26-options", - "status": 200, - "content_type": "application/json", - "body_sha256": "64889fe4b0e85dac1c361d29a4367ed27fe63ceb75d94166ee867ba11ceb83d3", - "identical": true - }, - { - "name": "jpeg", - "status": 200, - "content_type": "application/json", - "body_sha256": "aeedac2faea6c8d9cc6e8636751833f5ee03b88dc8092f8a9d93e4320eb3f887", - "identical": true - }, - { - "name": "medium", - "status": 200, - "content_type": "application/json", - "body_sha256": "1d6041c8144e070b2d766cc1c85c52764addcb4745da88e96f94c5058678e62c", - "identical": true - }, - { - "name": "single", - "status": 200, - "content_type": "application/json", - "body_sha256": "6e9194fc25bc7ec8741dabd52ee4dfb3412b5c18eac22c2935b3ba96989c7a8c", - "identical": true - }, - { - "name": "small", - "status": 200, - "content_type": "application/json", - "body_sha256": "51fa5f1cf910140c958cc2c6ca3738699527dd47cee58809060ac6017400602e", - "identical": true - }, - { - "name": "special-token", - "status": 200, - "content_type": "application/json", - "body_sha256": "38646659005cdc82b81a5bac4ab8120f7f1e1b16bdb1c610fa72e5794ed5a439", - "identical": true - }, - { - "name": "structured", - "status": 200, - "content_type": "application/json", - "body_sha256": "21fe1ab16324d937eb8f9c6c692ccb1d1ad41d9d01cb078440dee0fde8990920", - "identical": true - }, - { - "name": "two-questions", - "status": 200, - "content_type": "application/json", - "body_sha256": "3f8dd307c92bf5dbf7e1ff1c928d3961265516e587ce162f905ded363f6764c1", - "identical": true - }, - { - "name": "invalid", - "status": 422, - "content_type": "application/json", - "body_sha256": "866753c516b9090204f79318fa8d289f42ef1650821ef4b974bc915ec2836832", - "identical": true - }, - { - "name": "duplicate", - "status": 422, - "content_type": "application/json", - "body_sha256": "1175f6c373a386204fa01db801fac92f0b772d5f2ef09a43e788ac72ae156a27", - "identical": true - } -] \ No newline at end of file diff --git a/recipe/cua_s1/experiments/rtx4090/packages.txt b/recipe/cua_s1/experiments/rtx4090/packages.txt deleted file mode 100644 index 646c8bb..0000000 --- a/recipe/cua_s1/experiments/rtx4090/packages.txt +++ /dev/null @@ -1,62 +0,0 @@ -accelerate==1.15.0 -annotated-doc==0.0.5 -anyio==4.15.1 -certifi==2026.7.22 -click==8.5.0 -cuda-bindings==13.4.2 -cuda-pathfinder==1.8.2 -cuda-toolkit==13.0.3.0 -filelock==4.0.1 -fsspec==2026.9.0 -h11==0.16.0 -hf-xet==1.6.0 -httpcore==1.0.9 -httpx==0.28.1 -huggingface-hub==1.32.0 -idna==3.20 -iniconfig==2.3.0 -jinja2==3.1.6 -markdown-it-py==4.2.0 -markupsafe==3.0.3 -mdurl==0.1.2 -mpmath==1.3.0 -networkx==3.6.1 -numpy==2.5.3 -nvidia-cublas==13.1.1.3 -nvidia-cuda-cupti==13.0.85 -nvidia-cuda-nvrtc==13.0.88 -nvidia-cuda-runtime==13.0.96 -nvidia-cudnn-cu13==9.24.0.43 -nvidia-cufft==12.0.0.61 -nvidia-cufile==1.15.1.6 -nvidia-curand==10.4.0.35 -nvidia-cusolver==12.0.4.66 -nvidia-cusparse==12.6.3.3 -nvidia-cusparselt-cu13==0.8.1 -nvidia-nccl-cu13==2.30.7 -nvidia-nvjitlink==13.4.92 -nvidia-nvshmem-cu13==3.4.5 -nvidia-nvtx==13.0.85 -packaging==26.3 -peft==0.21.0 -pillow==11.3.0 -pluggy==1.6.0 -psutil==7.2.2 -pygments==2.21.0 -pytest==9.1.1 -pyyaml==6.0.3 -regex==2026.9.10 -rich==15.0.0 -ruff==0.16.8 -safetensors==0.8.0 -setuptools==84.0.0 -shellingham==1.5.4 -sympy==1.14.0 -tokenizers==0.23.2 -torch==2.14.0 -torchvision==0.29.0 -tqdm==4.70.1 -transformers==5.17.0 -triton==3.8.0 -typer==0.27.2 -typing-extensions==4.16.0 diff --git a/recipe/cua_s1/experiments/rtx4090/reference.json b/recipe/cua_s1/experiments/rtx4090/reference.json deleted file mode 100644 index eaba46f..0000000 --- a/recipe/cua_s1/experiments/rtx4090/reference.json +++ /dev/null @@ -1,527 +0,0 @@ -{ - "environment": { - "python": "3.12.13", - "gpu": "NVIDIA GeForce RTX 4090", - "compute_capability": [ - 8, - 9 - ], - "cuda": "13.0", - "driver": "595.71.05", - "torch_num_threads": 64, - "torch_num_interop_threads": 64, - "packages": { - "torch": "2.14.0", - "torchvision": "0.29.0", - "transformers": "5.17.0", - "peft": "0.21.0", - "pillow": "11.3.0", - "safetensors": "0.8.0", - "triton": "3.8.0" - }, - "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", - "dtype": "bfloat16", - "adapter_merged": false - }, - "mode": "reference", - "cases": [ - { - "name": "small", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" - }, - "attention_mask": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" - }, - "pixel_values": { - "shape": [ - 320, - 1536 - ], - "dtype": "torch.float32", - "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" - } - }, - "probabilities": [ - 0.9958575367927551, - 0.0005507932510226965, - 0.003591623157262802 - ], - "latency_ms": 147.01103698462248 - }, - { - "name": "medium", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" - }, - "attention_mask": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.9958575367927551, - 0.0005507932510226965, - 0.003591623157262802 - ], - "latency_ms": 149.6963743120432 - }, - { - "name": "jpeg", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "1e44867ea2f54730abf507d49e4db88735868bfc133cf98d4cb5c49fc001adad" - }, - "attention_mask": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "c7e61368825c7c260dfd74e531d207b536301fb298dbd9b8cf341fc9ba9d42ff" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 456 - ], - "dtype": "torch.int64", - "sha256": "90259c2d1bd52b91ab67913e12e53db66a07bfc19ffefa42978366f637e9d0ef" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "5a78a7f25b40888ad5c134d02826d691c27707384603242b99897cc623817bd8" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.9936226010322571, - 0.0011634123511612415, - 0.005214052740484476 - ], - "latency_ms": 137.3040061444044 - }, - { - "name": "single", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 431 - ], - "dtype": "torch.int64", - "sha256": "c4d7b1102e94984504efa938052a0bba3639ba85f9917de3e8af5426ceefd16b" - }, - "attention_mask": { - "shape": [ - 1, - 431 - ], - "dtype": "torch.int64", - "sha256": "3419eb840bd7f9d9e4b1fa6aaea783bae7a4752068701794687337702c2d8968" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 431 - ], - "dtype": "torch.int64", - "sha256": "077d91126e88fd3e8ba97a2d8079c28a4d5da388e56c89ca0e584fd5ad4b3641" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 1.0 - ], - "latency_ms": 135.01203525811434 - }, - { - "name": "26-options", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 752 - ], - "dtype": "torch.int64", - "sha256": "726f7d304140077e32490233aba734494431d93b7e2885bcb72d8b1c730e022b" - }, - "attention_mask": { - "shape": [ - 1, - 752 - ], - "dtype": "torch.int64", - "sha256": "389336aff30134c9af5da75eb0a9694eef903cde261cd8741f8aafae5ff94dc7" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 752 - ], - "dtype": "torch.int64", - "sha256": "e4cb23e2c28d9f1909d0269d3b145ba19f7efb4176b5832347ba58fed9cfa084" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.07546718418598175, - 0.00034948240499943495, - 0.263406366109848, - 0.557631254196167, - 0.009013270027935505, - 0.01908109337091446, - 0.007019542157649994, - 0.016839005053043365, - 0.009013270027935505, - 0.0025823453906923532, - 0.004824455827474594, - 0.003315796609967947, - 0.002011132426559925, - 0.0022789116483181715, - 0.00619472423568368, - 0.0010764816543087363, - 0.0037572900764644146, - 0.0006529190577566624, - 0.002011132426559925, - 0.0012198134791105986, - 0.00044874430750496686, - 0.0005084939184598625, - 0.003315796609967947, - 0.0013822298496961594, - 0.0017748181708157063, - 0.004824455827474594 - ], - "latency_ms": 314.77690767496824 - }, - { - "name": "structured", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 469 - ], - "dtype": "torch.int64", - "sha256": "51435b8464d5ef7dde621ff5dd7679036480601f39496e693687a2b81d178553" - }, - "attention_mask": { - "shape": [ - 1, - 469 - ], - "dtype": "torch.int64", - "sha256": "75a539e8d5305e6bc6b6d4f53e3c5e06be4f04fcaa8bdb5995794b7cf09fe80f" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 469 - ], - "dtype": "torch.int64", - "sha256": "269da87e532542e6e885dfd9ac112e706ee7f033044955ba5020f6b844c86a0c" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.07443060725927353, - 0.9067503809928894, - 0.01881900243461132 - ], - "latency_ms": 141.84184558689594 - }, - { - "name": "special-token", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 441 - ], - "dtype": "torch.int64", - "sha256": "6abf0a9290f3d2a19d5063d24e5c831eae8697c604e7206e7fcf83c4f1b013fe" - }, - "attention_mask": { - "shape": [ - 1, - 441 - ], - "dtype": "torch.int64", - "sha256": "07d34f72fa2eb8b18b38092dfb3bb5364429d5008739c6290d61fd68d500a2b0" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 441 - ], - "dtype": "torch.int64", - "sha256": "7ffc7201ce92882bf791a82175d36c9f7910e5bdba09d335cac9b8d6640b6caa" - }, - "pixel_values": { - "shape": [ - 1200, - 1536 - ], - "dtype": "torch.float32", - "sha256": "92648e27fcd15cac68708dbe53974f3cd64c4f5bcc32d17944d7cead68aa469c" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e" - } - }, - "probabilities": [ - 0.531209409236908, - 0.4687906503677368 - ], - "latency_ms": 140.13662841171026 - }, - { - "name": "two-questions", - "question": "next", - "inputs": { - "input_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21309638ff0c5ca8b7d1bf4b61905de3f31a216667ebc20fec9cdfaacdf08d48" - }, - "attention_mask": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "1b489e733e23e18297cee9103f1b583499b76d29ab36a61129ac4fabc6dbd27a" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 236 - ], - "dtype": "torch.int64", - "sha256": "21b9904b99d4c8ade165eae791ba8439afbbe6ee838c32df37f14031ab51bb3d" - }, - "pixel_values": { - "shape": [ - 320, - 1536 - ], - "dtype": "torch.float32", - "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" - } - }, - "probabilities": [ - 0.9958575367927551, - 0.0005507932510226965, - 0.003591623157262802 - ], - "latency_ms": 114.18061796575785 - }, - { - "name": "two-questions", - "question": "second", - "inputs": { - "input_ids": { - "shape": [ - 1, - 215 - ], - "dtype": "torch.int64", - "sha256": "1bc665c3f5aa0134e368d1a9edf0e652ec604cf43e969d59b14f696a4cf5c9c3" - }, - "attention_mask": { - "shape": [ - 1, - 215 - ], - "dtype": "torch.int64", - "sha256": "bbdd8ad88f622ea9585a6b471f3ef6810a0c3374e43dbcb747e3077de4e6e674" - }, - "mm_token_type_ids": { - "shape": [ - 1, - 215 - ], - "dtype": "torch.int64", - "sha256": "6dc930bb14d93a028e6d2db398b391c4fe5af99184ec6625d38d2f7820e182f3" - }, - "pixel_values": { - "shape": [ - 320, - 1536 - ], - "dtype": "torch.float32", - "sha256": "8fd8d3e52f9c4d9fcc14ae0467e7d0642ce620e733705e6a77edd1f4e7940776" - }, - "image_grid_thw": { - "shape": [ - 1, - 3 - ], - "dtype": "torch.int64", - "sha256": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329" - } - }, - "probabilities": [ - 0.22270014882087708, - 0.7772998809814453 - ], - "latency_ms": 117.08385031670332 - } - ], - "reference_source": { - "path": "libs/cua-s1/python/src/cua_s1/four_b.py", - "sha256": "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7", - "revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f" - }, - "artifact_verification_ms": 9541.830752044916, - "load_ms": 16820.161445997655, - "warmup_ms": 3665.197447873652, - "peak_allocated_bytes": 9655206912 -} \ No newline at end of file From 9093fd820e737a185aa72f2123d0f6e28b6f06ae Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Wed, 30 Sep 2026 00:35:46 +0800 Subject: [PATCH 3/4] Move Cua-S1 HTTP serving out of models and reuse upstream weight manifest --- .github/workflows/cua-s1-multimodal.yml | 6 +- README.md | 2 +- recipe/cua_s1/README.md | 14 ++- recipe/cua_s1/download_weights.py | 20 ++-- .../server.py => frontend/cua_s1.py} | 4 +- .../cua_s1/multimodal/THIRD_PARTY_NOTICES.md | 21 ---- src/models/cua_s1/multimodal/model.py | 12 ++- src/models/cua_s1/multimodal/protocol.py | 22 +++- .../cua_s1/multimodal/weights.lock.json | 102 ------------------ tests/cua_s1/test_download_weights.py | 83 ++++++++++++++ tests/cua_s1/test_model.py | 14 ++- tests/cua_s1/test_server.py | 2 +- 12 files changed, 160 insertions(+), 142 deletions(-) rename src/{models/cua_s1/multimodal/server.py => frontend/cua_s1.py} (97%) delete mode 100644 src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md delete mode 100644 src/models/cua_s1/multimodal/weights.lock.json create mode 100644 tests/cua_s1/test_download_weights.py diff --git a/.github/workflows/cua-s1-multimodal.yml b/.github/workflows/cua-s1-multimodal.yml index 9329244..9d4fd7b 100644 --- a/.github/workflows/cua-s1-multimodal.yml +++ b/.github/workflows/cua-s1-multimodal.yml @@ -3,6 +3,7 @@ on: pull_request: paths: - 'src/models/cua_s1/multimodal/**' + - 'src/frontend/cua_s1.py' - 'tests/cua_s1/**' - 'recipe/cua_s1/**' - '.github/workflows/cua-s1-multimodal.yml' @@ -10,6 +11,7 @@ on: branches: [main] paths: - 'src/models/cua_s1/multimodal/**' + - 'src/frontend/cua_s1.py' - 'tests/cua_s1/**' - 'recipe/cua_s1/**' - '.github/workflows/cua-s1-multimodal.yml' @@ -26,5 +28,5 @@ jobs: python-version: '3.12' - run: python -m pip install Pillow==11.3.0 pytest==9.1.1 ruff==0.16.8 - run: PYTHONPATH=src python -m pytest tests/cua_s1 -q - - run: ruff check --isolated --select E4,E7,E9,F,I src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 - - run: ruff format --isolated --check src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 + - run: ruff check --isolated --select E4,E7,E9,F,I src/frontend/cua_s1.py src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 + - run: ruff format --isolated --check src/frontend/cua_s1.py src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 diff --git a/README.md b/README.md index 2cca0d9..17d41db 100644 --- a/README.md +++ b/README.md @@ -41,7 +41,7 @@ Implementation code lives under `src/`; recipes and documentation stay at the re | Directory | Responsibility | | --- | --- | -| [`src/frontend/`](src/frontend/) | Rust serving code and the small engine interface. | +| [`src/frontend/`](src/frontend/) | Rust serving code, Python worker adapters, and the small engine interface. | | [`src/models/`](src/models/) | Model implementations, one directory per model: preprocessing, batching, state, execution, and output processing. | | [`src/backends/cuda/`](src/backends/cuda/) | NVIDIA GPU operations and kernel integration. | | [`src/backends/metal/`](src/backends/metal/) | Apple GPU operations and kernel integration. | diff --git a/recipe/cua_s1/README.md b/recipe/cua_s1/README.md index 741a29e..753eef7 100644 --- a/recipe/cua_s1/README.md +++ b/recipe/cua_s1/README.md @@ -22,17 +22,23 @@ python3.12 -m venv .venv . .venv/bin/activate pip install -r recipe/cua_s1/requirements-multimodal.txt PYTHONPATH=src python recipe/cua_s1/download_weights.py --dest weights -PYTHONPATH=src python -m models.cua_s1.multimodal.server \ +PYTHONPATH=src python -m frontend.cua_s1 \ --base weights/Qwen3.5-4B \ --adapter weights/cua-s1-4b-0.2/multimodal ``` -`weights.lock.json` pins and checks every loaded artifact's size and SHA-256. +The downloader fetches the upstream manifest at the fixed reference revision, +checks its pinned SHA-256 and saves it as `weights/weights.lock.json`. The worker +reads this local manifest and checks every loaded artifact's size and SHA-256. +Keep the manifest next to the base checkpoint directory when moving weights. Extra files are rejected, except Hugging Face's `.cache` metadata, so another checkpoint cannot silently override verified shards. Downloads require roughly 9 GB plus cache/install space. Loading is offline after the download completes. Weights are not included in this repository. +The HTTP adapter lives in `src/frontend/cua_s1.py`; model execution stays in +`src/models/cua_s1/multimodal/`. + The worker binds to `127.0.0.1:8000` only after loading and a successful warmup. `GET /health` returns `{"status":"ready","modality":"multimodal"}`. One request runs at a time; concurrent requests return `503`. This is a loopback model worker, @@ -138,8 +144,8 @@ CPU-only validation: ```sh pip install Pillow==11.3.0 pytest==9.1.1 ruff==0.16.8 PYTHONPATH=src python -m pytest tests/cua_s1 -q -ruff check --select E4,E7,E9,F,I src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 -ruff format --check src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 +ruff check --select E4,E7,E9,F,I src/frontend/cua_s1.py src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 +ruff format --check src/frontend/cua_s1.py src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 ``` Metal, native CUDA kernels, text-adapter serving, batching, caching and training diff --git a/recipe/cua_s1/download_weights.py b/recipe/cua_s1/download_weights.py index 104b393..f1474bb 100644 --- a/recipe/cua_s1/download_weights.py +++ b/recipe/cua_s1/download_weights.py @@ -1,23 +1,30 @@ """Download the upstream-pinned artifacts and verify their checksums.""" import argparse -import json from pathlib import Path +from urllib.request import urlopen from huggingface_hub import snapshot_download -from models.cua_s1.multimodal.model import verify_weights +from models.cua_s1.multimodal.model import ( + REFERENCE_REVISION, + parse_weights_manifest, + verify_weights, +) def main(): parser = argparse.ArgumentParser(description=__doc__) parser.add_argument("--dest", type=Path, required=True) args = parser.parse_args() - lock = ( - Path(__file__).resolve().parents[2] - / "src/models/cua_s1/multimodal/weights.lock.json" + url = ( + f"https://raw.githubusercontent.com/trycua/cua/{REFERENCE_REVISION}/" + "libs/cua-s1/ci/weights.lock.json" ) - for artifact in json.loads(lock.read_text())["artifacts"]: + with urlopen(url, timeout=30) as response: + raw = response.read() + manifest = parse_weights_manifest(raw) + for artifact in manifest["artifacts"]: snapshot_download( repo_id=artifact["repo_id"], revision=artifact["revision"], @@ -25,6 +32,7 @@ def main(): allow_patterns=list(artifact["files"]), token=False, ) + (args.dest / "weights.lock.json").write_bytes(raw) verify_weights(args.dest / "Qwen3.5-4B", args.dest / "cua-s1-4b-0.2/multimodal") print("Pinned base and multimodal adapter checksums verified.") diff --git a/src/models/cua_s1/multimodal/server.py b/src/frontend/cua_s1.py similarity index 97% rename from src/models/cua_s1/multimodal/server.py rename to src/frontend/cua_s1.py index c414bc9..bde882c 100644 --- a/src/models/cua_s1/multimodal/server.py +++ b/src/frontend/cua_s1.py @@ -9,7 +9,7 @@ import threading from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer -from .protocol import ( +from models.cua_s1.multimodal.protocol import ( MAX_BODY, InvalidRequest, MalformedJSON, @@ -107,7 +107,7 @@ def do_POST(self): def main(): - from .model import MultimodalEngine + from models.cua_s1.multimodal.model import MultimodalEngine p = argparse.ArgumentParser(description=__doc__) p.add_argument( diff --git a/src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md b/src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md deleted file mode 100644 index b8b198c..0000000 --- a/src/models/cua_s1/multimodal/THIRD_PARTY_NOTICES.md +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2025 Cua AI, Inc. - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/src/models/cua_s1/multimodal/model.py b/src/models/cua_s1/multimodal/model.py index db55070..e3655de 100644 --- a/src/models/cua_s1/multimodal/model.py +++ b/src/models/cua_s1/multimodal/model.py @@ -12,6 +12,9 @@ BASE_REVISION = "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a" ADAPTER_REVISION = "16818868b0cc7813808aae4e87b417657046ab79" IDENTITY = f"cua-ai/cua-s1-4b-0.2@{ADAPTER_REVISION}:multimodal" +WEIGHTS_MANIFEST_SHA256 = ( + "9820bd232c5762f114e19680c0f8203d7e1faaf8a60c196cfe01964d6d8a6c09" +) MAX_TOKENS = 4096 @@ -47,9 +50,16 @@ def validate_adapter_config(config: dict): raise ValueError("expected the pinned 0.2 multimodal LoRA adapter") +def parse_weights_manifest(raw: bytes) -> dict: + """Accept only the manifest from the pinned upstream reference commit.""" + if hashlib.sha256(raw).hexdigest() != WEIGHTS_MANIFEST_SHA256: + raise ValueError("upstream weights manifest checksum mismatch") + return json.loads(raw) + + def verify_weights(base: Path, adapter: Path): """Check local artifacts before assigning the pinned identity to responses.""" - lock = json.loads(Path(__file__).with_name("weights.lock.json").read_text()) + lock = parse_weights_manifest((base.parent / "weights.lock.json").read_bytes()) allowed = {base: set(), adapter: set()} for artifact in lock["artifacts"]: for name, expected in artifact["files"].items(): diff --git a/src/models/cua_s1/multimodal/protocol.py b/src/models/cua_s1/multimodal/protocol.py index edd207d..4346257 100644 --- a/src/models/cua_s1/multimodal/protocol.py +++ b/src/models/cua_s1/multimodal/protocol.py @@ -21,7 +21,27 @@ MAX_TEXT = 16384 # Prompt and letter layout follow trycua/cua at 0e75660ce4c2edda519e0c795fa3ad98abf4e76f. -# See THIRD_PARTY_NOTICES.md for the upstream MIT notice. +# MIT License +# +# Copyright (c) 2025 Cua AI, Inc. +# +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. SYSTEM_PROMPT = ( "You are a one-pass computer-use decision model. You are shown the " "current state of a screen and a fixed, closed list of candidate " diff --git a/src/models/cua_s1/multimodal/weights.lock.json b/src/models/cua_s1/multimodal/weights.lock.json deleted file mode 100644 index dfb9748..0000000 --- a/src/models/cua_s1/multimodal/weights.lock.json +++ /dev/null @@ -1,102 +0,0 @@ -{ - "schema": "cua-s1/weights-lock/v1", - "description": "Pinned public Hugging Face artifacts for the weights-backed Cua-S1-4B smoke. Every listed file is downloaded at the pinned revision and verified by size and SHA-256; keep in sync with libs/cua-s1/README.md.", - "artifacts": [ - { - "name": "Qwen3.5-4B", - "repo_id": "Qwen/Qwen3.5-4B", - "revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "role": "base", - "files": { - ".gitattributes": { - "size": 1570, - "sha256": "34448b82c17d60fec9b65b1f093c115ddbaadc04beb1b0140b6bfed2e012a930" - }, - "LICENSE": { - "size": 11544, - "sha256": "bbedc3fda3305820b977265f01b8619d87570a6739de3a5582c3464840f1e57a" - }, - "README.md": { - "size": 77661, - "sha256": "1406be1b6b8fd8a6545870da516912804756593628a1d0fb0a7965211e82a7bb" - }, - "chat_template.jinja": { - "size": 7756, - "sha256": "a4aee8afcf2e0711942cf848899be66016f8d14a889ff9ede07bca099c28f715" - }, - "config.json": { - "size": 3161, - "sha256": "ddc63e1c717afa86c865bb5e01313d89d72bb53b97ad4a8a03ba8510c0621670" - }, - "merges.txt": { - "size": 3353259, - "sha256": "a9d356d7bdf1ef4949e3e748e95b8e10ad9d4e2e838eddc38a0a7b6b94d1db8d" - }, - "model.safetensors-00001-of-00002.safetensors": { - "size": 5329398688, - "sha256": "26a93f066e1916adb13453dae5a0c707c0fbc71299ed98779571a907b8e74c61" - }, - "model.safetensors-00002-of-00002.safetensors": { - "size": 3990429408, - "sha256": "cb544bd9bfae93dc59b0f22b292f5933573854a7f9b97835c67060d7d910e188" - }, - "model.safetensors.index.json": { - "size": 76196, - "sha256": "cf3f798ee02ba45f9622aa8892a47369ab667d0afbf154ee7c2212de42e6302d" - }, - "preprocessor_config.json": { - "size": 390, - "sha256": "27225450ac9c6529872ee1924fcb0962ff5634834f817040f444118116f4e516" - }, - "tokenizer.json": { - "size": 12807982, - "sha256": "5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42" - }, - "tokenizer_config.json": { - "size": 16710, - "sha256": "316230d6a809701f4db5ea8f8fc862bc3a6f3229c937c174e674ff3ca0a64ac8" - }, - "video_preprocessor_config.json": { - "size": 385, - "sha256": "7768af27c1fafa9cc9011c1dc20067e03f8915e03b63504550e11d5066986d13" - }, - "vocab.json": { - "size": 6722759, - "sha256": "ce99b4cb2983d118806ce0a8b777a35b093e2000a503ebde25853284c9dfa003" - } - } - }, - { - "name": "cua-s1-4b-0.2", - "repo_id": "cua-ai/cua-s1-4b-0.2", - "revision": "16818868b0cc7813808aae4e87b417657046ab79", - "role": "adapter", - "files": { - ".gitattributes": { - "size": 1519, - "sha256": "11ad7efa24975ee4b0c3c3a38ed18737f0658a5f75a0a96787b576a78a023361" - }, - "README.md": { - "size": 1134, - "sha256": "0e89cb4aea205e08c13f8af7fbfdfd0854ba0f6c57602071238771d0e1f8c7cd" - }, - "multimodal/adapter_config.json": { - "size": 1129, - "sha256": "f80edac43dd7605317ae8189613c75d13a029fa5d8a4870c04f1ba7097a8fab5" - }, - "multimodal/adapter_model.safetensors": { - "size": 101665112, - "sha256": "38ecd5a9191a436c1739db29104517f0aff4c70d0736d9cd810d3d49cfd652ea" - }, - "text/adapter_config.json": { - "size": 1091, - "sha256": "c246fce1fe1d44160ae5f246f9881dfeabb4fcd67fe4a777cef10a1dbcf16540" - }, - "text/adapter_model.safetensors": { - "size": 84968408, - "sha256": "9b59c5aed96171a50b26526613766bbf44347a5c7af70f81efe6bcc6e9dbfb0e" - } - } - } - ] -} diff --git a/tests/cua_s1/test_download_weights.py b/tests/cua_s1/test_download_weights.py new file mode 100644 index 0000000..5fb9939 --- /dev/null +++ b/tests/cua_s1/test_download_weights.py @@ -0,0 +1,83 @@ +"""Download setup uses the upstream manifest without checking a copy into source.""" + +import importlib.util +import io +import json +import sys +from pathlib import Path +from types import SimpleNamespace + +import pytest + + +@pytest.mark.parametrize("corrupt", [False, True]) +def test_manifest_is_verified_before_downloading_weights( + tmp_path, monkeypatch, corrupt +): + from models.cua_s1.multimodal import model + + raw = json.dumps( + { + "artifacts": [ + { + "repo_id": "test/base", + "revision": "fixed", + "name": "Qwen3.5-4B", + "files": {"config.json": {}}, + } + ] + } + ).encode() + import hashlib + + monkeypatch.setattr( + model, "WEIGHTS_MANIFEST_SHA256", hashlib.sha256(raw).hexdigest() + ) + downloads = [] + + def download(**kwargs): + downloads.append(kwargs) + kwargs["local_dir"].mkdir(parents=True) + + monkeypatch.setitem( + sys.modules, "huggingface_hub", SimpleNamespace(snapshot_download=download) + ) + path = Path(__file__).resolve().parents[2] / "recipe/cua_s1/download_weights.py" + spec = importlib.util.spec_from_file_location("download_weights_test", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + payload = raw + b" " if corrupt else raw + urls = [] + + def fetch(url, timeout): + urls.append(url) + assert timeout == 30 + return io.BytesIO(payload) + + monkeypatch.setattr(module, "urlopen", fetch) + verified = [] + + def verify(base, adapter): + assert (base.parent / "weights.lock.json").read_bytes() == raw + verified.append((base, adapter)) + + monkeypatch.setattr(module, "verify_weights", verify) + dest = tmp_path / "weights" + monkeypatch.setattr(sys, "argv", [str(path), "--dest", str(dest)]) + if corrupt: + with pytest.raises(ValueError, match="manifest checksum"): + module.main() + assert not downloads and not verified and not dest.exists() + else: + module.main() + assert downloads == [ + { + "repo_id": "test/base", + "revision": "fixed", + "local_dir": dest / "Qwen3.5-4B", + "allow_patterns": ["config.json"], + "token": False, + } + ] + assert verified == [(dest / "Qwen3.5-4B", dest / "cua-s1-4b-0.2/multimodal")] + assert model.REFERENCE_REVISION in urls[0] diff --git a/tests/cua_s1/test_model.py b/tests/cua_s1/test_model.py index 228e3ac..d37728f 100644 --- a/tests/cua_s1/test_model.py +++ b/tests/cua_s1/test_model.py @@ -84,7 +84,11 @@ def test_unlisted_model_files_cannot_override_verified_shards(tmp_path, monkeypa } ) ) - monkeypatch.setattr(model, "__file__", str(tmp_path / "model.py")) + monkeypatch.setattr( + model, + "WEIGHTS_MANIFEST_SHA256", + hashlib.sha256(manifest.read_bytes()).hexdigest(), + ) model.verify_weights(base, adapter) (base / "model.safetensors").write_bytes(b"override") with pytest.raises(ValueError, match="unlisted"): @@ -110,3 +114,11 @@ def prepare(image, question): with pytest.raises(InvalidRequest, match="4096"): engine.predict(Request(None, (first, second))) assert forwarded == [] + + +def test_weights_manifest_must_match_pinned_upstream_digest(tmp_path): + from models.cua_s1.multimodal.model import verify_weights + + (tmp_path / "weights.lock.json").write_text('{"artifacts": []}') + with pytest.raises(ValueError, match="manifest checksum"): + verify_weights(tmp_path / "base", tmp_path / "adapter") diff --git a/tests/cua_s1/test_server.py b/tests/cua_s1/test_server.py index 6238598..50bd2cb 100644 --- a/tests/cua_s1/test_server.py +++ b/tests/cua_s1/test_server.py @@ -6,8 +6,8 @@ import pytest from test_protocol import image_url, request +from frontend.cua_s1 import Handler, WorkerServer from models.cua_s1.multimodal.protocol import MAX_BODY -from models.cua_s1.multimodal.server import Handler, WorkerServer class Engine: From 934e1ad1dca8f2f447c935d267e01b9216ccf337 Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Wed, 30 Sep 2026 09:08:53 +0800 Subject: [PATCH 4/4] chore: limit PR to multimodal runtime code Remove supplemental tests, recipe tools, documentation, CI, and configuration changes from the PR diff. Runtime Python source and inline third-party notice are unchanged. Validation uses the pre-cleanup test/tool snapshot outside the checkout. --- .github/workflows/cua-s1-multimodal.yml | 32 --- .gitignore | 13 +- README.md | 12 +- recipe/cua_s1/README.md | 153 ---------- recipe/cua_s1/check_frontend.py | 75 ----- recipe/cua_s1/download_weights.py | 41 --- recipe/cua_s1/evaluate_multimodal.py | 322 ---------------------- recipe/cua_s1/make_example.py | 71 ----- recipe/cua_s1/requirements-multimodal.txt | 11 - src/models/cua_s1/README.md | 4 +- tests/cua_s1/test_download_weights.py | 83 ------ tests/cua_s1/test_evaluation.py | 65 ----- tests/cua_s1/test_model.py | 124 --------- tests/cua_s1/test_protocol.py | 172 ------------ tests/cua_s1/test_server.py | 204 -------------- 15 files changed, 9 insertions(+), 1373 deletions(-) delete mode 100644 .github/workflows/cua-s1-multimodal.yml delete mode 100644 recipe/cua_s1/README.md delete mode 100644 recipe/cua_s1/check_frontend.py delete mode 100644 recipe/cua_s1/download_weights.py delete mode 100644 recipe/cua_s1/evaluate_multimodal.py delete mode 100644 recipe/cua_s1/make_example.py delete mode 100644 recipe/cua_s1/requirements-multimodal.txt delete mode 100644 tests/cua_s1/test_download_weights.py delete mode 100644 tests/cua_s1/test_evaluation.py delete mode 100644 tests/cua_s1/test_model.py delete mode 100644 tests/cua_s1/test_protocol.py delete mode 100644 tests/cua_s1/test_server.py diff --git a/.github/workflows/cua-s1-multimodal.yml b/.github/workflows/cua-s1-multimodal.yml deleted file mode 100644 index 9d4fd7b..0000000 --- a/.github/workflows/cua-s1-multimodal.yml +++ /dev/null @@ -1,32 +0,0 @@ -name: Cua-S1 multimodal CPU checks -on: - pull_request: - paths: - - 'src/models/cua_s1/multimodal/**' - - 'src/frontend/cua_s1.py' - - 'tests/cua_s1/**' - - 'recipe/cua_s1/**' - - '.github/workflows/cua-s1-multimodal.yml' - push: - branches: [main] - paths: - - 'src/models/cua_s1/multimodal/**' - - 'src/frontend/cua_s1.py' - - 'tests/cua_s1/**' - - 'recipe/cua_s1/**' - - '.github/workflows/cua-s1-multimodal.yml' -permissions: - contents: read -jobs: - cpu: - runs-on: ubuntu-latest - timeout-minutes: 10 - steps: - - uses: actions/checkout@v4 - - uses: actions/setup-python@v5 - with: - python-version: '3.12' - - run: python -m pip install Pillow==11.3.0 pytest==9.1.1 ruff==0.16.8 - - run: PYTHONPATH=src python -m pytest tests/cua_s1 -q - - run: ruff check --isolated --select E4,E7,E9,F,I src/frontend/cua_s1.py src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 - - run: ruff format --isolated --check src/frontend/cua_s1.py src/models/cua_s1/multimodal tests/cua_s1 recipe/cua_s1 diff --git a/.gitignore b/.gitignore index b42bbc8..3857937 100644 --- a/.gitignore +++ b/.gitignore @@ -20,16 +20,7 @@ target # option (not recommended) you can uncomment the following to ignore the entire idea folder. #.idea/ -# Python workers and local model artifacts -__pycache__/ -.pytest_cache/ -.ruff_cache/ +# Local Python worker environment .venv/ -weights/ - -# macOS metadata +__pycache__/ .DS_Store - -# Local experiment outputs and assistant planning notes -/recipe/cua_s1/experiments/ -/docs/superpowers/ diff --git a/README.md b/README.md index 17d41db..76944aa 100644 --- a/README.md +++ b/README.md @@ -2,7 +2,7 @@ A community-maintained inference engine for prefill-only System1-Omni models, designed around a Rust frontend, model-owned execution, and high-performance CUDA and Metal backends. -The Rust frontend forwards requests to a separately running model worker. The Cua-S1 multimodal worker has been validated on CUDA through Transformers and PEFT; native GPU backends remain planned. +The Rust frontend forwards requests to a separately running model worker. In-repository model engines and GPU backends are not implemented yet. ## Run the frontend @@ -17,8 +17,7 @@ OMNI_JEV_BACKEND_URL=http://127.0.0.1:8000 \ Start the worker separately. See the [frontend documentation](src/frontend/README.md) for the HTTP interface and configuration, or the [Laya recipe](recipe/laya/README.md) -for a CPU text worker and response checks. The [Cua-S1 recipe](recipe/cua_s1/README.md) -provides a CUDA worker for screenshot-conditioned choices. +for a CPU text worker and response checks. ## Architecture @@ -41,23 +40,22 @@ Implementation code lives under `src/`; recipes and documentation stay at the re | Directory | Responsibility | | --- | --- | -| [`src/frontend/`](src/frontend/) | Rust serving code, Python worker adapters, and the small engine interface. | +| [`src/frontend/`](src/frontend/) | Rust serving code and the small engine interface. | | [`src/models/`](src/models/) | Model implementations, one directory per model: preprocessing, batching, state, execution, and output processing. | | [`src/backends/cuda/`](src/backends/cuda/) | NVIDIA GPU operations and kernel integration. | | [`src/backends/metal/`](src/backends/metal/) | Apple GPU operations and kernel integration. | | [`recipe/`](recipe/) | Model setup instructions, launch commands, configuration examples, and example requests. | | [`docs/`](docs/) | Project documentation and architecture assets. | -The frontend is a Cargo workspace member. Model and backend directories document ownership, with implementations added incrementally; they do not prescribe process boundaries. +The frontend is a Cargo workspace member. Model and backend directories currently document planned work; they do not prescribe process boundaries. ## Supported models -Validated coverage is listed by modality and execution path: +LAYA can run as an external Python worker for text requests. Its in-repository model engine is still planned: | Model | Status | | --- | --- | | LAYA | [External worker](recipe/laya/README.md); model engine planned | -| [Cua-S1 4B 0.2](recipe/cua_s1/README.md) | Multimodal screenshot choices via Transformers/PEFT on RTX 4090 CUDA; [parity and measurements](https://github.com/Levius-Fubuki/system1-omni/blob/683d470669f19d5e9530cd8233727a0d4f1f8d29/recipe/cua_s1/experiments/README.md). Text serving, native CUDA kernels and Metal deferred. | CUDA and Metal coverage will be documented per model as implementations are added and validated. diff --git a/recipe/cua_s1/README.md b/recipe/cua_s1/README.md deleted file mode 100644 index 753eef7..0000000 --- a/recipe/cua_s1/README.md +++ /dev/null @@ -1,153 +0,0 @@ -# Cua-S1 0.2 multimodal CUDA worker - -This recipe adds screenshot decisions using the multimodal adapter discussed in -[#10](https://github.com/ThinkFlowLab/system1-omni/issues/10). It loads Transformers -and PEFT directly. Upstream `FourBModel` is used only as an independent parity -oracle. The worker owns image decoding, the processor/chat template, vision and -language LoRA loading, and the candidate-letter probability readout. - -The text mapping and pinned revisions follow -[PR #11](https://github.com/ThinkFlowLab/system1-omni/pull/11). The image `state` -format below is this PR's proposed extension. A separately launched multimodal -worker uses the same request model name; deployment routing selects its modality. - -## Setup - -Run from this repository's root on Linux with an NVIDIA GPU. The measured CUDA -wheel, driver, GPU memory and results are recorded in [experiments](https://github.com/Levius-Fubuki/system1-omni/blob/683d470669f19d5e9530cd8233727a0d4f1f8d29/recipe/cua_s1/experiments/README.md). -Python 3.12 is required by the pinned environment. - -```sh -python3.12 -m venv .venv -. .venv/bin/activate -pip install -r recipe/cua_s1/requirements-multimodal.txt -PYTHONPATH=src python recipe/cua_s1/download_weights.py --dest weights -PYTHONPATH=src python -m frontend.cua_s1 \ - --base weights/Qwen3.5-4B \ - --adapter weights/cua-s1-4b-0.2/multimodal -``` - -The downloader fetches the upstream manifest at the fixed reference revision, -checks its pinned SHA-256 and saves it as `weights/weights.lock.json`. The worker -reads this local manifest and checks every loaded artifact's size and SHA-256. -Keep the manifest next to the base checkpoint directory when moving weights. -Extra files are rejected, except Hugging Face's `.cache` metadata, so another -checkpoint cannot silently override verified shards. Downloads require roughly -9 GB plus cache/install space. Loading is offline after the download completes. -Weights are not included in this repository. - -The HTTP adapter lives in `src/frontend/cua_s1.py`; model execution stays in -`src/models/cua_s1/multimodal/`. - -The worker binds to `127.0.0.1:8000` only after loading and a successful warmup. -`GET /health` returns `{"status":"ready","modality":"multimodal"}`. One request -runs at a time; concurrent requests return `503`. This is a loopback model worker, -with the Rust frontend and an ingress responsible for public serving. - -To use the Rust frontend included in this repository: - -```sh -cargo build --release --locked -OMNI_JEV_BIND=127.0.0.1:8080 \ -OMNI_JEV_BACKEND_URL=http://127.0.0.1:8000 ./target/release/omni-jev -``` - -## Request and response - -Generate a self-contained example with a synthetic settings screenshot: - -```sh -python recipe/cua_s1/make_example.py --output /tmp/cua-example.json -curl -sS http://127.0.0.1:8000/v1/systemone \ - -H 'Content-Type: application/json' --data-binary @/tmp/cua-example.json -``` - -Replace the port with `8080` to send the same request through the frontend. -The request shape is: - -```json -{ - "model": "cua-s1-4b-0.2", - "state": {"image": "data:image/png;base64,"}, - "questions": { - "next": { - "type": "choice", - "instructions": "Save the changes", - "criteria": {"save": "Save changes", "cancel": "Cancel"} - } - } -} -``` - -- `state` contains exactly one inline PNG or JPEG data URL. Images are decoded to - RGB. Local filenames, remote URLs, video, animation and mixed text/image state - are unsupported. -- There are 1–8 questions and 1–26 options per question. Option order assigns - letters A–Z. Labels are strings, objects, arrays or `null` (which uses the key). - Structured values use Python `json.dumps(..., ensure_ascii=False)` followed by - the upstream chooser's label escaping. Instructions accept strings, objects - or arrays and must be present; an empty string or `null` omits the goal block. -- Limits: 8 MiB body, 4 MiB decoded image, 2048 pixels per side, 1,048,576 pixels - total, at most 200:1 aspect ratio in either orientation, 16,384 characters per - question and 4096 processed tokens per question. - Every question is validated/preprocessed before any forward pass begins. -- `<|image_pad|>`, `<|video_pad|>`, `<|vision_start|>` and `<|vision_end|>` are - rejected in user text because the processor interprets them as media controls. - Other special-token spellings retain upstream tokenization behavior. -- Malformed JSON (including duplicate keys, non-finite numbers, invalid UTF-8, - lone surrogates and non-object bodies) returns `400`. Well-formed unsupported - inputs, missing `instructions`, unsupported models and `score`/`noul` questions - return `422`. Error bodies use `{"detail": ""}`. Oversized bodies - return `413`; chunked uploads return `411`. Send `Content-Length` and `Content-Type: application/json`. - -Each answer has `type`, `choice`, `probabilities` and `confidence`. The readout -uses the last position's candidate-letter logits, casts to fp32 and applies -softmax over those letters only. There is no decode. Ties select the earliest -option; confidence is `1 - H(p)/ln(n)`, or 1 for one option. Each question has a -separate forward pass over the same screenshot. Usage sums processed input tokens -and reports zero output tokens. - -Response identity: -`cua-ai/cua-s1-4b-0.2@16818868b0cc7813808aae4e87b417657046ab79:multimodal`. -The base is BF16; PEFT's rank-16, alpha-32 adapter remains unmerged with fp32 LoRA -branches, including 50 vision projection modules (178 total adapted modules). - -## Reproduce correctness and profiling - -```sh -git clone https://github.com/trycua/cua.git /tmp/cua-reference -git -C /tmp/cua-reference checkout 0e75660ce4c2edda519e0c795fa3ad98abf4e76f -for mode in reference candidate benchmark; do - PYTHONPATH=src python recipe/cua_s1/evaluate_multimodal.py \ - --weights weights --reference /tmp/cua-reference \ - --output /tmp/cua-evidence --mode "$mode" -done -``` - -The evaluator hashes the pinned `four_b.py` before importing it and checks report -provenance/environment before comparison. Reference and candidate are separate -processes to avoid keeping two models in GPU memory. Eight synthetic requests -(nine question forwards) cover two resolutions, PNG/JPEG, 1/26 candidates, -structured/Unicode labels, special-token text and multiple questions. It requires -identical processor tensor shapes/dtypes/hashes and identical fp32 candidate -probabilities; numerical tolerance is zero. These are integration/parity fixtures, -not an evaluation of GUI task success. - -Benchmark mode records two runs of 50 serial requests after five warmups on the -640×480 fixture, with synchronized end-to-end engine latency, p50/p95, serial -throughput, allocated/reserved GPU peaks and a separate operator profile. Load -and warmup are recorded separately; candidate load time includes artifact hash -verification. See the experiment report for measured scope and limitations. - -CPU-only validation: - -```sh -pip install Pillow==11.3.0 pytest==9.1.1 ruff==0.16.8 -PYTHONPATH=src python -m pytest tests/cua_s1 -q -ruff check --select E4,E7,E9,F,I src/frontend/cua_s1.py src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 -ruff format --check src/frontend/cua_s1.py src/models/cua_s1/multimodal recipe/cua_s1/*.py tests/cua_s1 -``` - -Metal, native CUDA kernels, text-adapter serving, batching, caching and training -are outside this worker's scope. This implementation does not import or modify -another contributor's text engine. diff --git a/recipe/cua_s1/check_frontend.py b/recipe/cua_s1/check_frontend.py deleted file mode 100644 index 37fe0a0..0000000 --- a/recipe/cua_s1/check_frontend.py +++ /dev/null @@ -1,75 +0,0 @@ -"""Compare HTTP response status, content type and bytes through Rust and directly.""" - -import argparse -import hashlib -import json -import urllib.error -import urllib.request -from pathlib import Path - - -def exchange(base, route, body=None): - request = urllib.request.Request( - base.rstrip("/") + route, - data=body, - headers={"Content-Type": "application/json"}, - ) - try: - response = urllib.request.urlopen(request, timeout=60) - except urllib.error.HTTPError as exc: - response = exc - with response: - return response.status, response.headers.get("Content-Type"), response.read() - - -def main(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--worker", default="http://127.0.0.1:8000") - parser.add_argument("--frontend", default="http://127.0.0.1:8080") - parser.add_argument("--fixtures", type=Path, required=True) - parser.add_argument("--output", type=Path, required=True) - args = parser.parse_args() - fixtures = sorted(args.fixtures.glob("*.json")) - if not fixtures: - raise ValueError("at least one valid fixture is required") - checks = [("health", "/health", None)] - checks += [(p.stem, "/v1/systemone", p.read_bytes()) for p in fixtures] - checks += [ - ("invalid", "/v1/systemone", b"{}"), - ("duplicate", "/v1/systemone", b'{"model":1,"model":2}'), - ] - missing = json.loads(fixtures[0].read_bytes()) - next(iter(missing["questions"].values())).pop("instructions") - checks.append( - ("missing-instructions", "/v1/systemone", json.dumps(missing).encode()) - ) - report = [] - for name, route, body in checks: - direct = exchange(args.worker, route, body) - proxied = exchange(args.frontend, route, body) - assert direct == proxied, f"frontend changed response: {name}" - expected_status = { - "invalid": 422, - "duplicate": 400, - "missing-instructions": 422, - }.get(name, 200) - assert direct[0] == expected_status, f"unexpected status: {name}: {direct[0]}" - if expected_status >= 400: - assert set(json.loads(direct[2])) == {"detail"}, ( - f"wrong error envelope: {name}" - ) - report.append( - { - "name": name, - "status": direct[0], - "content_type": direct[1], - "body_sha256": hashlib.sha256(direct[2]).hexdigest(), - "identical": True, - } - ) - print(f"{name}: HTTP {direct[0]}, identical response", flush=True) - args.output.write_text(json.dumps(report, indent=2)) - - -if __name__ == "__main__": - main() diff --git a/recipe/cua_s1/download_weights.py b/recipe/cua_s1/download_weights.py deleted file mode 100644 index f1474bb..0000000 --- a/recipe/cua_s1/download_weights.py +++ /dev/null @@ -1,41 +0,0 @@ -"""Download the upstream-pinned artifacts and verify their checksums.""" - -import argparse -from pathlib import Path -from urllib.request import urlopen - -from huggingface_hub import snapshot_download - -from models.cua_s1.multimodal.model import ( - REFERENCE_REVISION, - parse_weights_manifest, - verify_weights, -) - - -def main(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--dest", type=Path, required=True) - args = parser.parse_args() - url = ( - f"https://raw.githubusercontent.com/trycua/cua/{REFERENCE_REVISION}/" - "libs/cua-s1/ci/weights.lock.json" - ) - with urlopen(url, timeout=30) as response: - raw = response.read() - manifest = parse_weights_manifest(raw) - for artifact in manifest["artifacts"]: - snapshot_download( - repo_id=artifact["repo_id"], - revision=artifact["revision"], - local_dir=args.dest / artifact["name"], - allow_patterns=list(artifact["files"]), - token=False, - ) - (args.dest / "weights.lock.json").write_bytes(raw) - verify_weights(args.dest / "Qwen3.5-4B", args.dest / "cua-s1-4b-0.2/multimodal") - print("Pinned base and multimodal adapter checksums verified.") - - -if __name__ == "__main__": - main() diff --git a/recipe/cua_s1/evaluate_multimodal.py b/recipe/cua_s1/evaluate_multimodal.py deleted file mode 100644 index ca3e535..0000000 --- a/recipe/cua_s1/evaluate_multimodal.py +++ /dev/null @@ -1,322 +0,0 @@ -"""Reproduce upstream parity and bounded GPU measurements on fixed synthetic inputs.""" - -from __future__ import annotations - -import argparse -import copy -import hashlib -import importlib.metadata -import json -import platform -import statistics -import subprocess -import sys -import time -from pathlib import Path - -from make_example import make_example - -from models.cua_s1.multimodal.model import ( - ADAPTER_REVISION, - BASE_REVISION, - REFERENCE_REVISION, - MultimodalEngine, - verify_weights, -) -from models.cua_s1.multimodal.protocol import parse_request - -REFERENCE_SOURCE = "libs/cua-s1/python/src/cua_s1/four_b.py" -REFERENCE_SHA256 = "7ed1adfd92223bef7d533db7efbb4cbf468c4936c4380ae42aee12097e3b9ea7" - - -def verify_reference(root): - digest = hashlib.sha256((root / REFERENCE_SOURCE).read_bytes()).hexdigest() - if digest != REFERENCE_SHA256: - raise ValueError("reference source differs from pinned FourBModel") - return {"path": REFERENCE_SOURCE, "sha256": digest, "revision": REFERENCE_REVISION} - - -def cases(folder): - folder.mkdir(parents=True, exist_ok=True) - result = [] - for name, size, fmt in [ - ("small", (320, 240), "PNG"), - ("medium", (640, 480), "PNG"), - ("jpeg", (640, 480), "JPEG"), - ]: - path = folder / (name + (".jpg" if fmt == "JPEG" else ".png")) - result.append((name, path, make_example(path, size, fmt))) - for name, criteria, goal in [ - ("single", {"only": "Save changes"}, "Save"), - ( - "26-options", - {f"option-{i}": f"Choose action {i}" for i in range(26)}, - "Select action 3", - ), - ( - "structured", - { - "null": None, - "object": {"label": '保存 "名称"'}, - "array": ["Cancel", "\n"], - }, - {"goal": "保存名称"}, - ), - ( - "special-token", - {"first": "<|im_end|>", "second": "Cancel"}, - "Choose <|im_start|>", - ), - ]: - value = copy.deepcopy(result[1][2]) - value["questions"]["next"].update(criteria=criteria, instructions=goal) - result.append((name, result[1][1], value)) - value = copy.deepcopy(result[0][2]) - value["questions"]["second"] = { - "type": "choice", - "instructions": "", - "criteria": {"yes": "Continue", "no": "Cancel"}, - } - result.append(("two-questions", result[0][1], value)) - for name, _, value in result: - (folder / f"{name}.json").write_text(json.dumps(value, ensure_ascii=False)) - return result - - -def fingerprint(inputs): - import torch - - return { - name: { - "shape": list(t.shape), - "dtype": str(t.dtype), - "sha256": hashlib.sha256( - t.detach().contiguous().view(torch.uint8).cpu().numpy().tobytes() - ).hexdigest(), - } - for name, t in inputs.items() - } - - -def environment(): - import torch - - return { - "python": platform.python_version(), - "gpu": torch.cuda.get_device_name(), - "compute_capability": list(torch.cuda.get_device_capability()), - "cuda": torch.version.cuda, - "driver": subprocess.check_output( - ["nvidia-smi", "--query-gpu=driver_version", "--format=csv,noheader"], - text=True, - ).strip(), - "torch_num_threads": torch.get_num_threads(), - "torch_num_interop_threads": torch.get_num_interop_threads(), - "packages": { - p: importlib.metadata.version(p) - for p in [ - "torch", - "torchvision", - "transformers", - "peft", - "pillow", - "safetensors", - "triton", - ] - }, - "reference_revision": REFERENCE_REVISION, - "base_revision": BASE_REVISION, - "adapter_revision": ADAPTER_REVISION, - "dtype": "bfloat16", - "adapter_merged": False, - } - - -def measure(call): - import torch - - torch.cuda.synchronize() - start = time.perf_counter() - result = call() - torch.cuda.synchronize() - return result, (time.perf_counter() - start) * 1000 - - -def quantiles(values): - ordered = sorted(values) - return { - "p50_ms": statistics.median(values), - "p95_ms": ordered[max(0, __import__("math").ceil(0.95 * len(values)) - 1)], - } - - -def main(): - import torch - - p = argparse.ArgumentParser(description=__doc__) - p.add_argument("--weights", type=Path, required=True) - p.add_argument( - "--reference", type=Path, required=True, help="checkout of pinned trycua/cua" - ) - p.add_argument("--output", type=Path, required=True) - p.add_argument( - "--mode", choices=["reference", "candidate", "benchmark"], required=True - ) - args = p.parse_args() - args.output.mkdir(parents=True, exist_ok=True) - base = str(args.weights / "Qwen3.5-4B") - adapter = str(args.weights / "cua-s1-4b-0.2/multimodal") - fixture_set = cases(args.output / "fixtures") - report = {"environment": environment(), "mode": args.mode, "cases": []} - report["reference_source"] = verify_reference(args.reference) - if args.mode == "reference": - _, report["artifact_verification_ms"] = measure( - lambda: verify_weights(Path(base), Path(adapter)) - ) - sys.path.insert(0, str(args.reference / "libs/cua-s1/python/src")) - from cua_s1.four_b import FourBModel, Option, assign_letters, build_prompt - - model = FourBModel( - base_model=base, lora_adapter_path=adapter, modality="multimodal" - ) - _, report["load_ms"] = measure(model.load) - first = True - for name, path, value in fixture_set: - request = parse_request(value) - for q in request.questions: - options = [ - Option(element_id=k, role="Decision", label=v, action="select") - for k, v in zip(q.keys, q.labels) - ] - messages = build_prompt( - assign_letters(options), - app="Cua Driver", - task_family="closed-candidate decision", - screenshot=path, - modality="multimodal", - goal=q.goal, - ) - text = model._processor.apply_chat_template( - messages, tokenize=False, add_generation_prompt=True - ) - inputs = model._processor( - text=[text], images=[request.image], return_tensors="pt" - ) - - def call(): - return model.forward( - options, - app="Cua Driver", - task_family="closed-candidate decision", - screenshot=path, - goal=q.goal, - ) - - if first: - _, report["warmup_ms"] = measure(call) - first = False - output, elapsed = measure(call) - report["cases"].append( - { - "name": name, - "question": q.name, - "inputs": fingerprint(inputs), - "probabilities": [x.probability for x in output], - "latency_ms": elapsed, - } - ) - print(f"reference {name}/{q.name}: {elapsed:.1f} ms", flush=True) - else: - engine, report["load_ms"] = measure(lambda: MultimodalEngine(base, adapter)) - _, report["warmup_ms"] = measure(engine.warmup) - report["adapter_modules"] = engine.adapter_modules - if args.mode == "candidate": - reference = json.loads((args.output / "reference.json").read_text()) - if ( - reference["reference_source"] != report["reference_source"] - or reference["environment"] != report["environment"] - ): - raise ValueError("reference provenance or environment mismatch") - expected = {(x["name"], x["question"]): x for x in reference["cases"]} - for name, _, value in fixture_set: - request = parse_request(value) - for q in request.questions: - inputs = engine.prepare(request.image, q) - probabilities, elapsed = measure(lambda: engine.score(inputs, q)) - ref = expected[name, q.name] - difference = max( - abs(a - b) for a, b in zip(ref["probabilities"], probabilities) - ) - assert fingerprint(inputs) == ref["inputs"], ( - f"preprocessing mismatch: {name}" - ) - assert probabilities == ref["probabilities"], ( - f"probability mismatch: {name}: {difference}" - ) - report["cases"].append( - { - "name": name, - "question": q.name, - "inputs": fingerprint(inputs), - "probabilities": probabilities, - "max_abs_difference": difference, - "latency_ms": elapsed, - } - ) - print(f"candidate {name}/{q.name}: exact parity", flush=True) - else: - request = parse_request(fixture_set[1][2]) - for _ in range(5): - engine.predict(request) - report["benchmark"] = { - "case": "medium", - "batch_size": 1, - "concurrency": 1, - "warmup_requests": 5, - "runs": [], - } - for run in range(2): - torch.cuda.reset_peak_memory_stats() - latencies = [ - measure(lambda: engine.predict(request))[1] for _ in range(50) - ] - report["benchmark"]["runs"].append( - { - "run": run + 1, - "latencies_ms": latencies, - **quantiles(latencies), - "requests_per_second_serial": 1000 / statistics.mean(latencies), - "peak_allocated_bytes": torch.cuda.max_memory_allocated(), - "peak_reserved_bytes": torch.cuda.max_memory_reserved(), - } - ) - print(f"benchmark run {run + 1}: {quantiles(latencies)}", flush=True) - with torch.profiler.profile( - activities=[ - torch.profiler.ProfilerActivity.CPU, - torch.profiler.ProfilerActivity.CUDA, - ], - record_shapes=True, - ) as prof: - engine.predict(request) - torch.cuda.synchronize() - report["profile"] = [ - { - "op": x.key, - "count": x.count, - "device_time_us": x.device_time_total, - "self_device_time_us": x.self_device_time_total, - "cpu_time_us": x.cpu_time_total, - "self_cpu_time_us": x.self_cpu_time_total, - } - for x in sorted( - prof.key_averages(), key=lambda x: x.device_time_total, reverse=True - )[:30] - ] - report["peak_allocated_bytes"] = torch.cuda.max_memory_allocated() - (args.output / f"{args.mode}.json").write_text(json.dumps(report, indent=2)) - print(f"Saved {args.mode}.json", flush=True) - - -if __name__ == "__main__": - main() diff --git a/recipe/cua_s1/make_example.py b/recipe/cua_s1/make_example.py deleted file mode 100644 index 8adbc53..0000000 --- a/recipe/cua_s1/make_example.py +++ /dev/null @@ -1,71 +0,0 @@ -"""Create redistributable synthetic GUI fixtures and inline-image requests.""" - -from __future__ import annotations - -import argparse -import base64 -import json -from pathlib import Path - -from PIL import Image, ImageDraw - - -def make_example(path: Path, size=(640, 480), fmt="PNG") -> dict: - image = Image.new("RGB", size, "#f4f6f8") - draw = ImageDraw.Draw(image) - w, h = size - draw.rectangle( - (w // 10, h // 8, w * 9 // 10, h * 7 // 8), - fill="white", - outline="#8899aa", - width=2, - ) - draw.text( - (w // 7, h // 5), "Account settings", fill="black", font_size=max(12, w // 25) - ) - draw.text( - (w // 7, h // 3), - "Display name: Alice", - fill="black", - font_size=max(10, w // 32), - ) - draw.rectangle((w // 7, h // 2, w * 4 // 7, h * 2 // 3), fill="#1460b4") - draw.text( - (w // 6, h * 13 // 24), "Save changes", fill="white", font_size=max(10, w // 32) - ) - draw.text( - (w * 5 // 8, h * 13 // 24), "Cancel", fill="black", font_size=max(10, w // 32) - ) - image.save(path, format=fmt) - mime = "jpeg" if fmt == "JPEG" else "png" - return { - "model": "cua-s1-4b-0.2", - "state": { - "image": f"data:image/{mime};base64," - + base64.b64encode(path.read_bytes()).decode() - }, - "questions": { - "next": { - "type": "choice", - "instructions": "Save the changed display name.", - "criteria": { - "save": "Click Save changes", - "cancel": "Click Cancel", - "wait": "Wait", - }, - } - }, - } - - -def main(): - parser = argparse.ArgumentParser(description=__doc__) - parser.add_argument("--output", type=Path, default=Path("example.json")) - args = parser.parse_args() - args.output.parent.mkdir(parents=True, exist_ok=True) - value = make_example(args.output.with_suffix(".png")) - args.output.write_text(json.dumps(value, ensure_ascii=False)) - - -if __name__ == "__main__": - main() diff --git a/recipe/cua_s1/requirements-multimodal.txt b/recipe/cua_s1/requirements-multimodal.txt deleted file mode 100644 index afb7422..0000000 --- a/recipe/cua_s1/requirements-multimodal.txt +++ /dev/null @@ -1,11 +0,0 @@ -# Reference package versions from trycua/cua 0e75660, four-b uv.lock. -# Lock the CUDA wheel/driver in the experiment report for the target GPU. -torch==2.14.0 -torchvision==0.29.0 -transformers==5.17.0 -peft==0.21.0 -accelerate==1.15.0 -Pillow==11.3.0 -safetensors==0.8.0 -huggingface-hub==1.32.0 -tokenizers==0.23.2 diff --git a/src/models/cua_s1/README.md b/src/models/cua_s1/README.md index 2646837..6a07260 100644 --- a/src/models/cua_s1/README.md +++ b/src/models/cua_s1/README.md @@ -2,7 +2,7 @@ This directory owns Cua-S1 4B 0.2 ([#10](https://github.com/ThinkFlowLab/system1-omni/issues/10)): request mapping, prompt construction, adapter selection, execution, and the answer-letter readout. This page records the pinned upstream revisions, the inference contract an implementation must match, and how its outputs will be compared with the upstream reference. -Status: the `multimodal` adapter has a Transformers/PEFT CUDA worker validated on RTX 4090. See the [multimodal recipe](../../../recipe/cua_s1/README.md) for its screenshot request extension, input limits and archived GPU parity results. The `text` worker is tracked separately in [PR #13](https://github.com/ThinkFlowLab/system1-omni/pull/13); the text mapping below remains its contract. +Status: planned; nothing is implemented or validated yet. The first target is the `text` adapter on CUDA, starting with a worker that loads the model directly through Hugging Face Transformers and PEFT. The `multimodal` adapter is deferred; see [Not covered yet](#not-covered-yet). ## Pinned revisions @@ -105,7 +105,7 @@ The bfloat16 worker's own difference from the fp32 worker is reported next to ea ## Not covered yet -- Text-adapter serving in this checkout (tracked in PR #13), and native GPU kernels. +- The `multimodal` adapter: image preprocessing, the vision tower and the vision LoRA. This is tracked in [#10](https://github.com/ThinkFlowLab/system1-omni/issues/10). - `score` and `noul` questions. - More than 26 options per question. - The Metal backend. diff --git a/tests/cua_s1/test_download_weights.py b/tests/cua_s1/test_download_weights.py deleted file mode 100644 index 5fb9939..0000000 --- a/tests/cua_s1/test_download_weights.py +++ /dev/null @@ -1,83 +0,0 @@ -"""Download setup uses the upstream manifest without checking a copy into source.""" - -import importlib.util -import io -import json -import sys -from pathlib import Path -from types import SimpleNamespace - -import pytest - - -@pytest.mark.parametrize("corrupt", [False, True]) -def test_manifest_is_verified_before_downloading_weights( - tmp_path, monkeypatch, corrupt -): - from models.cua_s1.multimodal import model - - raw = json.dumps( - { - "artifacts": [ - { - "repo_id": "test/base", - "revision": "fixed", - "name": "Qwen3.5-4B", - "files": {"config.json": {}}, - } - ] - } - ).encode() - import hashlib - - monkeypatch.setattr( - model, "WEIGHTS_MANIFEST_SHA256", hashlib.sha256(raw).hexdigest() - ) - downloads = [] - - def download(**kwargs): - downloads.append(kwargs) - kwargs["local_dir"].mkdir(parents=True) - - monkeypatch.setitem( - sys.modules, "huggingface_hub", SimpleNamespace(snapshot_download=download) - ) - path = Path(__file__).resolve().parents[2] / "recipe/cua_s1/download_weights.py" - spec = importlib.util.spec_from_file_location("download_weights_test", path) - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - payload = raw + b" " if corrupt else raw - urls = [] - - def fetch(url, timeout): - urls.append(url) - assert timeout == 30 - return io.BytesIO(payload) - - monkeypatch.setattr(module, "urlopen", fetch) - verified = [] - - def verify(base, adapter): - assert (base.parent / "weights.lock.json").read_bytes() == raw - verified.append((base, adapter)) - - monkeypatch.setattr(module, "verify_weights", verify) - dest = tmp_path / "weights" - monkeypatch.setattr(sys, "argv", [str(path), "--dest", str(dest)]) - if corrupt: - with pytest.raises(ValueError, match="manifest checksum"): - module.main() - assert not downloads and not verified and not dest.exists() - else: - module.main() - assert downloads == [ - { - "repo_id": "test/base", - "revision": "fixed", - "local_dir": dest / "Qwen3.5-4B", - "allow_patterns": ["config.json"], - "token": False, - } - ] - assert verified == [(dest / "Qwen3.5-4B", dest / "cua-s1-4b-0.2/multimodal")] - assert model.REFERENCE_REVISION in urls[0] diff --git a/tests/cua_s1/test_evaluation.py b/tests/cua_s1/test_evaluation.py deleted file mode 100644 index cca8a45..0000000 --- a/tests/cua_s1/test_evaluation.py +++ /dev/null @@ -1,65 +0,0 @@ -"""Provenance checks must fail before a changed oracle can certify parity.""" - -import importlib.util -from pathlib import Path - -import pytest - - -def test_modified_reference_is_rejected(tmp_path, monkeypatch): - recipe = Path(__file__).resolve().parents[2] / "recipe/cua_s1" - monkeypatch.syspath_prepend(str(recipe)) - spec = importlib.util.spec_from_file_location( - "evaluation", recipe / "evaluate_multimodal.py" - ) - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - source = tmp_path / module.REFERENCE_SOURCE - source.parent.mkdir(parents=True) - source.write_text("modified oracle") - with pytest.raises(ValueError, match="reference source differs"): - module.verify_reference(tmp_path) - - -def test_environment_survives_json_roundtrip(monkeypatch): - import json - import sys - from types import SimpleNamespace - - recipe = Path(__file__).resolve().parents[2] / "recipe/cua_s1" - monkeypatch.syspath_prepend(str(recipe)) - spec = importlib.util.spec_from_file_location( - "evaluation", recipe / "evaluate_multimodal.py" - ) - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - torch = SimpleNamespace( - cuda=SimpleNamespace( - get_device_name=lambda: "GPU", get_device_capability=lambda: (8, 9) - ), - version=SimpleNamespace(cuda="13.0"), - get_num_threads=lambda: 16, - get_num_interop_threads=lambda: 16, - ) - monkeypatch.setitem(sys.modules, "torch", torch) - monkeypatch.setattr(module.importlib.metadata, "version", lambda name: "pinned") - monkeypatch.setattr(module.subprocess, "check_output", lambda *a, **kw: "driver\n") - environment = module.environment() - assert json.loads(json.dumps(environment)) == environment - - -def test_generated_fixtures_satisfy_required_instructions(tmp_path, monkeypatch): - from models.cua_s1.multimodal.protocol import parse_request - - recipe = Path(__file__).resolve().parents[2] / "recipe/cua_s1" - monkeypatch.syspath_prepend(str(recipe)) - spec = importlib.util.spec_from_file_location( - "evaluation", recipe / "evaluate_multimodal.py" - ) - module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) - fixtures = module.cases(tmp_path) - requests = [parse_request(value) for _, _, value in fixtures] - assert len(requests) == 8 - assert sum(len(value.questions) for value in requests) == 9 - assert requests[-1].questions[1].goal == "" diff --git a/tests/cua_s1/test_model.py b/tests/cua_s1/test_model.py deleted file mode 100644 index d37728f..0000000 --- a/tests/cua_s1/test_model.py +++ /dev/null @@ -1,124 +0,0 @@ -import pytest - -from models.cua_s1.multimodal.model import letter_ids, validate_adapter_config - - -class Tokenizer: - def encode(self, text, add_special_tokens): - assert add_special_tokens is False - return [ord(text) - 33] - - -def test_letter_readout_is_in_candidate_order(): - assert letter_ids(Tokenizer(), 3) == [32, 33, 34] - - -def test_multitoken_letters_are_rejected(): - class Bad: - def encode(self, *args, **kwargs): - return [1, 2] - - with pytest.raises(ValueError, match="single token"): - letter_ids(Bad(), 2) - - -def test_text_adapter_is_rejected_before_model_load(): - with pytest.raises(ValueError, match="multimodal"): - validate_adapter_config( - { - "peft_type": "LORA", - "r": 16, - "lora_alpha": 32, - "target_modules": ["q_proj", "k_proj"], - } - ) - - -def test_multimodal_adapter_contract(): - validate_adapter_config( - { - "peft_type": "LORA", - "r": 16, - "lora_alpha": 32, - "base_model_name_or_path": "Qwen/Qwen3.5-4B", - "target_modules": [ - "q_proj", - "k_proj", - "v_proj", - "o_proj", - "gate_proj", - "up_proj", - "down_proj", - "linear_fc1", - "linear_fc2", - ], - } - ) - - -def test_unlisted_model_files_cannot_override_verified_shards(tmp_path, monkeypatch): - import hashlib - import json - - from models.cua_s1.multimodal import model - - base, adapter = tmp_path / "base", tmp_path / "adapter" - base.mkdir() - adapter.mkdir() - (base / "config.json").write_bytes(b"{}") - manifest = tmp_path / "weights.lock.json" - manifest.write_text( - json.dumps( - { - "artifacts": [ - { - "role": "base", - "files": { - "config.json": { - "size": 2, - "sha256": hashlib.sha256(b"{}").hexdigest(), - } - }, - } - ] - } - ) - ) - monkeypatch.setattr( - model, - "WEIGHTS_MANIFEST_SHA256", - hashlib.sha256(manifest.read_bytes()).hexdigest(), - ) - model.verify_weights(base, adapter) - (base / "model.safetensors").write_bytes(b"override") - with pytest.raises(ValueError, match="unlisted"): - model.verify_weights(base, adapter) - - -def test_all_question_lengths_are_checked_before_inference(): - from models.cua_s1.multimodal.model import MultimodalEngine - from models.cua_s1.multimodal.protocol import InvalidRequest, Question, Request - - engine = object.__new__(MultimodalEngine) - first = Question("first", ("a",), ("A",), "") - second = Question("second", ("b",), ("B",), "") - forwarded = [] - - def prepare(image, question): - if question.name == "second": - raise InvalidRequest("processed prompt exceeds 4096 tokens") - return {} - - engine.prepare = prepare - engine.score = lambda inputs, q: forwarded.append(q) - with pytest.raises(InvalidRequest, match="4096"): - engine.predict(Request(None, (first, second))) - assert forwarded == [] - - -def test_weights_manifest_must_match_pinned_upstream_digest(tmp_path): - from models.cua_s1.multimodal.model import verify_weights - - (tmp_path / "weights.lock.json").write_text('{"artifacts": []}') - with pytest.raises(ValueError, match="manifest checksum"): - verify_weights(tmp_path / "base", tmp_path / "adapter") diff --git a/tests/cua_s1/test_protocol.py b/tests/cua_s1/test_protocol.py deleted file mode 100644 index 00ecd76..0000000 --- a/tests/cua_s1/test_protocol.py +++ /dev/null @@ -1,172 +0,0 @@ -import base64 -import io -import json - -import pytest -from PIL import Image - -from models.cua_s1.multimodal.protocol import ( - InvalidRequest, - answer, - build_messages, - decode_request, - parse_request, -) - - -def image_url(fmt="PNG", size=(32, 32)): - out = io.BytesIO() - Image.new("RGB", size, "white").save(out, format=fmt) - mime = "jpeg" if fmt == "JPEG" else "png" - return f"data:image/{mime};base64," + base64.b64encode(out.getvalue()).decode() - - -def request(): - return { - "model": "cua-s1-4b-0.2", - "state": {"image": image_url()}, - "questions": { - "next": { - "type": "choice", - "instructions": "Submit the form", - "criteria": {"submit": "Submit", "cancel": "Cancel"}, - } - }, - } - - -def test_image_and_order_are_preserved(): - r = parse_request(request()) - assert r.image.mode == "RGB" and r.image.size == (32, 32) - assert r.questions[0].keys == ("submit", "cancel") - assert r.questions[0].labels == ("Submit", "Cancel") - - -def test_prompt_keeps_image_block_and_upstream_text(): - q = parse_request(request()).questions[0] - msg = build_messages(q) - assert msg[1]["content"][0]["type"] == "image" - assert msg[1]["content"][1]["text"] == ( - "Goal: Submit the form\n\nApp: Cua Driver\nTask family: closed-candidate decision\n\n" - "The current screenshot is attached.\n\nOptions:\n" - 'A. Decision "Submit" -> select\nB. Decision "Cancel" -> select\n\n' - "Answer with a single letter." - ) - - -def test_structured_values_and_escaping(): - r = request() - r["questions"]["next"]["instructions"] = {"目标": "提交"} - r["questions"]["next"]["criteria"] = {"fallback": None, "obj": {"x": '"\n'}} - q = parse_request(r).questions[0] - assert q.goal == '{"目标": "提交"}' - assert q.labels == ( - "fallback", - json.dumps(json.dumps({"x": '"\n'}, ensure_ascii=False), ensure_ascii=False)[ - 1:-1 - ], - ) - - -@pytest.mark.parametrize( - "state", - [ - {}, - {"image": "/etc/passwd"}, - {"image": "https://example.com/a.png"}, - {"image": "data:image/png;base64,!!"}, - {"image": image_url(), "text": "ignored"}, - ], -) -def test_invalid_images_and_unknown_state_fields(state): - r = request() - r["state"] = state - with pytest.raises(InvalidRequest): - parse_request(r) - - -def test_mime_mismatch_and_oversized_dimensions(): - for url in [ - image_url().replace("image/png", "image/jpeg"), - image_url(size=(2049, 1)), - ]: - r = request() - r["state"]["image"] = url - with pytest.raises(InvalidRequest): - parse_request(r) - - -@pytest.mark.parametrize( - "criteria", [{}, {str(i): "x" for i in range(27)}, {"a": 1}, {"a": True}] -) -def test_invalid_candidates(criteria): - r = request() - r["questions"]["next"]["criteria"] = criteria - with pytest.raises(InvalidRequest): - parse_request(r) - - -@pytest.mark.parametrize("kind", ["score", "noul"]) -def test_unsupported_question_rejects_entire_request(kind): - r = request() - r["questions"]["bad"] = {"type": kind, "instructions": "x"} - with pytest.raises(InvalidRequest): - parse_request(r) - - -def test_duplicate_keys_and_nonfinite_json(): - for raw in [ - b'{"model":1,"model":2}', - b'{"x":NaN}', - b'{"x":Infinity}', - b"[]", - b"not json", - ]: - with pytest.raises(InvalidRequest): - decode_request(raw) - - -def test_entropy_confidence_and_earliest_tie(): - q = parse_request(request()).questions[0] - result = answer(q, [0.5, 0.5]) - assert result["choice"] == "submit" - assert result["confidence"] == 0.0 - assert result["probabilities"] == {"submit": 0.5, "cancel": 0.5} - r = request() - r["questions"]["next"]["criteria"] = {"only": "Only"} - assert answer(parse_request(r).questions[0], [1.0])["confidence"] == 1.0 - - -def test_jpeg_supported(): - r = request() - r["state"]["image"] = image_url("JPEG") - assert parse_request(r).image.mode == "RGB" - - -@pytest.mark.parametrize( - "token", ["<|image_pad|>", "<|video_pad|>", "<|vision_start|>", "<|vision_end|>"] -) -@pytest.mark.parametrize("field", ["instructions", "criteria"]) -def test_media_control_tokens_are_rejected(token, field): - value = request() - value["questions"]["next"][field] = ( - token if field == "instructions" else {"a": {"text": token}} - ) - with pytest.raises(InvalidRequest, match="control token"): - parse_request(value) - - -@pytest.mark.parametrize("instructions", ["", None]) -def test_explicit_empty_or_null_instructions_omits_goal(instructions): - value = request() - value["questions"]["next"]["instructions"] = instructions - question = parse_request(value).questions[0] - assert question.goal == "" - assert "Goal:" not in build_messages(question)[1]["content"][1]["text"] - - -@pytest.mark.parametrize("size", [(200, 1), (1, 200), (199, 1), (1, 199)]) -def test_supported_aspect_ratio_boundary(size): - value = request() - value["state"]["image"] = image_url(size=size) - assert parse_request(value).image.size == size diff --git a/tests/cua_s1/test_server.py b/tests/cua_s1/test_server.py deleted file mode 100644 index 50bd2cb..0000000 --- a/tests/cua_s1/test_server.py +++ /dev/null @@ -1,204 +0,0 @@ -import json -import threading -import urllib.error -import urllib.request - -import pytest -from test_protocol import image_url, request - -from frontend.cua_s1 import Handler, WorkerServer -from models.cua_s1.multimodal.protocol import MAX_BODY - - -class Engine: - def predict(self, parsed): - return { - "model": "test:multimodal", - "answers": {q.name: {"type": "choice"} for q in parsed.questions}, - } - - -@pytest.fixture -def worker(): - server = WorkerServer(("127.0.0.1", 0), Engine()) - thread = threading.Thread(target=server.serve_forever, daemon=True) - thread.start() - yield server, f"http://127.0.0.1:{server.server_port}" - server.shutdown() - server.server_close() - thread.join() - - -@pytest.fixture -def deferred_cleanup(worker, monkeypatch): - """Hold cleanup until the test starts waiting for the inference lock.""" - server, _ = worker - lock = server.inference_lock - allow_cleanup = threading.Event() - - class DeferredLock: - def acquire(self, blocking=True, timeout=-1): - if blocking: - allow_cleanup.set() - return lock.acquire(blocking, timeout) - - def release(self): - lock.release() - - def locked(self): - return lock.locked() - - server.inference_lock = DeferredLock() - send_json = Handler.send_json - - def send_then_wait(handler, status, value): - send_json(handler, status, value) - # The client has the complete response, but finally has not run yet. - if not allow_cleanup.wait(timeout=5): - raise AssertionError("test did not wait for handler cleanup") - - monkeypatch.setattr(Handler, "send_json", send_then_wait) - try: - yield worker - finally: - allow_cleanup.set() - - -def call(url, body=None, **headers): - data = ( - body if isinstance(body, bytes) or body is None else json.dumps(body).encode() - ) - req = urllib.request.Request( - url, data=data, headers={"Content-Type": "application/json", **headers} - ) - try: - with urllib.request.urlopen(req, timeout=5) as response: - return response.status, json.load(response) - except urllib.error.HTTPError as response: - return response.code, json.load(response) - - -def test_health_and_prediction(worker): - _, url = worker - assert call(url + "/health")[0] == 200 - status, body = call(url + "/v1/systemone", request()) - assert status == 200 and "next" in body["answers"] - - -def test_invalid_question_never_reaches_model(worker): - _, url = worker - r = request() - r["questions"]["next"]["type"] = "noul" - status, body = call(url + "/v1/systemone", r) - assert status == 422 and set(body) == {"detail"} - - -def test_busy_worker_rejects_instead_of_queueing_gpu_work(worker): - server, url = worker - server.inference_lock.acquire() - try: - assert call(url + "/health")[0] == 200 - assert call(url + "/v1/systemone", request())[0] == 503 - finally: - server.inference_lock.release() - - -def test_body_limit_checked_before_reading(worker): - _, url = worker - assert ( - call(url + "/v1/systemone", {}, **{"Content-Length": str(MAX_BODY + 1)})[0] - == 413 - ) - - -def test_model_failure_is_not_reported_as_success(deferred_cleanup): - server, url = deferred_cleanup - - def fail(_): - raise RuntimeError("private file or input must not leak") - - server.engine.predict = fail - status, body = call(url + "/v1/systemone", request()) - assert status == 500 and body == {"detail": "inference failed"} - assert server.inference_lock.acquire(timeout=2), "handler did not release the lock" - server.inference_lock.release() - - -@pytest.mark.parametrize( - "raw", - [ - b"not json", - b'{"x":1,"x":2}', - b'{"x":{"y":1,"y":2}}', - b'{"x":NaN}', - b'{"x":Infinity}', - b'{"x":1e400}', - b"[]", - b"null", - b'{"x":"\xff"}', - b'{"x":"\\ud800"}', - b'{"\\ud800":1}', - b"[" * 2000 + b"]" * 2000, - "{}".encode("utf-16"), - ], -) -def test_malformed_json_returns_400_without_inference(deferred_cleanup, raw): - server, url = deferred_cleanup - - def unexpected(_): - raise AssertionError("malformed JSON reached inference") - - server.engine.predict = unexpected - status, body = call(url + "/v1/systemone", raw) - assert status == 400 and set(body) == {"detail"} - assert server.inference_lock.acquire(timeout=2), "handler did not release the lock" - server.inference_lock.release() - - -def test_missing_instructions_rejects_whole_request(worker): - server, url = worker - value = request() - value["questions"]["second"] = {"type": "choice", "criteria": {"a": "A"}} - - def unexpected(_): - raise AssertionError("invalid question reached inference") - - server.engine.predict = unexpected - status, body = call(url + "/v1/systemone", value) - assert status == 422 and set(body) == {"detail"} - assert "instructions" in body["detail"] - - -@pytest.mark.parametrize( - "route,body,headers,status", - [ - ("/missing", None, {}, 404), - ("/missing", {}, {}, 404), - ("/v1/systemone", {}, {"Content-Type": "text/plain"}, 415), - ("/v1/systemone", {}, {"Transfer-Encoding": "chunked"}, 411), - ("/v1/systemone", {}, {"Content-Length": "invalid"}, 400), - ("/v1/systemone", {}, {"Content-Length": str(MAX_BODY + 1)}, 413), - ], -) -def test_transport_errors_use_detail(worker, route, body, headers, status): - _, url = worker - actual, response = call(url + route, body, **headers) - assert actual == status and set(response) == {"detail"} - - -@pytest.mark.parametrize("size", [(2048, 1), (1, 2048), (201, 1), (1, 201)]) -def test_unsupported_image_aspect_ratio_returns_422_before_inference(worker, size): - server, url = worker - reached_engine = [] - - def unexpected(parsed): - reached_engine.append(parsed) - raise ValueError("unsupported processor input") - - server.engine.predict = unexpected - value = request() - value["state"]["image"] = image_url(size=size) - status, body = call(url + "/v1/systemone", value) - assert status == 422 and set(body) == {"detail"} - assert "aspect ratio" in body["detail"] - assert not reached_engine