From a173126ed03005b78f818ecf729a221f05a3f3ae Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Wed, 30 Sep 2026 21:11:42 +0800 Subject: [PATCH 1/2] Add reproducible Cua-S1 multimodal reference tensor exports --- .github/workflows/cua-s1-reference.yml | 31 ++ recipe/README.md | 3 + recipe/cua_s1/export_multimodal_reference.py | 334 ++++++++++++++++++ recipe/cua_s1/reference-validation.json | 294 +++++++++++++++ recipe/cua_s1/reference.md | 154 ++++++++ recipe/cua_s1/requirements-reference.txt | 13 + recipe/cua_s1/verify_multimodal_reference.py | 239 +++++++++++++ .../test_export_multimodal_reference.py | 260 ++++++++++++++ 8 files changed, 1328 insertions(+) create mode 100644 .github/workflows/cua-s1-reference.yml create mode 100644 recipe/cua_s1/export_multimodal_reference.py create mode 100644 recipe/cua_s1/reference-validation.json create mode 100644 recipe/cua_s1/reference.md create mode 100644 recipe/cua_s1/requirements-reference.txt create mode 100644 recipe/cua_s1/verify_multimodal_reference.py create mode 100644 tests/cua_s1/test_export_multimodal_reference.py diff --git a/.github/workflows/cua-s1-reference.yml b/.github/workflows/cua-s1-reference.yml new file mode 100644 index 0000000..4e59165 --- /dev/null +++ b/.github/workflows/cua-s1-reference.yml @@ -0,0 +1,31 @@ +name: Cua-S1 reference recipe +on: + pull_request: + paths: + - 'recipe/cua_s1/**' + - 'tests/cua_s1/test_export_multimodal_reference.py' + - 'src/models/cua_s1/multimodal/**' + - '.github/workflows/cua-s1-reference.yml' + push: + branches: [main] + paths: + - 'recipe/cua_s1/**' + - 'tests/cua_s1/test_export_multimodal_reference.py' + - 'src/models/cua_s1/multimodal/**' + - '.github/workflows/cua-s1-reference.yml' +permissions: + contents: read +jobs: + cpu: + runs-on: ubuntu-latest + timeout-minutes: 15 + steps: + - uses: actions/checkout@v5 + - uses: actions/setup-python@v5 + with: + python-version: '3.12' + - run: python -m pip install torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu + - run: python -m pip install Pillow==11.3.0 numpy==2.5.3 safetensors==0.8.0 pytest==9.1.1 ruff==0.16.8 + - run: PYTHONPATH=src python -m pytest tests/cua_s1/test_export_multimodal_reference.py -q + - run: ruff check --isolated --select E4,E7,E9,F,I recipe/cua_s1/export_multimodal_reference.py recipe/cua_s1/verify_multimodal_reference.py tests/cua_s1/test_export_multimodal_reference.py + - run: ruff format --isolated --check recipe/cua_s1/export_multimodal_reference.py recipe/cua_s1/verify_multimodal_reference.py tests/cua_s1/test_export_multimodal_reference.py diff --git a/recipe/README.md b/recipe/README.md index 4d0bf6c..05c2334 100644 --- a/recipe/README.md +++ b/recipe/README.md @@ -3,5 +3,8 @@ - [Laya text worker](laya/README.md): start the external Python worker, connect the Rust frontend and compare direct and proxied responses. +- [Cua-S1 multimodal reference tensors](cua_s1/reference.md): export fixed inputs + and actual vision/language boundary tensors for native integration. + Recipes contain setup, launch commands and examples. Reusable implementation code belongs under `src/`. diff --git a/recipe/cua_s1/export_multimodal_reference.py b/recipe/cua_s1/export_multimodal_reference.py new file mode 100644 index 0000000..89961a6 --- /dev/null +++ b/recipe/cua_s1/export_multimodal_reference.py @@ -0,0 +1,334 @@ +"""Export the #12 reference's actual multimodal forward inputs and readout.""" + +from __future__ import annotations + +import argparse +import base64 +import hashlib +import importlib.metadata +import importlib.util +import json +import os +import platform +import subprocess +from pathlib import Path + +from PIL import Image, ImageDraw + +from models.cua_s1.multimodal.model import ( + ADAPTER_REVISION, + BASE_REVISION, + REFERENCE_REVISION, + WEIGHTS_MANIFEST_SHA256, + MultimodalEngine, + letter_ids, +) +from models.cua_s1.multimodal.protocol import ( + answer, + build_messages, + decode_request, + parse_request, +) + +SCHEMA = "cua-s1-multimodal-reference-v1" +PACKAGES = { + "torch": "2.14.0", + "torchvision": "0.29.0", + "transformers": "5.17.0", + "peft": "0.21.0", + "accelerate": "1.15.0", + "Pillow": "11.3.0", + "safetensors": "0.8.0", + "huggingface-hub": "1.32.0", + "tokenizers": "0.23.2", + "numpy": "2.5.3", +} + + +def sha256(raw): + return hashlib.sha256(raw).hexdigest() + + +def write_json(path, value): + path.write_text( + json.dumps(value, ensure_ascii=False, indent=2, allow_nan=False) + "\n" + ) + + +def make_cases(folder): + folder.mkdir(parents=True) + cases = [] + for name, size, fmt in [ + ("small", (320, 240), "PNG"), + ("wide", (640, 320), "PNG"), + ("portrait", (320, 640), "PNG"), + ("jpeg", (640, 480), "JPEG"), + ("single-option", (320, 240), "PNG"), + ("26-options", (256, 256), "PNG"), + ("two-questions", (320, 240), "PNG"), + ]: + image = Image.new("RGB", size, "#f4f6f8") + draw = ImageDraw.Draw(image) + width, height = size + draw.rectangle( + (16, 16, width - 16, height - 16), fill="white", outline="#8899aa" + ) + draw.text((24, 24), "Account settings", fill="black") + draw.text((24, 48), "Display name: Alice", fill="black") + draw.rectangle((24, height // 2, width // 2, height // 2 + 32), fill="#1460b4") + draw.text((28, height // 2 + 8), "Save", fill="white") + draw.text((width // 2 + 16, height // 2 + 8), "Cancel", fill="black") + image_path = folder / (name + (".jpg" if fmt == "JPEG" else ".png")) + image.save(image_path, format=fmt) + criteria = {"save": "Click Save", "cancel": "Click Cancel", "wait": "Wait"} + if name == "single-option": + criteria = {"save": "Click Save"} + elif name == "26-options": + criteria = {f"option-{i}": f"Choose action {i}" for i in range(26)} + questions = { + "next": { + "type": "choice", + "instructions": "Save the changed display name.", + "criteria": criteria, + } + } + if name == "two-questions": + questions["second"] = { + "type": "choice", + "instructions": {"goal": "保存名称"}, + "criteria": {"continue": {"label": "Save"}, "cancel": None}, + } + mime = "jpeg" if fmt == "JPEG" else "png" + request = { + "model": "cua-s1-4b-0.2", + "state": { + "image": f"data:image/{mime};base64," + + base64.b64encode(image_path.read_bytes()).decode() + }, + "questions": questions, + } + write_json(folder / f"{name}.json", request) + cases.append({"name": name, "image": image_path.name, "request": request}) + return cases + + +def tensor_info(tensor): + import torch + + value = tensor.detach().cpu().contiguous() + return { + "shape": list(value.shape), + "dtype": str(value.dtype).removeprefix("torch."), + "sha256": sha256(value.view(torch.uint8).numpy().tobytes()), + } + + +def save_tensors(path, tensors): + from safetensors.torch import save_file + + # Clone individually: safetensors refuses shared storage, even for equal inputs. + tensors = { + name: value.detach().cpu().contiguous().clone() + for name, value in tensors.items() + } + save_file(tensors, str(path), metadata={"schema": SCHEMA}) + return {name: tensor_info(value) for name, value in tensors.items()} + + +def capture(engine, inputs, question): + import torch + + tensors = {} + + def keep(name, value): + if name in tensors: + raise RuntimeError(f"expected one forward per question: duplicate {name}") + tensors[name] = value.detach().cpu().contiguous().clone() + + def vision_hook(module, args, output): + keep("image_features", output.pooler_output) + + def language_pre_hook(module, args, kwargs): + keep("inputs_embeds", kwargs["inputs_embeds"]) + keep("position_ids", kwargs["position_ids"]) + + def language_hook(module, args, output): + keep("last_hidden_state", output.last_hidden_state[:, -1, :]) + + core = engine.model.get_base_model().model + hooks = [ + core.visual.register_forward_hook(vision_hook), + core.language_model.register_forward_pre_hook( + language_pre_hook, with_kwargs=True + ), + core.language_model.register_forward_hook(language_hook), + ] + try: + with torch.no_grad(): + output = engine.model( + **{ + name: value.to(engine.model.device) + for name, value in inputs.items() + } + ) + ids = torch.tensor( + letter_ids(engine.tokenizer, len(question.keys)), + device=output.logits.device, + ) + logits = output.logits[0, -1, ids] + keep("candidate_token_ids", ids) + keep("candidate_logits", logits) + keep("candidate_probabilities", torch.softmax(logits.float(), dim=-1)) + finally: + for handle in hooks: + handle.remove() + for name, value in inputs.items(): + keep(name, value) + keep("rope_deltas", core.rope_deltas) + indices = ( + (inputs["input_ids"][0] == engine.model.config.image_token_id) + .nonzero() + .flatten() + ) + keep("image_token_indices", indices) + if not torch.equal(tensors["inputs_embeds"][0, indices], tensors["image_features"]): + raise RuntimeError("image feature insertion differs from language input") + return tensors + + +def environment(): + import torch + from transformers.models.qwen3_5 import modeling_qwen3_5 + + packages = {name: importlib.metadata.version(name) for name in PACKAGES} + for name, expected in PACKAGES.items(): + if packages[name].split("+")[0] != expected: + raise ValueError(f"{name} must be {expected}, got {packages[name]}") + if any( + importlib.util.find_spec(name) is not None for name in ("fla", "causal_conv1d") + ): + raise ValueError( + "reference export requires the PyTorch DeltaNet path, without FLA" + ) + return { + "python": platform.python_version(), + "torch_build": str(torch.__version__), + "torch_num_threads": torch.get_num_threads(), + "torch_num_interop_threads": torch.get_num_interop_threads(), + "packages": packages, + "cuda": torch.version.cuda, + "gpu": torch.cuda.get_device_name(), + "compute_capability": list(torch.cuda.get_device_capability()), + "driver": subprocess.check_output( + ["nvidia-smi", "--query-gpu=driver_version", "--format=csv,noheader"], + text=True, + ).strip(), + "transformers_source_sha256": sha256( + Path(modeling_qwen3_5.__file__).read_bytes() + ), + "dtype": "bfloat16", + "adapter_merged": False, + "tf32": False, + "deterministic_algorithms": True, + "cublas_workspace_config": os.environ["CUBLAS_WORKSPACE_CONFIG"], + } + + +def export(weights, output): + if output.exists(): + raise FileExistsError(f"output must be a new directory: {output}") + # Must be set before Torch initializes CUDA, including model construction. + os.environ["CUBLAS_WORKSPACE_CONFIG"] = ":4096:8" + import torch + + torch.manual_seed(0) + torch.use_deterministic_algorithms(True) + torch.backends.cuda.matmul.allow_tf32 = False + torch.backends.cudnn.allow_tf32 = False + report = { + "schema": SCHEMA, + "reference_revision": REFERENCE_REVISION, + "base_revision": BASE_REVISION, + "adapter_revision": ADAPTER_REVISION, + "weights_manifest_sha256": WEIGHTS_MANIFEST_SHA256, + "environment": environment(), + "files": {}, + "questions": [], + } + engine = MultimodalEngine( + str(weights / "Qwen3.5-4B"), str(weights / "cua-s1-4b-0.2/multimodal") + ) + core = engine.model.get_base_model().model + report["execution"] = { + "visual_attention": core.visual.config._attn_implementation, + "text_attention": core.language_model.config._attn_implementation, + "processor_class": type(engine.processor).__name__, + "image_processor_class": type(engine.processor.image_processor).__name__, + } + output.mkdir(parents=True) + cases = make_cases(output / "inputs") + (output / "tensors").mkdir() + (output / "configs").mkdir() + for label, path in [ + ("base", weights / "Qwen3.5-4B/config.json"), + ("processor", weights / "Qwen3.5-4B/preprocessor_config.json"), + ("adapter", weights / "cua-s1-4b-0.2/multimodal/adapter_config.json"), + ]: + (output / "configs" / f"{label}.json").write_bytes(path.read_bytes()) + root = Path(__file__).resolve().parents[2] + report["source_sha256"] = { + name: sha256((root / name).read_bytes()) + for name in [ + "recipe/cua_s1/export_multimodal_reference.py", + "src/models/cua_s1/multimodal/model.py", + "src/models/cua_s1/multimodal/protocol.py", + ] + } + for case in cases: + request_path = output / "inputs" / f"{case['name']}.json" + request = parse_request(decode_request(request_path.read_bytes())) + for index, question in enumerate(request.questions): + inputs = engine.prepare(request.image, question) + tensors = capture(engine, inputs, question) + probabilities = tensors["candidate_probabilities"].tolist() + if probabilities != engine.score(inputs, question): + raise RuntimeError("hooked readout differs from ordinary #12 score") + relative = f"tensors/{case['name']}-{index}.safetensors" + entry = { + "case": case["name"], + "question": question.name, + "request": request_path.relative_to(output).as_posix(), + "image": f"inputs/{case['image']}", + "image_size_wh": list(request.image.size), + "option_keys": list(question.keys), + "prompt": engine.processor.apply_chat_template( + build_messages(question), tokenize=False, add_generation_prompt=True + ), + "tensors_file": relative, + "tensors": save_tensors(output / relative, tensors), + "answer": answer(question, probabilities), + "ordinary_score_equal": True, + } + report["questions"].append(entry) + print(f"exported {case['name']}/{question.name}", flush=True) + for path in sorted(output.rglob("*")): + if path.is_file(): + report["files"][path.relative_to(output).as_posix()] = { + "sha256": sha256(path.read_bytes()), + "size": path.stat().st_size, + } + # A manifest is written only after every forward and ordinary-score check passed. + write_json(output / "manifest.json", report) + return report + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--weights", type=Path, required=True) + parser.add_argument("--output", type=Path, required=True) + args = parser.parse_args() + export(args.weights, args.output) + + +if __name__ == "__main__": + main() diff --git a/recipe/cua_s1/reference-validation.json b/recipe/cua_s1/reference-validation.json new file mode 100644 index 0000000..f85f09f --- /dev/null +++ b/recipe/cua_s1/reference-validation.json @@ -0,0 +1,294 @@ +{ + "base_commit": "b50aa28", + "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", + "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", + "weights_manifest_sha256": "9820bd232c5762f114e19680c0f8203d7e1faaf8a60c196cfe01964d6d8a6c09", + "source_sha256": { + "recipe/cua_s1/export_multimodal_reference.py": "9ff84a0980d843f5015e319622533da4bdc9fd1ae74ed09be7e9dc4b08fa8cd2", + "src/models/cua_s1/multimodal/model.py": "e37a01feb72eaf60a9a3a36c0ddae7a78b466c6e1add5e2f1dc93c5b1947d2dc", + "src/models/cua_s1/multimodal/protocol.py": "eefc421dc485ab0a23d475a3a733b5d46f33cbe9febd6295e5d794e7fdfcec5f" + }, + "environment": { + "python": "3.12.3", + "torch_build": "2.14.0+cu130", + "torch_num_threads": 64, + "torch_num_interop_threads": 64, + "packages": { + "torch": "2.14.0", + "torchvision": "0.29.0", + "transformers": "5.17.0", + "peft": "0.21.0", + "accelerate": "1.15.0", + "Pillow": "11.3.0", + "safetensors": "0.8.0", + "huggingface-hub": "1.32.0", + "tokenizers": "0.23.2", + "numpy": "2.5.3" + }, + "cuda": "13.0", + "gpu": "NVIDIA GeForce RTX 4090", + "compute_capability": [ + 8, + 9 + ], + "driver": "595.71.05", + "transformers_source_sha256": "762feb6c7426a7f15b5bf830df54c07438bf9e7c27b8cdb23179045920412c3b", + "dtype": "bfloat16", + "adapter_merged": false, + "tf32": false, + "deterministic_algorithms": true, + "cublas_workspace_config": ":4096:8" + }, + "execution": { + "visual_attention": "sdpa", + "text_attention": "sdpa", + "processor_class": "Qwen3VLProcessor", + "image_processor_class": "Qwen2VLImageProcessor" + }, + "verification": { + "questions": 8, + "tensors": 112, + "files": 25, + "integrity_and_relations": "pass", + "independent_export_equality": "pass" + }, + "tests": { + "command": "PYTHONPATH=src python -m pytest tests/cua_s1/test_export_multimodal_reference.py -q", + "passed": 11, + "cuda_visible_devices": "" + }, + "ordinary_score_equality": { + "questions_per_run": 8, + "independent_runs": 2, + "comparison": "exact" + }, + "cpu_softmax_max_absolute_difference": 3.725290298461914e-09, + "questions": [ + { + "case": "small", + "question": "next", + "image_size_wh": [ + 320, + 240 + ], + "prompt_tokens": 235, + "patches": 320, + "image_tokens": 80, + "choice": "save", + "tensor_sha256": { + "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", + "inputs_embeds": "7761598e9b268125c2da4519a3a1a29724b4529f8b304413f80d43b516e98cf8", + "position_ids": "5f2b2d115b061ae076d3e08a6c6d1cb80e55afda4ae9d1e533d398d73c830449", + "last_hidden_state": "09d0e1519397478402144e773fc68e0dc23ca74b1be5ba8e06b8e9e3fc84016d", + "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", + "candidate_logits": "0f43844e3ae275526d3a659484f1d2ffad9fd23a1896b3717069ecef23679f16", + "candidate_probabilities": "b98be35be7e3291eab531ab69fbec351e175f6d99c009650dc81ae713d1c7f68", + "input_ids": "d6313d4f14a714ce7017e1c18c0c3c6332d65662c0662f327f3d646bde794aed", + "attention_mask": "f72d07ac7afaa9e3a46d1e407a315b679c4782b07cf2e7099a57ca64971b9e43", + "mm_token_type_ids": "b9cced2b89f7bc1a3951b6e33c8f64a4ef0948c4010fe75f415e1d9acdbef42c", + "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", + "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", + "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", + "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" + } + }, + { + "case": "wide", + "question": "next", + "image_size_wh": [ + 640, + 320 + ], + "prompt_tokens": 355, + "patches": 800, + "image_tokens": 200, + "choice": "save", + "tensor_sha256": { + "image_features": "2a61a3942c1ae3da659af50774f8df2d9c54fe378d8d6af9b4bff8556c827cac", + "inputs_embeds": "c2798ff2878c65703b72f602dfc3c02bddaf4cf7e052ce9cc343a4712f56b794", + "position_ids": "884820e1bdbb6d8430c45dd5e9736a84daa78c7f8687d7cd066e2696d42b8748", + "last_hidden_state": "0f52c02be99a39235599db4d07a87d6d7279313fd50d768b68af28e09ddb6386", + "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", + "candidate_logits": "d4dfc2bbd444653083bd5da94cf1ec3aa05ee0e8a58b86dfda59a952fed459e1", + "candidate_probabilities": "db172e375ef8b73e81e1634bb651703282ffd2461adfeb3d4a2a3fbb54bc2295", + "input_ids": "1a4f08196defb2d7c6978066e62172ce4cffd79261814028ee690b662a0cf4d1", + "attention_mask": "0a9895710d7a304bd8fcb401caa711bcfa5cffd2dc60effae67ec591ca5cdabd", + "mm_token_type_ids": "6b62dbc92ec3e227317458df9e81142df1239943eed4bf615a14d342ae1dd4e1", + "pixel_values": "e97e8e954e053f7bfe80988638b38a3de84d1900728c457c7021e8e36cba4dff", + "image_grid_thw": "dec6e7d9d02c96647505413493ba0829a487fc6fd7faaedb1677f8ea61bd3b0d", + "rope_deltas": "0aeec9fa170cb9adb0bafac34bac3430d494f79e0461520c9ca98bce2da6e9ed", + "image_token_indices": "cbb735a43a94e383737745b05d3304a1946232052a7f23fbf95001388d8674ab" + } + }, + { + "case": "portrait", + "question": "next", + "image_size_wh": [ + 320, + 640 + ], + "prompt_tokens": 355, + "patches": 800, + "image_tokens": 200, + "choice": "save", + "tensor_sha256": { + "image_features": "7576dc0965b6dbfc1e70681292be52239e474b4da39348e430c20425d5c610df", + "inputs_embeds": "7f20690d8364ee699f5d2b013b7255de1aab0f732b562128e60557003deee9e0", + "position_ids": "a1bc3878c5e02ae0fdec46bc2c67ae283b0e800e003abdc58ca465a745ea2b2c", + "last_hidden_state": "a74a601724461343b9da0d4e9fdcc95d987291fc369a77a4f89147904b4fd01b", + "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", + "candidate_logits": "ac44163a1f33f5b8bdf49c949477867ba5729ee8b9a7a40e47d34e62f2d9b7d0", + "candidate_probabilities": "fde4386614042b0c3bd2686796d13c754951cb996f56096a68e31df8f4733c25", + "input_ids": "1a4f08196defb2d7c6978066e62172ce4cffd79261814028ee690b662a0cf4d1", + "attention_mask": "0a9895710d7a304bd8fcb401caa711bcfa5cffd2dc60effae67ec591ca5cdabd", + "mm_token_type_ids": "6b62dbc92ec3e227317458df9e81142df1239943eed4bf615a14d342ae1dd4e1", + "pixel_values": "34764f3da7a09dae7003ee88ab200296693ec0d4f3791c0380fd08bf2b6f21f6", + "image_grid_thw": "f781bf73a563dadd6c35c03c0d07e3ef167084c7198eda620d9a5f2c6c2672b0", + "rope_deltas": "0aeec9fa170cb9adb0bafac34bac3430d494f79e0461520c9ca98bce2da6e9ed", + "image_token_indices": "cbb735a43a94e383737745b05d3304a1946232052a7f23fbf95001388d8674ab" + } + }, + { + "case": "jpeg", + "question": "next", + "image_size_wh": [ + 640, + 480 + ], + "prompt_tokens": 455, + "patches": 1200, + "image_tokens": 300, + "choice": "save", + "tensor_sha256": { + "image_features": "8110266107776e3a68e1270c696800cb0fbe54324e29b1e063cf99d99e07f7d1", + "inputs_embeds": "78e300616c864ac7d5d169e584257fbe61085a92abaec350c803cd84a6bad633", + "position_ids": "c4455b06669d378155a252796d98864fff869e59d7c369f873a4032802bbc845", + "last_hidden_state": "4e98f5b4529bb2c460be64c5b6564e6515a4eba40356e3676e52d821baa4a2f9", + "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", + "candidate_logits": "73887875f2f3d1d483d38649473ed99b836d9e7a5876eb196f04c533a6b1b3c5", + "candidate_probabilities": "f3f70b0f4c2da5fb277bd1dca5819d733541619405ae69b2451912beb7b438db", + "input_ids": "3a1cbc0fa6773db9ff9197682f3bb7ad9cfceb2246ae6b7e914d0dd920fb9be8", + "attention_mask": "b3a444dc9afd4f01b52288183c330714bb9d91bf7e6826d41506a94d008d33be", + "mm_token_type_ids": "4f71def3dd6649b5d23f6686470e003f056b91c53c1e44525846408acbcb01b4", + "pixel_values": "9fda7dd06d02e0ab99b8773f19967027da315168e33e879956816550f86c3a6a", + "image_grid_thw": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e", + "rope_deltas": "2e2a6247a269325c7f8a545ac50011fdffb5752820c5059993d2adc3a4db2fae", + "image_token_indices": "ea9df864fd6cd15d8cbd525c9fec290a8b386419da93b8897cdc8b99a720dcaf" + } + }, + { + "case": "single-option", + "question": "next", + "image_size_wh": [ + 320, + 240 + ], + "prompt_tokens": 216, + "patches": 320, + "image_tokens": 80, + "choice": "save", + "tensor_sha256": { + "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", + "inputs_embeds": "d22355d1127e40d8f8b4c703a70aa11216676187d4e8066cc89b5fed53c6653b", + "position_ids": "d2af297496cb55a053629fe3f11a2b2fbc3e7acd9d7fa0872a4f130cf4a36efa", + "last_hidden_state": "bf7f36eda00adc4019dadefbcdc45e9f2c220e2f33122bd4bffb147d2d2a056a", + "candidate_token_ids": "9d4ac218fb54041e3a70a8e14db1ea1af9f570f4842b4702ef9323f1ad8f0ec4", + "candidate_logits": "c156769c9825ace28120add7875f05a82356739495cf5e77f3085f5e0e699ef9", + "candidate_probabilities": "e00e5eb9444182f352323374ef4e08ebcb784725fdd4fd612d7730540b3e0c8c", + "input_ids": "54c997a6bd3c87420e06814199bdd4c6250e8bd7a119905dc6b70bb9547cb275", + "attention_mask": "f2a58a8823b814fb09efb464a7cbd713a267a0330a3950d4c30794b042cdbbf3", + "mm_token_type_ids": "60264dae568d914f055d42dadf102ac26543bfb3c59c16c6ffcc6b4537556382", + "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", + "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", + "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", + "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" + } + }, + { + "case": "26-options", + "question": "next", + "image_size_wh": [ + 256, + 256 + ], + "prompt_tokens": 518, + "patches": 256, + "image_tokens": 64, + "choice": "option-0", + "tensor_sha256": { + "image_features": "27776425b299d2365ccb7c85a0ac978437330bb768870e7d5eccd9049aebe75c", + "inputs_embeds": "e5aae2f7a83de111de48d21ae88d88ffdf762ad7870006d5a764a179ee2755a5", + "position_ids": "f5d23ba88480091b2a48b402a09649426935e9c9705da3a193ff863608c50e52", + "last_hidden_state": "08fd4d02c72c6d8a14d8ac25907da2b694d65dbbdbc5138146585d3215cbbee1", + "candidate_token_ids": "3d8eb6eccd7f0b19cdf184a00d3a66c055c51144fd1fca423947c4e6ba8e0ffb", + "candidate_logits": "ec389d03d0dfd6aae31d453339ec39adc99958a7f35bc560eaad72e906451c8f", + "candidate_probabilities": "93ffe78005decc4c142827283b7f8f77c3a91f981657f3bc51b8c2a3da76e9a7", + "input_ids": "4ff5dc819a82ea350fdbaac5614d02dd0c5fcb598ec254d2509e32d75564c8ec", + "attention_mask": "897e287da60216749858043c3f41203172b7b95e03f529598c64d6b4f7de8fd5", + "mm_token_type_ids": "1ec7514267935156e9ece070c26c9ba76a41f897edf6338fce634774dadf735b", + "pixel_values": "e4bc5f6a335986e8f845008a409c8e30106b42351f3dbeef508671abfd04d87c", + "image_grid_thw": "53389fa912f56f81bbbe0669ab378c489b4cfaf9501876dcaec451ae86db735b", + "rope_deltas": "57e3269d7d7589d5903723ce291461fcabd8c9b96180084d0c8dee6f8830a977", + "image_token_indices": "54bdbec761821f053fad9eab88372b9a5224cb34de6693e73364ea8e0f4612a1" + } + }, + { + "case": "two-questions", + "question": "next", + "image_size_wh": [ + 320, + 240 + ], + "prompt_tokens": 235, + "patches": 320, + "image_tokens": 80, + "choice": "save", + "tensor_sha256": { + "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", + "inputs_embeds": "7761598e9b268125c2da4519a3a1a29724b4529f8b304413f80d43b516e98cf8", + "position_ids": "5f2b2d115b061ae076d3e08a6c6d1cb80e55afda4ae9d1e533d398d73c830449", + "last_hidden_state": "09d0e1519397478402144e773fc68e0dc23ca74b1be5ba8e06b8e9e3fc84016d", + "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", + "candidate_logits": "0f43844e3ae275526d3a659484f1d2ffad9fd23a1896b3717069ecef23679f16", + "candidate_probabilities": "b98be35be7e3291eab531ab69fbec351e175f6d99c009650dc81ae713d1c7f68", + "input_ids": "d6313d4f14a714ce7017e1c18c0c3c6332d65662c0662f327f3d646bde794aed", + "attention_mask": "f72d07ac7afaa9e3a46d1e407a315b679c4782b07cf2e7099a57ca64971b9e43", + "mm_token_type_ids": "b9cced2b89f7bc1a3951b6e33c8f64a4ef0948c4010fe75f415e1d9acdbef42c", + "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", + "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", + "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", + "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" + } + }, + { + "case": "two-questions", + "question": "second", + "image_size_wh": [ + 320, + 240 + ], + "prompt_tokens": 229, + "patches": 320, + "image_tokens": 80, + "choice": "continue", + "tensor_sha256": { + "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", + "inputs_embeds": "c64343cbde260b2137678280ada592565424642328e4c1f68998c66d70e74613", + "position_ids": "ee8ee859bb4c6fc19bbb21afb5858bc34f3c77cc639e6b92b2667b4a9d68850f", + "last_hidden_state": "3990053f803cee4dcf6bfb2f0f720e9626a4293e16c7aa2f1edbcc290ebf5839", + "candidate_token_ids": "69f367cf75d7888b5f7dd9a1742839d327c2e189173b60a39197d55f338f9ab7", + "candidate_logits": "04a5adbc4b329516181953dc4d3bec78d9d7373e776fa959770698a943e43832", + "candidate_probabilities": "7309d5948eec9911a1c24ad1306f3a44741bd7b4ccdc8efcec716a3b3f281d74", + "input_ids": "0314c4622ca5ef8f0423172fe100739b7da8b8a2a0dfaed669c67daf93175b24", + "attention_mask": "1d0fecdc1aef66d694d1a03853d4d98ad3e43dc08cf99a6f77b466737b679eaf", + "mm_token_type_ids": "0f510d2f44a832df16a9ccd4f19cf61918614f5730950e0c8fe62993583f97ec", + "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", + "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", + "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", + "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" + } + } + ], + "verification_source_sha256": "107fb249e556acf798e756d5e9e553feda03dfe3e1caaef3854075d64dcb3997" +} diff --git a/recipe/cua_s1/reference.md b/recipe/cua_s1/reference.md new file mode 100644 index 0000000..d5615f0 --- /dev/null +++ b/recipe/cua_s1/reference.md @@ -0,0 +1,154 @@ +# Cua-S1 multimodal reference tensors + +This offline recipe observes the Transformers/PEFT implementation merged in +[#12](https://github.com/ThinkFlowLab/system1-omni/pull/12). It provides concrete +inputs for the native vision integration discussed in +[#10](https://github.com/ThinkFlowLab/system1-omni/issues/10#issuecomment-5862276038). +The adapter stays **unmerged**, including the vision LoRA. It does not change the +worker or add a native vision implementation. + +## Reproduce + +Use Python 3.12 and an NVIDIA GPU with enough memory for Qwen3.5-4B and the full +reference output head. The validated setup is an RTX 4090 (24 GiB), driver +595.71.05, Torch 2.14.0+cu130 and the packages in +[requirements-reference.txt](requirements-reference.txt). Use a fresh environment +without flash-linear-attention, causal-conv1d or explicitly enabled hub kernels. +The model chooses its default attention implementation; the manifest records the +actual text and vision choices. These are reference exports, not timed benchmarks. + +From the repository root: + +```sh +python3.12 -m venv .venv +.venv/bin/python -m pip install torch==2.14.0 torchvision==0.29.0 \ + --index-url https://download.pytorch.org/whl/cu130 +.venv/bin/python -m pip install -r recipe/cua_s1/requirements-reference.txt + +git clone https://github.com/trycua/cua.git /tmp/cua-reference +git -C /tmp/cua-reference checkout 0e75660ce4c2edda519e0c795fa3ad98abf4e76f +.venv/bin/python /tmp/cua-reference/libs/cua-s1/ci/fetch_pinned_weights.py \ + --dest /tmp/cua-weights +cp /tmp/cua-reference/libs/cua-s1/ci/weights.lock.json /tmp/cua-weights/weights.lock.json + +PYTHONPATH=src HF_HUB_OFFLINE=1 TOKENIZERS_PARALLELISM=false \ + .venv/bin/python recipe/cua_s1/export_multimodal_reference.py \ + --weights /tmp/cua-weights --output /tmp/cua-reference-a +PYTHONPATH=src HF_HUB_OFFLINE=1 TOKENIZERS_PARALLELISM=false \ + .venv/bin/python recipe/cua_s1/export_multimodal_reference.py \ + --weights /tmp/cua-weights --output /tmp/cua-reference-b +PYTHONPATH=src .venv/bin/python recipe/cua_s1/verify_multimodal_reference.py \ + /tmp/cua-reference-a --compare /tmp/cua-reference-b +``` + +The model verifies the upstream manifest's pinned SHA-256 and the size/hash of +every loaded base/adapter file before loading. For existing verified weights, +provide the same directory layout and the upstream manifest next to `Qwen3.5-4B/`. +The model pins are recorded in [the model contract](../../src/models/cua_s1/README.md#pinned-revisions). + +Each output must be a new directory. An existing output is rejected before model +loading. `manifest.json` is written last; a directory without it is an incomplete +export. Save generated bundles outside the checkout. A bundle is about 44 MiB on +the validated setup and contains no checkpoint weights. + +## Fixed inputs + +The recipe generates redistributable synthetic account-settings screens using +Pillow; it needs no user screenshots or external images. The image bytes and the +full inline-image request JSON are included in `inputs/`. + +| Case | Original image, width × height | Format | Questions / options | +| --- | --- | --- | --- | +| small | 320 × 240 | PNG | 1 / 3 | +| wide | 640 × 320 | PNG | 1 / 3 | +| portrait | 320 × 640 | PNG | 1 / 3 | +| jpeg | 640 × 480 | JPEG | 1 / 3 | +| single-option | 320 × 240 | PNG | 1 / 1 | +| 26-options | 256 × 256 | PNG | 1 / 26 | +| two-questions | 320 × 240 | PNG | 2 / 3 and 2 | + +The second question includes structured, non-ASCII instructions and an object and +null criterion. Each question is an independent unpadded forward over the same +request image, preserving the #12 behavior and option order. Identical image bytes +must give identical preprocessing and adapted vision features across questions. + +## Bundle format and native boundary + +`manifest.json` uses schema `cua-s1-multimodal-reference-v1`. It records pinned +revisions, verified weight-manifest hash, package/GPU/CUDA/driver information, +source hashes, execution configuration, file sizes and SHA-256s. Each ordered +question entry contains its prompt, request/image paths, option keys, final answer, +and a safetensors file. Every tensor has its shape, dtype and SHA-256 of contiguous +CPU bytes, with no dtype conversion. BF16 bytes are retained directly. + +The bundle also includes base, processor and adapter JSON configurations in +`configs/`. Tensor dimensions below use `S` for prompt length, `P = T×H×W` for raw +patch count and `I = P / merge_size²` for merged image-token count. In this pinned +model, hidden width `D = 2560`, patch size 16, temporal patch size 2 and merge size 2. +A still image has `T = 1`; the processor duplicates its frame for temporal patches. + +| Tensor | Shape | Dtype | Meaning | +| --- | --- | --- | --- | +| `pixel_values` | `[P, 1536]` | float32 | Processor-normalized, flattened image patches; not an RGB image or NCHW tensor | +| `image_grid_thw` | `[1, 3]` | int64 | Raw temporal/height/width patch grid before spatial merging | +| `input_ids` | `[1, S]` | int64 | Fully expanded chat prompt, including `I` image placeholders | +| `attention_mask` | `[1, S]` | int64 | All ones: one unpadded prompt | +| `mm_token_type_ids` | `[1, S]` | int64 | Text 0, image 1 | +| `image_features` | `[I, D]` | bfloat16 | Actual adapted vision output after the merger, before insertion into the language input | +| `image_token_indices` | `[I]` | int64 | Positions where the image features replace token embeddings, in sequence order | +| `inputs_embeds` | `[1, S, D]` | bfloat16 | Actual language-model input after image insertion | +| `position_ids` | `[3, 1, S]` | int64 | Actual temporal/height/width M-RoPE positions at the language-model boundary | +| `rope_deltas` | `[1, 1]` | int64 | `max(position_ids) + 1 - S`, recorded from the model | +| `last_hidden_state` | `[1, D]` | bfloat16 | Final normalized language hidden state at the last sequence position | +| `candidate_token_ids` | `[C]` | int64 | A–Z letter ids, aligned with `option_keys` | +| `candidate_logits` | `[C]` | bfloat16 | Final-position logits for the candidates | +| `candidate_probabilities` | `[C]` | float32 | GPU fp32 softmax of the candidate logits | + +Temporary hooks observe the **actual** PEFT forward, including the vision merger +and the language input/last hidden state. They return no replacement outputs and +are removed in `finally`, including on failure. Position ids come from the model's +forward, rather than a separately reimplemented M-RoPE algorithm. Each captured +readout is checked against a second call to the ordinary #12 `score()` method; +the probabilities must match exactly before an export is marked complete. + +For a native language-path test, use `inputs_embeds`, `position_ids`, the base text +configuration and the multimodal-adapted language weights. The text adapter from +#19 is a different adapter. For a native vision test, start from `pixel_values` and +`image_grid_thw` and compare with `image_features`. The exported embedding contains +the unmerged reference adapter's BF16 rounding behavior; merging LoRA may change +that behavior and needs its own declared tolerance. + +## Verification and limits + +The verifier checks file and tensor fingerprints, exact keys/shapes/dtypes, +finite values, patch/merge counts, image-placeholder ordering, feature insertion, +position delta, candidate ordering, answer reconstruction and same-image reuse. +CPU softmax reconstruction allows absolute error `1e-7` with zero relative +allowance because CPU and GPU reduction implementations can round differently. +`--compare` additionally requires the two manifests to match exactly, including +all file and tensor hashes, execution settings and source/environment metadata. +The safetensors header key order is not part of a portable format guarantee. + +The checked [validation summary](reference-validation.json) records two +independent exports on the RTX 4090: 8 questions, 112 tensors, 25 files per bundle, +identical raw tensor bytes and identical ordinary/captured probabilities. CPU +softmax reconstruction differed by at most `3.73e-9`. This verifies the reference +artifact pipeline; it makes no native accuracy or speed claim. Bitwise equality +has only been checked on the recorded software/GPU setup. Other GPUs, CUDA builds, +FP32 controls, video, multiple images, padding/batching and Metal are unverified. +When evaluating a new native implementation, declare numerical tolerances before +comparison; the `--compare` mode is for repeating the same reference export. + +Run the focused tests without weights or a GPU: + +```sh +python -m pip install torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu +python -m pip install Pillow==11.3.0 numpy==2.5.3 safetensors==0.8.0 pytest==9.1.1 +PYTHONPATH=src python -m pytest tests/cua_s1/test_export_multimodal_reference.py -q +``` + +Tests cover deterministic cases, existing-output rejection, raw BF16 bytes, +actual hook capture and cleanup on success/failure, safetensors round trips, +corruption/unlisted-file detection, safe paths, missing tensor rejection and CPU +softmax roundoff versus drift. Without Torch, the two image/output tests run and tensor tests are skipped; +the recipe CI installs CPU Torch and executes the complete suite. diff --git a/recipe/cua_s1/requirements-reference.txt b/recipe/cua_s1/requirements-reference.txt new file mode 100644 index 0000000..e2a0b84 --- /dev/null +++ b/recipe/cua_s1/requirements-reference.txt @@ -0,0 +1,13 @@ +# Reference environment from the pinned trycua/cua four-b lock. +# Install the CUDA Torch/Torchvision wheels first, as described in reference.md. +torch==2.14.0 +torchvision==0.29.0 +transformers==5.17.0 +peft==0.21.0 +accelerate==1.15.0 +Pillow==11.3.0 +safetensors==0.8.0 +huggingface-hub==1.32.0 +tokenizers==0.23.2 +# NumPy is used only for raw tensor-byte serialization. +numpy==2.5.3 diff --git a/recipe/cua_s1/verify_multimodal_reference.py b/recipe/cua_s1/verify_multimodal_reference.py new file mode 100644 index 0000000..0f9a7a2 --- /dev/null +++ b/recipe/cua_s1/verify_multimodal_reference.py @@ -0,0 +1,239 @@ +"""Verify export integrity and multimodal tensor relations; optionally compare runs.""" + +from __future__ import annotations + +import argparse +import json +from pathlib import Path + +import torch +from export_multimodal_reference import SCHEMA, sha256, tensor_info +from safetensors.torch import load_file + +from models.cua_s1.multimodal.protocol import answer, decode_request, parse_request + +REQUIRED = { + "input_ids", + "attention_mask", + "mm_token_type_ids", + "pixel_values", + "image_grid_thw", + "image_features", + "image_token_indices", + "inputs_embeds", + "position_ids", + "rope_deltas", + "last_hidden_state", + "candidate_token_ids", + "candidate_logits", + "candidate_probabilities", +} + + +def require(condition, message): + if not condition: + raise ValueError(message) + + +def check_files(folder, manifest): + for name, info in manifest["files"].items(): + path = folder / name + require( + path.resolve().is_relative_to(folder.resolve()), f"unsafe file path: {name}" + ) + require(path.is_file(), f"missing file: {name}") + require( + path.stat().st_size == info["size"] + and sha256(path.read_bytes()) == info["sha256"], + f"file checksum: {name}", + ) + + actual = { + path.relative_to(folder).as_posix() + for path in folder.rglob("*") + if path.is_file() and path != folder / "manifest.json" + } + require(actual == set(manifest["files"]), "file inventory mismatch") + + +def check_tensors(tensors, entry, configs): + require( + set(tensors) == REQUIRED, + f"missing tensors or unknown keys: {set(tensors) ^ REQUIRED}", + ) + require( + {name: tensor_info(value) for name, value in tensors.items()} + == entry["tensors"], + "tensor fingerprint mismatch", + ) + ids = tensors["input_ids"] + sequence = ids.shape[-1] + hidden = configs["base"]["text_config"]["hidden_size"] + vision = configs["base"]["vision_config"] + merge = vision["spatial_merge_size"] + patch = vision["patch_size"] + temporal = vision["temporal_patch_size"] + grid = tensors["image_grid_thw"] + require( + grid.shape == (1, 3) and grid[0, 0].item() == 1, "expected one still-image grid" + ) + require( + grid[0, 1].item() % merge == grid[0, 2].item() % merge == 0, + "grid merge alignment", + ) + image_tokens = grid.prod().item() // merge**2 + expected_indices = (ids[0] == configs["base"]["image_token_id"]).nonzero().flatten() + shapes = { + "input_ids": (1, sequence), + "attention_mask": (1, sequence), + "mm_token_type_ids": (1, sequence), + "pixel_values": (grid.prod().item(), 3 * temporal * patch**2), + "image_features": (image_tokens, hidden), + "image_token_indices": (image_tokens,), + "inputs_embeds": (1, sequence, hidden), + "position_ids": (3, 1, sequence), + "rope_deltas": (1, 1), + "last_hidden_state": (1, hidden), + "candidate_token_ids": (len(entry["option_keys"]),), + "candidate_logits": (len(entry["option_keys"]),), + "candidate_probabilities": (len(entry["option_keys"]),), + } + for name, shape in shapes.items(): + value = tensors[name] + require(tuple(value.shape) == shape, f"shape mismatch: {name}") + require(bool(torch.isfinite(value).all()), f"nonfinite tensor: {name}") + for name in [ + "input_ids", + "attention_mask", + "mm_token_type_ids", + "image_grid_thw", + "image_token_indices", + "position_ids", + "rope_deltas", + "candidate_token_ids", + ]: + require(tensors[name].dtype == torch.int64, f"expected int64: {name}") + for name in [ + "image_features", + "inputs_embeds", + "last_hidden_state", + "candidate_logits", + ]: + require(tensors[name].dtype == torch.bfloat16, f"expected bfloat16: {name}") + require( + tensors["pixel_values"].dtype + == tensors["candidate_probabilities"].dtype + == torch.float32, + "expected float32 pixels/probabilities", + ) + require(bool((tensors["attention_mask"] == 1).all()), "expected unpadded prompt") + require( + torch.equal(expected_indices, tensors["image_token_indices"]), + "image token indices mismatch", + ) + require( + torch.equal( + expected_indices, (tensors["mm_token_type_ids"][0] == 1).nonzero().flatten() + ), + "image token types mismatch", + ) + require( + torch.equal( + tensors["inputs_embeds"][0, expected_indices], tensors["image_features"] + ), + "image insertion mismatch", + ) + require( + tensors["rope_deltas"].item() + == tensors["position_ids"].max().item() + 1 - sequence, + "rope delta mismatch", + ) + require( + torch.equal( + tensors["candidate_token_ids"], + torch.arange(32, 32 + len(entry["option_keys"])), + ), + "candidate token ordering", + ) + require( + torch.allclose( + torch.softmax(tensors["candidate_logits"].float(), dim=-1), + tensors["candidate_probabilities"], + atol=1e-7, + rtol=0, + ), + "readout mismatch", + ) + + +def verify(folder): + manifest = json.loads((folder / "manifest.json").read_text()) + require(manifest["schema"] == SCHEMA, "unsupported schema") + check_files(folder, manifest) + configs = { + name: json.loads((folder / f"configs/{name}.json").read_text()) + for name in ["base", "processor", "adapter"] + } + seen_images = {} + count = 0 + for entry in manifest["questions"]: + require(entry["tensors_file"] in manifest["files"], "unlisted tensor file") + require( + entry["request"] in manifest["files"] + and entry["image"] in manifest["files"], + "unlisted input file", + ) + tensors = load_file(str(folder / entry["tensors_file"])) + check_tensors(tensors, entry, configs) + request = parse_request( + decode_request((folder / entry["request"]).read_bytes()) + ) + question = next(q for q in request.questions if q.name == entry["question"]) + require( + list(request.image.size) == entry["image_size_wh"], "image size mismatch" + ) + require(list(question.keys) == entry["option_keys"], "option ordering mismatch") + require( + answer(question, tensors["candidate_probabilities"].tolist()) + == entry["answer"], + "answer mismatch", + ) + require(entry["ordinary_score_equal"] is True, "ordinary score check missing") + image = manifest["files"][entry["image"]]["sha256"] + features = { + name: entry["tensors"][name] + for name in ["pixel_values", "image_grid_thw", "image_features"] + } + require( + image not in seen_images or seen_images[image] == features, + "same-image vision mismatch", + ) + seen_images[image] = features + count += len(tensors) + require(len(manifest["questions"]) == 8, "expected eight questions") + return manifest, { + "questions": 8, + "tensors": count, + "files": len(manifest["files"]), + "integrity_and_relations": "pass", + } + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("bundle", type=Path) + parser.add_argument("--compare", type=Path) + args = parser.parse_args() + manifest, summary = verify(args.bundle) + if args.compare: + other, _ = verify(args.compare) + require( + manifest == other, + "exports differ (environment, metadata, files or tensors)", + ) + summary["independent_export_equality"] = "pass" + print(json.dumps(summary, indent=2)) + + +if __name__ == "__main__": + main() diff --git a/tests/cua_s1/test_export_multimodal_reference.py b/tests/cua_s1/test_export_multimodal_reference.py new file mode 100644 index 0000000..e165ea8 --- /dev/null +++ b/tests/cua_s1/test_export_multimodal_reference.py @@ -0,0 +1,260 @@ +"""CPU checks for the offline reference export; tensor checks need optional Torch.""" + +import base64 +import importlib.util +import json +from pathlib import Path +from types import SimpleNamespace + +import pytest + +ROOT = Path(__file__).resolve().parents[2] +SPEC = importlib.util.spec_from_file_location( + "export_multimodal_reference", ROOT / "recipe/cua_s1/export_multimodal_reference.py" +) + + +def recipe(): + module = importlib.util.module_from_spec(SPEC) + SPEC.loader.exec_module(module) + return module + + +def test_cases_reproduce_images_and_requests(tmp_path): + module = recipe() + left = module.make_cases(tmp_path / "left") + right = module.make_cases(tmp_path / "right") + assert len(left) == len(right) == 7 + assert sum(len(case["request"]["questions"]) for case in left) == 8 + for a, b in zip(left, right): + assert a["name"] == b["name"] + assert a["request"] == b["request"] + image = base64.b64decode(a["request"]["state"]["image"].split(",")[1]) + assert (tmp_path / "left" / a["image"]).read_bytes() == image + assert (tmp_path / "right" / b["image"]).read_bytes() == image + from models.cua_s1.multimodal.protocol import parse_request + + requests = [parse_request(case["request"]) for case in left] + assert len({request.image.size for request in requests}) >= 4 + assert {len(q.keys) for request in requests for q in request.questions} >= { + 1, + 3, + 26, + } + + +def test_existing_output_is_rejected_before_model_load(tmp_path): + module = recipe() + with pytest.raises(FileExistsError): + module.export(tmp_path / "missing-weights", tmp_path) + assert not list(tmp_path.iterdir()) + + +def test_fingerprint_preserves_bfloat16_bytes(): + torch = pytest.importorskip("torch") + module = recipe() + tensor = torch.tensor([[1.0, -2.0]], dtype=torch.bfloat16) + info = module.tensor_info(tensor) + assert info["shape"] == [1, 2] + assert info["dtype"] == "bfloat16" + assert info["sha256"] == module.sha256(tensor.view(torch.uint8).numpy().tobytes()) + assert module.tensor_info(tensor.t().contiguous().t()) == info + + +def toy_engine(torch, fail=False): + class Visual(torch.nn.Module): + def forward(self, pixels, **kwargs): + return SimpleNamespace(pooler_output=pixels.to(torch.bfloat16)) + + class Language(torch.nn.Module): + def forward(self, inputs_embeds=None, position_ids=None, **kwargs): + if fail: + raise RuntimeError("deliberate forward failure") + return SimpleNamespace(last_hidden_state=inputs_embeds + 1) + + class Model(torch.nn.Module): + def __init__(self): + super().__init__() + self.model = torch.nn.Module() + self.model.visual = Visual() + self.model.language_model = Language() + self.model.rope_deltas = torch.tensor([[-1]]) + self.device = torch.device("cpu") + self.config = SimpleNamespace(image_token_id=99) + + def get_base_model(self): + return self + + def forward(self, input_ids, pixel_values, **kwargs): + features = self.model.visual(pixel_values).pooler_output + embeds = torch.zeros(1, input_ids.shape[1], 2, dtype=torch.bfloat16) + embeds[0, input_ids[0] == 99] = features + positions = torch.arange(input_ids.shape[1]).expand(3, 1, -1) + result = self.model.language_model( + inputs_embeds=embeds, position_ids=positions + ) + logits = result.last_hidden_state[..., :1].expand(-1, -1, 100).clone() + return SimpleNamespace(logits=logits) + + model = Model() + tokenizer = SimpleNamespace(encode=lambda letter, **kwargs: [ord(letter) - 33]) + return SimpleNamespace(model=model, tokenizer=tokenizer) + + +def test_capture_observes_forward_and_removes_hooks(): + torch = pytest.importorskip("torch") + module = recipe() + from models.cua_s1.multimodal.protocol import Question + + engine = toy_engine(torch) + inputs = { + "input_ids": torch.tensor([[1, 99, 99, 2]]), + "pixel_values": torch.tensor([[2.0, 3.0], [4.0, 5.0]]), + } + question = Question("test", ("a", "b"), ("First", "Second"), "Choose") + tensors = module.capture(engine, inputs, question) + assert torch.equal(tensors["inputs_embeds"][0, 1:3], tensors["image_features"]) + assert tensors["position_ids"].shape == (3, 1, 4) + assert tensors["last_hidden_state"].shape == (1, 2) + assert tensors["candidate_probabilities"].tolist() == [0.5, 0.5] + assert all(t.device.type == "cpu" for t in tensors.values()) + assert not engine.model.model.visual._forward_hooks + assert not engine.model.model.language_model._forward_pre_hooks + assert not engine.model.model.language_model._forward_hooks + + +def test_forward_failure_also_removes_hooks(): + torch = pytest.importorskip("torch") + module = recipe() + from models.cua_s1.multimodal.protocol import Question + + engine = toy_engine(torch, fail=True) + with pytest.raises(RuntimeError, match="deliberate forward failure"): + module.capture( + engine, + {"input_ids": torch.tensor([[99]]), "pixel_values": torch.ones(1, 2)}, + Question("test", ("a",), ("First",), "Choose"), + ) + assert not engine.model.model.visual._forward_hooks + assert not engine.model.model.language_model._forward_pre_hooks + assert not engine.model.model.language_model._forward_hooks + + +def test_bundle_roundtrip_and_independent_storage(tmp_path): + torch = pytest.importorskip("torch") + from safetensors.torch import load_file + + module = recipe() + original = torch.tensor([[1, 2]], dtype=torch.int64) + entries = module.save_tensors( + tmp_path / "q.safetensors", {"a": original, "b": original} + ) + loaded = load_file(str(tmp_path / "q.safetensors")) + assert torch.equal(loaded["a"], original) + assert entries["a"] == module.tensor_info(loaded["a"]) + assert entries["b"] == entries["a"] + json.dumps(entries, allow_nan=False) + + +def test_verifier_checks_files_and_rejects_corruption(tmp_path, monkeypatch): + torch = pytest.importorskip("torch") + monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) + import verify_multimodal_reference as verifier + + module = recipe() + bundle = tmp_path / "bundle" + bundle.mkdir() + tensor_path = bundle / "q.safetensors" + metadata = module.save_tensors(tensor_path, {"example": torch.ones(2)}) + manifest = { + "schema": module.SCHEMA, + "files": { + "q.safetensors": { + "size": tensor_path.stat().st_size, + "sha256": module.sha256(tensor_path.read_bytes()), + } + }, + "questions": [{"tensors_file": "q.safetensors", "tensors": metadata}], + } + module.write_json(bundle / "manifest.json", manifest) + # File integrity is independently testable before semantic validation. + verifier.check_files(bundle, manifest) + tensor_path.write_bytes(tensor_path.read_bytes() + b"corrupt") + with pytest.raises(ValueError, match="file checksum"): + verifier.check_files(bundle, manifest) + + +def test_verifier_does_not_accept_missing_tensors(tmp_path, monkeypatch): + pytest.importorskip("torch") + monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) + import verify_multimodal_reference as verifier + + with pytest.raises(ValueError, match="missing tensors"): + verifier.check_tensors({}, {}, {}) + + +def test_readout_allows_cpu_roundoff_but_rejects_drift(monkeypatch): + torch = pytest.importorskip("torch") + monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) + import verify_multimodal_reference as verifier + + module = recipe() + tensors = { + "input_ids": torch.tensor([[1, 99, 99, 2]]), + "attention_mask": torch.ones(1, 4, dtype=torch.int64), + "mm_token_type_ids": torch.tensor([[0, 1, 1, 0]]), + "pixel_values": torch.ones(2, 3), + "image_grid_thw": torch.tensor([[1, 1, 2]]), + "image_features": torch.ones(2, 2, dtype=torch.bfloat16), + "image_token_indices": torch.tensor([1, 2]), + "inputs_embeds": torch.ones(1, 4, 2, dtype=torch.bfloat16), + "position_ids": torch.arange(4).expand(3, 1, -1), + "rope_deltas": torch.tensor([[0]]), + "last_hidden_state": torch.ones(1, 2, dtype=torch.bfloat16), + "candidate_token_ids": torch.tensor([32, 33]), + "candidate_logits": torch.tensor([0, 1], dtype=torch.bfloat16), + "candidate_probabilities": torch.softmax(torch.tensor([0.0, 1.0]), dim=-1), + } + tensors["candidate_probabilities"][0] += 3e-8 + configs = { + "base": { + "text_config": {"hidden_size": 2}, + "vision_config": { + "spatial_merge_size": 1, + "patch_size": 1, + "temporal_patch_size": 1, + }, + "image_token_id": 99, + } + } + entry = { + "option_keys": ["a", "b"], + "tensors": {k: module.tensor_info(v) for k, v in tensors.items()}, + } + verifier.check_tensors(tensors, entry, configs) + tensors["candidate_probabilities"][0] += 1e-4 + entry["tensors"] = {k: module.tensor_info(v) for k, v in tensors.items()} + with pytest.raises(ValueError, match="readout mismatch"): + verifier.check_tensors(tensors, entry, configs) + + +@pytest.mark.parametrize("unsafe", [False, True]) +def test_verifier_rejects_unlisted_files_and_unsafe_paths( + tmp_path, monkeypatch, unsafe +): + pytest.importorskip("torch") + monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) + import verify_multimodal_reference as verifier + + module = recipe() + folder = tmp_path / "bundle" + folder.mkdir() + extra = (tmp_path if unsafe else folder) / "extra.json" + extra.write_text("{}") + manifest = {"files": {}} + if unsafe: + manifest["files"]["../extra.json"] = {"size": 2, "sha256": module.sha256(b"{}")} + with pytest.raises( + ValueError, match="unsafe file path" if unsafe else "file inventory" + ): + verifier.check_files(folder, manifest) From 1b64fa2ceb0a82b6a66a69ecdc9bc5cc1b1a0b66 Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Wed, 30 Sep 2026 21:16:08 +0800 Subject: [PATCH 2/2] Keep PR limited to the reference export and verification code --- .github/workflows/cua-s1-reference.yml | 31 -- recipe/README.md | 3 - recipe/cua_s1/reference-validation.json | 294 ------------------ recipe/cua_s1/reference.md | 154 --------- recipe/cua_s1/requirements-reference.txt | 13 - .../test_export_multimodal_reference.py | 260 ---------------- 6 files changed, 755 deletions(-) delete mode 100644 .github/workflows/cua-s1-reference.yml delete mode 100644 recipe/cua_s1/reference-validation.json delete mode 100644 recipe/cua_s1/reference.md delete mode 100644 recipe/cua_s1/requirements-reference.txt delete mode 100644 tests/cua_s1/test_export_multimodal_reference.py diff --git a/.github/workflows/cua-s1-reference.yml b/.github/workflows/cua-s1-reference.yml deleted file mode 100644 index 4e59165..0000000 --- a/.github/workflows/cua-s1-reference.yml +++ /dev/null @@ -1,31 +0,0 @@ -name: Cua-S1 reference recipe -on: - pull_request: - paths: - - 'recipe/cua_s1/**' - - 'tests/cua_s1/test_export_multimodal_reference.py' - - 'src/models/cua_s1/multimodal/**' - - '.github/workflows/cua-s1-reference.yml' - push: - branches: [main] - paths: - - 'recipe/cua_s1/**' - - 'tests/cua_s1/test_export_multimodal_reference.py' - - 'src/models/cua_s1/multimodal/**' - - '.github/workflows/cua-s1-reference.yml' -permissions: - contents: read -jobs: - cpu: - runs-on: ubuntu-latest - timeout-minutes: 15 - steps: - - uses: actions/checkout@v5 - - uses: actions/setup-python@v5 - with: - python-version: '3.12' - - run: python -m pip install torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu - - run: python -m pip install Pillow==11.3.0 numpy==2.5.3 safetensors==0.8.0 pytest==9.1.1 ruff==0.16.8 - - run: PYTHONPATH=src python -m pytest tests/cua_s1/test_export_multimodal_reference.py -q - - run: ruff check --isolated --select E4,E7,E9,F,I recipe/cua_s1/export_multimodal_reference.py recipe/cua_s1/verify_multimodal_reference.py tests/cua_s1/test_export_multimodal_reference.py - - run: ruff format --isolated --check recipe/cua_s1/export_multimodal_reference.py recipe/cua_s1/verify_multimodal_reference.py tests/cua_s1/test_export_multimodal_reference.py diff --git a/recipe/README.md b/recipe/README.md index 05c2334..4d0bf6c 100644 --- a/recipe/README.md +++ b/recipe/README.md @@ -3,8 +3,5 @@ - [Laya text worker](laya/README.md): start the external Python worker, connect the Rust frontend and compare direct and proxied responses. -- [Cua-S1 multimodal reference tensors](cua_s1/reference.md): export fixed inputs - and actual vision/language boundary tensors for native integration. - Recipes contain setup, launch commands and examples. Reusable implementation code belongs under `src/`. diff --git a/recipe/cua_s1/reference-validation.json b/recipe/cua_s1/reference-validation.json deleted file mode 100644 index f85f09f..0000000 --- a/recipe/cua_s1/reference-validation.json +++ /dev/null @@ -1,294 +0,0 @@ -{ - "base_commit": "b50aa28", - "reference_revision": "0e75660ce4c2edda519e0c795fa3ad98abf4e76f", - "base_revision": "851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", - "adapter_revision": "16818868b0cc7813808aae4e87b417657046ab79", - "weights_manifest_sha256": "9820bd232c5762f114e19680c0f8203d7e1faaf8a60c196cfe01964d6d8a6c09", - "source_sha256": { - "recipe/cua_s1/export_multimodal_reference.py": "9ff84a0980d843f5015e319622533da4bdc9fd1ae74ed09be7e9dc4b08fa8cd2", - "src/models/cua_s1/multimodal/model.py": "e37a01feb72eaf60a9a3a36c0ddae7a78b466c6e1add5e2f1dc93c5b1947d2dc", - "src/models/cua_s1/multimodal/protocol.py": "eefc421dc485ab0a23d475a3a733b5d46f33cbe9febd6295e5d794e7fdfcec5f" - }, - "environment": { - "python": "3.12.3", - "torch_build": "2.14.0+cu130", - "torch_num_threads": 64, - "torch_num_interop_threads": 64, - "packages": { - "torch": "2.14.0", - "torchvision": "0.29.0", - "transformers": "5.17.0", - "peft": "0.21.0", - "accelerate": "1.15.0", - "Pillow": "11.3.0", - "safetensors": "0.8.0", - "huggingface-hub": "1.32.0", - "tokenizers": "0.23.2", - "numpy": "2.5.3" - }, - "cuda": "13.0", - "gpu": "NVIDIA GeForce RTX 4090", - "compute_capability": [ - 8, - 9 - ], - "driver": "595.71.05", - "transformers_source_sha256": "762feb6c7426a7f15b5bf830df54c07438bf9e7c27b8cdb23179045920412c3b", - "dtype": "bfloat16", - "adapter_merged": false, - "tf32": false, - "deterministic_algorithms": true, - "cublas_workspace_config": ":4096:8" - }, - "execution": { - "visual_attention": "sdpa", - "text_attention": "sdpa", - "processor_class": "Qwen3VLProcessor", - "image_processor_class": "Qwen2VLImageProcessor" - }, - "verification": { - "questions": 8, - "tensors": 112, - "files": 25, - "integrity_and_relations": "pass", - "independent_export_equality": "pass" - }, - "tests": { - "command": "PYTHONPATH=src python -m pytest tests/cua_s1/test_export_multimodal_reference.py -q", - "passed": 11, - "cuda_visible_devices": "" - }, - "ordinary_score_equality": { - "questions_per_run": 8, - "independent_runs": 2, - "comparison": "exact" - }, - "cpu_softmax_max_absolute_difference": 3.725290298461914e-09, - "questions": [ - { - "case": "small", - "question": "next", - "image_size_wh": [ - 320, - 240 - ], - "prompt_tokens": 235, - "patches": 320, - "image_tokens": 80, - "choice": "save", - "tensor_sha256": { - "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", - "inputs_embeds": "7761598e9b268125c2da4519a3a1a29724b4529f8b304413f80d43b516e98cf8", - "position_ids": "5f2b2d115b061ae076d3e08a6c6d1cb80e55afda4ae9d1e533d398d73c830449", - "last_hidden_state": "09d0e1519397478402144e773fc68e0dc23ca74b1be5ba8e06b8e9e3fc84016d", - "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", - "candidate_logits": "0f43844e3ae275526d3a659484f1d2ffad9fd23a1896b3717069ecef23679f16", - "candidate_probabilities": "b98be35be7e3291eab531ab69fbec351e175f6d99c009650dc81ae713d1c7f68", - "input_ids": "d6313d4f14a714ce7017e1c18c0c3c6332d65662c0662f327f3d646bde794aed", - "attention_mask": "f72d07ac7afaa9e3a46d1e407a315b679c4782b07cf2e7099a57ca64971b9e43", - "mm_token_type_ids": "b9cced2b89f7bc1a3951b6e33c8f64a4ef0948c4010fe75f415e1d9acdbef42c", - "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", - "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", - "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", - "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" - } - }, - { - "case": "wide", - "question": "next", - "image_size_wh": [ - 640, - 320 - ], - "prompt_tokens": 355, - "patches": 800, - "image_tokens": 200, - "choice": "save", - "tensor_sha256": { - "image_features": "2a61a3942c1ae3da659af50774f8df2d9c54fe378d8d6af9b4bff8556c827cac", - "inputs_embeds": "c2798ff2878c65703b72f602dfc3c02bddaf4cf7e052ce9cc343a4712f56b794", - "position_ids": "884820e1bdbb6d8430c45dd5e9736a84daa78c7f8687d7cd066e2696d42b8748", - "last_hidden_state": "0f52c02be99a39235599db4d07a87d6d7279313fd50d768b68af28e09ddb6386", - "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", - "candidate_logits": "d4dfc2bbd444653083bd5da94cf1ec3aa05ee0e8a58b86dfda59a952fed459e1", - "candidate_probabilities": "db172e375ef8b73e81e1634bb651703282ffd2461adfeb3d4a2a3fbb54bc2295", - "input_ids": "1a4f08196defb2d7c6978066e62172ce4cffd79261814028ee690b662a0cf4d1", - "attention_mask": "0a9895710d7a304bd8fcb401caa711bcfa5cffd2dc60effae67ec591ca5cdabd", - "mm_token_type_ids": "6b62dbc92ec3e227317458df9e81142df1239943eed4bf615a14d342ae1dd4e1", - "pixel_values": "e97e8e954e053f7bfe80988638b38a3de84d1900728c457c7021e8e36cba4dff", - "image_grid_thw": "dec6e7d9d02c96647505413493ba0829a487fc6fd7faaedb1677f8ea61bd3b0d", - "rope_deltas": "0aeec9fa170cb9adb0bafac34bac3430d494f79e0461520c9ca98bce2da6e9ed", - "image_token_indices": "cbb735a43a94e383737745b05d3304a1946232052a7f23fbf95001388d8674ab" - } - }, - { - "case": "portrait", - "question": "next", - "image_size_wh": [ - 320, - 640 - ], - "prompt_tokens": 355, - "patches": 800, - "image_tokens": 200, - "choice": "save", - "tensor_sha256": { - "image_features": "7576dc0965b6dbfc1e70681292be52239e474b4da39348e430c20425d5c610df", - "inputs_embeds": "7f20690d8364ee699f5d2b013b7255de1aab0f732b562128e60557003deee9e0", - "position_ids": "a1bc3878c5e02ae0fdec46bc2c67ae283b0e800e003abdc58ca465a745ea2b2c", - "last_hidden_state": "a74a601724461343b9da0d4e9fdcc95d987291fc369a77a4f89147904b4fd01b", - "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", - "candidate_logits": "ac44163a1f33f5b8bdf49c949477867ba5729ee8b9a7a40e47d34e62f2d9b7d0", - "candidate_probabilities": "fde4386614042b0c3bd2686796d13c754951cb996f56096a68e31df8f4733c25", - "input_ids": "1a4f08196defb2d7c6978066e62172ce4cffd79261814028ee690b662a0cf4d1", - "attention_mask": "0a9895710d7a304bd8fcb401caa711bcfa5cffd2dc60effae67ec591ca5cdabd", - "mm_token_type_ids": "6b62dbc92ec3e227317458df9e81142df1239943eed4bf615a14d342ae1dd4e1", - "pixel_values": "34764f3da7a09dae7003ee88ab200296693ec0d4f3791c0380fd08bf2b6f21f6", - "image_grid_thw": "f781bf73a563dadd6c35c03c0d07e3ef167084c7198eda620d9a5f2c6c2672b0", - "rope_deltas": "0aeec9fa170cb9adb0bafac34bac3430d494f79e0461520c9ca98bce2da6e9ed", - "image_token_indices": "cbb735a43a94e383737745b05d3304a1946232052a7f23fbf95001388d8674ab" - } - }, - { - "case": "jpeg", - "question": "next", - "image_size_wh": [ - 640, - 480 - ], - "prompt_tokens": 455, - "patches": 1200, - "image_tokens": 300, - "choice": "save", - "tensor_sha256": { - "image_features": "8110266107776e3a68e1270c696800cb0fbe54324e29b1e063cf99d99e07f7d1", - "inputs_embeds": "78e300616c864ac7d5d169e584257fbe61085a92abaec350c803cd84a6bad633", - "position_ids": "c4455b06669d378155a252796d98864fff869e59d7c369f873a4032802bbc845", - "last_hidden_state": "4e98f5b4529bb2c460be64c5b6564e6515a4eba40356e3676e52d821baa4a2f9", - "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", - "candidate_logits": "73887875f2f3d1d483d38649473ed99b836d9e7a5876eb196f04c533a6b1b3c5", - "candidate_probabilities": "f3f70b0f4c2da5fb277bd1dca5819d733541619405ae69b2451912beb7b438db", - "input_ids": "3a1cbc0fa6773db9ff9197682f3bb7ad9cfceb2246ae6b7e914d0dd920fb9be8", - "attention_mask": "b3a444dc9afd4f01b52288183c330714bb9d91bf7e6826d41506a94d008d33be", - "mm_token_type_ids": "4f71def3dd6649b5d23f6686470e003f056b91c53c1e44525846408acbcb01b4", - "pixel_values": "9fda7dd06d02e0ab99b8773f19967027da315168e33e879956816550f86c3a6a", - "image_grid_thw": "d57d75cd100b84d70668fa90c17881ed8d1fcb7bb199a216e050a284b48dfe0e", - "rope_deltas": "2e2a6247a269325c7f8a545ac50011fdffb5752820c5059993d2adc3a4db2fae", - "image_token_indices": "ea9df864fd6cd15d8cbd525c9fec290a8b386419da93b8897cdc8b99a720dcaf" - } - }, - { - "case": "single-option", - "question": "next", - "image_size_wh": [ - 320, - 240 - ], - "prompt_tokens": 216, - "patches": 320, - "image_tokens": 80, - "choice": "save", - "tensor_sha256": { - "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", - "inputs_embeds": "d22355d1127e40d8f8b4c703a70aa11216676187d4e8066cc89b5fed53c6653b", - "position_ids": "d2af297496cb55a053629fe3f11a2b2fbc3e7acd9d7fa0872a4f130cf4a36efa", - "last_hidden_state": "bf7f36eda00adc4019dadefbcdc45e9f2c220e2f33122bd4bffb147d2d2a056a", - "candidate_token_ids": "9d4ac218fb54041e3a70a8e14db1ea1af9f570f4842b4702ef9323f1ad8f0ec4", - "candidate_logits": "c156769c9825ace28120add7875f05a82356739495cf5e77f3085f5e0e699ef9", - "candidate_probabilities": "e00e5eb9444182f352323374ef4e08ebcb784725fdd4fd612d7730540b3e0c8c", - "input_ids": "54c997a6bd3c87420e06814199bdd4c6250e8bd7a119905dc6b70bb9547cb275", - "attention_mask": "f2a58a8823b814fb09efb464a7cbd713a267a0330a3950d4c30794b042cdbbf3", - "mm_token_type_ids": "60264dae568d914f055d42dadf102ac26543bfb3c59c16c6ffcc6b4537556382", - "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", - "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", - "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", - "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" - } - }, - { - "case": "26-options", - "question": "next", - "image_size_wh": [ - 256, - 256 - ], - "prompt_tokens": 518, - "patches": 256, - "image_tokens": 64, - "choice": "option-0", - "tensor_sha256": { - "image_features": "27776425b299d2365ccb7c85a0ac978437330bb768870e7d5eccd9049aebe75c", - "inputs_embeds": "e5aae2f7a83de111de48d21ae88d88ffdf762ad7870006d5a764a179ee2755a5", - "position_ids": "f5d23ba88480091b2a48b402a09649426935e9c9705da3a193ff863608c50e52", - "last_hidden_state": "08fd4d02c72c6d8a14d8ac25907da2b694d65dbbdbc5138146585d3215cbbee1", - "candidate_token_ids": "3d8eb6eccd7f0b19cdf184a00d3a66c055c51144fd1fca423947c4e6ba8e0ffb", - "candidate_logits": "ec389d03d0dfd6aae31d453339ec39adc99958a7f35bc560eaad72e906451c8f", - "candidate_probabilities": "93ffe78005decc4c142827283b7f8f77c3a91f981657f3bc51b8c2a3da76e9a7", - "input_ids": "4ff5dc819a82ea350fdbaac5614d02dd0c5fcb598ec254d2509e32d75564c8ec", - "attention_mask": "897e287da60216749858043c3f41203172b7b95e03f529598c64d6b4f7de8fd5", - "mm_token_type_ids": "1ec7514267935156e9ece070c26c9ba76a41f897edf6338fce634774dadf735b", - "pixel_values": "e4bc5f6a335986e8f845008a409c8e30106b42351f3dbeef508671abfd04d87c", - "image_grid_thw": "53389fa912f56f81bbbe0669ab378c489b4cfaf9501876dcaec451ae86db735b", - "rope_deltas": "57e3269d7d7589d5903723ce291461fcabd8c9b96180084d0c8dee6f8830a977", - "image_token_indices": "54bdbec761821f053fad9eab88372b9a5224cb34de6693e73364ea8e0f4612a1" - } - }, - { - "case": "two-questions", - "question": "next", - "image_size_wh": [ - 320, - 240 - ], - "prompt_tokens": 235, - "patches": 320, - "image_tokens": 80, - "choice": "save", - "tensor_sha256": { - "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", - "inputs_embeds": "7761598e9b268125c2da4519a3a1a29724b4529f8b304413f80d43b516e98cf8", - "position_ids": "5f2b2d115b061ae076d3e08a6c6d1cb80e55afda4ae9d1e533d398d73c830449", - "last_hidden_state": "09d0e1519397478402144e773fc68e0dc23ca74b1be5ba8e06b8e9e3fc84016d", - "candidate_token_ids": "ed19debe0881d7ed8461ebfdb65c345f335dee64864e209d1d8aa821475d7ce9", - "candidate_logits": "0f43844e3ae275526d3a659484f1d2ffad9fd23a1896b3717069ecef23679f16", - "candidate_probabilities": "b98be35be7e3291eab531ab69fbec351e175f6d99c009650dc81ae713d1c7f68", - "input_ids": "d6313d4f14a714ce7017e1c18c0c3c6332d65662c0662f327f3d646bde794aed", - "attention_mask": "f72d07ac7afaa9e3a46d1e407a315b679c4782b07cf2e7099a57ca64971b9e43", - "mm_token_type_ids": "b9cced2b89f7bc1a3951b6e33c8f64a4ef0948c4010fe75f415e1d9acdbef42c", - "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", - "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", - "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", - "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" - } - }, - { - "case": "two-questions", - "question": "second", - "image_size_wh": [ - 320, - 240 - ], - "prompt_tokens": 229, - "patches": 320, - "image_tokens": 80, - "choice": "continue", - "tensor_sha256": { - "image_features": "047d4243c6a4272aa22131183eac0bb6500efd2af7b6d2a1dac878a4fee89aa0", - "inputs_embeds": "c64343cbde260b2137678280ada592565424642328e4c1f68998c66d70e74613", - "position_ids": "ee8ee859bb4c6fc19bbb21afb5858bc34f3c77cc639e6b92b2667b4a9d68850f", - "last_hidden_state": "3990053f803cee4dcf6bfb2f0f720e9626a4293e16c7aa2f1edbcc290ebf5839", - "candidate_token_ids": "69f367cf75d7888b5f7dd9a1742839d327c2e189173b60a39197d55f338f9ab7", - "candidate_logits": "04a5adbc4b329516181953dc4d3bec78d9d7373e776fa959770698a943e43832", - "candidate_probabilities": "7309d5948eec9911a1c24ad1306f3a44741bd7b4ccdc8efcec716a3b3f281d74", - "input_ids": "0314c4622ca5ef8f0423172fe100739b7da8b8a2a0dfaed669c67daf93175b24", - "attention_mask": "1d0fecdc1aef66d694d1a03853d4d98ad3e43dc08cf99a6f77b466737b679eaf", - "mm_token_type_ids": "0f510d2f44a832df16a9ccd4f19cf61918614f5730950e0c8fe62993583f97ec", - "pixel_values": "16e2585f739b0f79dabe6faef8b372d676c100452afab35f5a93f8c217b83ce7", - "image_grid_thw": "62ab45c5deaaea64b40e38c48f540fbdfb6235ee5d28cc2630dc82c526d25329", - "rope_deltas": "400c3ecb1bf85c0388e53f600e524fdd2716b0e0f973dec8a66f43d20eefd720", - "image_token_indices": "07f5346eb22ad4ad59cf7623ab212e93440e5bb2043f8b15299f6e527cc1a2d0" - } - } - ], - "verification_source_sha256": "107fb249e556acf798e756d5e9e553feda03dfe3e1caaef3854075d64dcb3997" -} diff --git a/recipe/cua_s1/reference.md b/recipe/cua_s1/reference.md deleted file mode 100644 index d5615f0..0000000 --- a/recipe/cua_s1/reference.md +++ /dev/null @@ -1,154 +0,0 @@ -# Cua-S1 multimodal reference tensors - -This offline recipe observes the Transformers/PEFT implementation merged in -[#12](https://github.com/ThinkFlowLab/system1-omni/pull/12). It provides concrete -inputs for the native vision integration discussed in -[#10](https://github.com/ThinkFlowLab/system1-omni/issues/10#issuecomment-5862276038). -The adapter stays **unmerged**, including the vision LoRA. It does not change the -worker or add a native vision implementation. - -## Reproduce - -Use Python 3.12 and an NVIDIA GPU with enough memory for Qwen3.5-4B and the full -reference output head. The validated setup is an RTX 4090 (24 GiB), driver -595.71.05, Torch 2.14.0+cu130 and the packages in -[requirements-reference.txt](requirements-reference.txt). Use a fresh environment -without flash-linear-attention, causal-conv1d or explicitly enabled hub kernels. -The model chooses its default attention implementation; the manifest records the -actual text and vision choices. These are reference exports, not timed benchmarks. - -From the repository root: - -```sh -python3.12 -m venv .venv -.venv/bin/python -m pip install torch==2.14.0 torchvision==0.29.0 \ - --index-url https://download.pytorch.org/whl/cu130 -.venv/bin/python -m pip install -r recipe/cua_s1/requirements-reference.txt - -git clone https://github.com/trycua/cua.git /tmp/cua-reference -git -C /tmp/cua-reference checkout 0e75660ce4c2edda519e0c795fa3ad98abf4e76f -.venv/bin/python /tmp/cua-reference/libs/cua-s1/ci/fetch_pinned_weights.py \ - --dest /tmp/cua-weights -cp /tmp/cua-reference/libs/cua-s1/ci/weights.lock.json /tmp/cua-weights/weights.lock.json - -PYTHONPATH=src HF_HUB_OFFLINE=1 TOKENIZERS_PARALLELISM=false \ - .venv/bin/python recipe/cua_s1/export_multimodal_reference.py \ - --weights /tmp/cua-weights --output /tmp/cua-reference-a -PYTHONPATH=src HF_HUB_OFFLINE=1 TOKENIZERS_PARALLELISM=false \ - .venv/bin/python recipe/cua_s1/export_multimodal_reference.py \ - --weights /tmp/cua-weights --output /tmp/cua-reference-b -PYTHONPATH=src .venv/bin/python recipe/cua_s1/verify_multimodal_reference.py \ - /tmp/cua-reference-a --compare /tmp/cua-reference-b -``` - -The model verifies the upstream manifest's pinned SHA-256 and the size/hash of -every loaded base/adapter file before loading. For existing verified weights, -provide the same directory layout and the upstream manifest next to `Qwen3.5-4B/`. -The model pins are recorded in [the model contract](../../src/models/cua_s1/README.md#pinned-revisions). - -Each output must be a new directory. An existing output is rejected before model -loading. `manifest.json` is written last; a directory without it is an incomplete -export. Save generated bundles outside the checkout. A bundle is about 44 MiB on -the validated setup and contains no checkpoint weights. - -## Fixed inputs - -The recipe generates redistributable synthetic account-settings screens using -Pillow; it needs no user screenshots or external images. The image bytes and the -full inline-image request JSON are included in `inputs/`. - -| Case | Original image, width × height | Format | Questions / options | -| --- | --- | --- | --- | -| small | 320 × 240 | PNG | 1 / 3 | -| wide | 640 × 320 | PNG | 1 / 3 | -| portrait | 320 × 640 | PNG | 1 / 3 | -| jpeg | 640 × 480 | JPEG | 1 / 3 | -| single-option | 320 × 240 | PNG | 1 / 1 | -| 26-options | 256 × 256 | PNG | 1 / 26 | -| two-questions | 320 × 240 | PNG | 2 / 3 and 2 | - -The second question includes structured, non-ASCII instructions and an object and -null criterion. Each question is an independent unpadded forward over the same -request image, preserving the #12 behavior and option order. Identical image bytes -must give identical preprocessing and adapted vision features across questions. - -## Bundle format and native boundary - -`manifest.json` uses schema `cua-s1-multimodal-reference-v1`. It records pinned -revisions, verified weight-manifest hash, package/GPU/CUDA/driver information, -source hashes, execution configuration, file sizes and SHA-256s. Each ordered -question entry contains its prompt, request/image paths, option keys, final answer, -and a safetensors file. Every tensor has its shape, dtype and SHA-256 of contiguous -CPU bytes, with no dtype conversion. BF16 bytes are retained directly. - -The bundle also includes base, processor and adapter JSON configurations in -`configs/`. Tensor dimensions below use `S` for prompt length, `P = T×H×W` for raw -patch count and `I = P / merge_size²` for merged image-token count. In this pinned -model, hidden width `D = 2560`, patch size 16, temporal patch size 2 and merge size 2. -A still image has `T = 1`; the processor duplicates its frame for temporal patches. - -| Tensor | Shape | Dtype | Meaning | -| --- | --- | --- | --- | -| `pixel_values` | `[P, 1536]` | float32 | Processor-normalized, flattened image patches; not an RGB image or NCHW tensor | -| `image_grid_thw` | `[1, 3]` | int64 | Raw temporal/height/width patch grid before spatial merging | -| `input_ids` | `[1, S]` | int64 | Fully expanded chat prompt, including `I` image placeholders | -| `attention_mask` | `[1, S]` | int64 | All ones: one unpadded prompt | -| `mm_token_type_ids` | `[1, S]` | int64 | Text 0, image 1 | -| `image_features` | `[I, D]` | bfloat16 | Actual adapted vision output after the merger, before insertion into the language input | -| `image_token_indices` | `[I]` | int64 | Positions where the image features replace token embeddings, in sequence order | -| `inputs_embeds` | `[1, S, D]` | bfloat16 | Actual language-model input after image insertion | -| `position_ids` | `[3, 1, S]` | int64 | Actual temporal/height/width M-RoPE positions at the language-model boundary | -| `rope_deltas` | `[1, 1]` | int64 | `max(position_ids) + 1 - S`, recorded from the model | -| `last_hidden_state` | `[1, D]` | bfloat16 | Final normalized language hidden state at the last sequence position | -| `candidate_token_ids` | `[C]` | int64 | A–Z letter ids, aligned with `option_keys` | -| `candidate_logits` | `[C]` | bfloat16 | Final-position logits for the candidates | -| `candidate_probabilities` | `[C]` | float32 | GPU fp32 softmax of the candidate logits | - -Temporary hooks observe the **actual** PEFT forward, including the vision merger -and the language input/last hidden state. They return no replacement outputs and -are removed in `finally`, including on failure. Position ids come from the model's -forward, rather than a separately reimplemented M-RoPE algorithm. Each captured -readout is checked against a second call to the ordinary #12 `score()` method; -the probabilities must match exactly before an export is marked complete. - -For a native language-path test, use `inputs_embeds`, `position_ids`, the base text -configuration and the multimodal-adapted language weights. The text adapter from -#19 is a different adapter. For a native vision test, start from `pixel_values` and -`image_grid_thw` and compare with `image_features`. The exported embedding contains -the unmerged reference adapter's BF16 rounding behavior; merging LoRA may change -that behavior and needs its own declared tolerance. - -## Verification and limits - -The verifier checks file and tensor fingerprints, exact keys/shapes/dtypes, -finite values, patch/merge counts, image-placeholder ordering, feature insertion, -position delta, candidate ordering, answer reconstruction and same-image reuse. -CPU softmax reconstruction allows absolute error `1e-7` with zero relative -allowance because CPU and GPU reduction implementations can round differently. -`--compare` additionally requires the two manifests to match exactly, including -all file and tensor hashes, execution settings and source/environment metadata. -The safetensors header key order is not part of a portable format guarantee. - -The checked [validation summary](reference-validation.json) records two -independent exports on the RTX 4090: 8 questions, 112 tensors, 25 files per bundle, -identical raw tensor bytes and identical ordinary/captured probabilities. CPU -softmax reconstruction differed by at most `3.73e-9`. This verifies the reference -artifact pipeline; it makes no native accuracy or speed claim. Bitwise equality -has only been checked on the recorded software/GPU setup. Other GPUs, CUDA builds, -FP32 controls, video, multiple images, padding/batching and Metal are unverified. -When evaluating a new native implementation, declare numerical tolerances before -comparison; the `--compare` mode is for repeating the same reference export. - -Run the focused tests without weights or a GPU: - -```sh -python -m pip install torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu -python -m pip install Pillow==11.3.0 numpy==2.5.3 safetensors==0.8.0 pytest==9.1.1 -PYTHONPATH=src python -m pytest tests/cua_s1/test_export_multimodal_reference.py -q -``` - -Tests cover deterministic cases, existing-output rejection, raw BF16 bytes, -actual hook capture and cleanup on success/failure, safetensors round trips, -corruption/unlisted-file detection, safe paths, missing tensor rejection and CPU -softmax roundoff versus drift. Without Torch, the two image/output tests run and tensor tests are skipped; -the recipe CI installs CPU Torch and executes the complete suite. diff --git a/recipe/cua_s1/requirements-reference.txt b/recipe/cua_s1/requirements-reference.txt deleted file mode 100644 index e2a0b84..0000000 --- a/recipe/cua_s1/requirements-reference.txt +++ /dev/null @@ -1,13 +0,0 @@ -# Reference environment from the pinned trycua/cua four-b lock. -# Install the CUDA Torch/Torchvision wheels first, as described in reference.md. -torch==2.14.0 -torchvision==0.29.0 -transformers==5.17.0 -peft==0.21.0 -accelerate==1.15.0 -Pillow==11.3.0 -safetensors==0.8.0 -huggingface-hub==1.32.0 -tokenizers==0.23.2 -# NumPy is used only for raw tensor-byte serialization. -numpy==2.5.3 diff --git a/tests/cua_s1/test_export_multimodal_reference.py b/tests/cua_s1/test_export_multimodal_reference.py deleted file mode 100644 index e165ea8..0000000 --- a/tests/cua_s1/test_export_multimodal_reference.py +++ /dev/null @@ -1,260 +0,0 @@ -"""CPU checks for the offline reference export; tensor checks need optional Torch.""" - -import base64 -import importlib.util -import json -from pathlib import Path -from types import SimpleNamespace - -import pytest - -ROOT = Path(__file__).resolve().parents[2] -SPEC = importlib.util.spec_from_file_location( - "export_multimodal_reference", ROOT / "recipe/cua_s1/export_multimodal_reference.py" -) - - -def recipe(): - module = importlib.util.module_from_spec(SPEC) - SPEC.loader.exec_module(module) - return module - - -def test_cases_reproduce_images_and_requests(tmp_path): - module = recipe() - left = module.make_cases(tmp_path / "left") - right = module.make_cases(tmp_path / "right") - assert len(left) == len(right) == 7 - assert sum(len(case["request"]["questions"]) for case in left) == 8 - for a, b in zip(left, right): - assert a["name"] == b["name"] - assert a["request"] == b["request"] - image = base64.b64decode(a["request"]["state"]["image"].split(",")[1]) - assert (tmp_path / "left" / a["image"]).read_bytes() == image - assert (tmp_path / "right" / b["image"]).read_bytes() == image - from models.cua_s1.multimodal.protocol import parse_request - - requests = [parse_request(case["request"]) for case in left] - assert len({request.image.size for request in requests}) >= 4 - assert {len(q.keys) for request in requests for q in request.questions} >= { - 1, - 3, - 26, - } - - -def test_existing_output_is_rejected_before_model_load(tmp_path): - module = recipe() - with pytest.raises(FileExistsError): - module.export(tmp_path / "missing-weights", tmp_path) - assert not list(tmp_path.iterdir()) - - -def test_fingerprint_preserves_bfloat16_bytes(): - torch = pytest.importorskip("torch") - module = recipe() - tensor = torch.tensor([[1.0, -2.0]], dtype=torch.bfloat16) - info = module.tensor_info(tensor) - assert info["shape"] == [1, 2] - assert info["dtype"] == "bfloat16" - assert info["sha256"] == module.sha256(tensor.view(torch.uint8).numpy().tobytes()) - assert module.tensor_info(tensor.t().contiguous().t()) == info - - -def toy_engine(torch, fail=False): - class Visual(torch.nn.Module): - def forward(self, pixels, **kwargs): - return SimpleNamespace(pooler_output=pixels.to(torch.bfloat16)) - - class Language(torch.nn.Module): - def forward(self, inputs_embeds=None, position_ids=None, **kwargs): - if fail: - raise RuntimeError("deliberate forward failure") - return SimpleNamespace(last_hidden_state=inputs_embeds + 1) - - class Model(torch.nn.Module): - def __init__(self): - super().__init__() - self.model = torch.nn.Module() - self.model.visual = Visual() - self.model.language_model = Language() - self.model.rope_deltas = torch.tensor([[-1]]) - self.device = torch.device("cpu") - self.config = SimpleNamespace(image_token_id=99) - - def get_base_model(self): - return self - - def forward(self, input_ids, pixel_values, **kwargs): - features = self.model.visual(pixel_values).pooler_output - embeds = torch.zeros(1, input_ids.shape[1], 2, dtype=torch.bfloat16) - embeds[0, input_ids[0] == 99] = features - positions = torch.arange(input_ids.shape[1]).expand(3, 1, -1) - result = self.model.language_model( - inputs_embeds=embeds, position_ids=positions - ) - logits = result.last_hidden_state[..., :1].expand(-1, -1, 100).clone() - return SimpleNamespace(logits=logits) - - model = Model() - tokenizer = SimpleNamespace(encode=lambda letter, **kwargs: [ord(letter) - 33]) - return SimpleNamespace(model=model, tokenizer=tokenizer) - - -def test_capture_observes_forward_and_removes_hooks(): - torch = pytest.importorskip("torch") - module = recipe() - from models.cua_s1.multimodal.protocol import Question - - engine = toy_engine(torch) - inputs = { - "input_ids": torch.tensor([[1, 99, 99, 2]]), - "pixel_values": torch.tensor([[2.0, 3.0], [4.0, 5.0]]), - } - question = Question("test", ("a", "b"), ("First", "Second"), "Choose") - tensors = module.capture(engine, inputs, question) - assert torch.equal(tensors["inputs_embeds"][0, 1:3], tensors["image_features"]) - assert tensors["position_ids"].shape == (3, 1, 4) - assert tensors["last_hidden_state"].shape == (1, 2) - assert tensors["candidate_probabilities"].tolist() == [0.5, 0.5] - assert all(t.device.type == "cpu" for t in tensors.values()) - assert not engine.model.model.visual._forward_hooks - assert not engine.model.model.language_model._forward_pre_hooks - assert not engine.model.model.language_model._forward_hooks - - -def test_forward_failure_also_removes_hooks(): - torch = pytest.importorskip("torch") - module = recipe() - from models.cua_s1.multimodal.protocol import Question - - engine = toy_engine(torch, fail=True) - with pytest.raises(RuntimeError, match="deliberate forward failure"): - module.capture( - engine, - {"input_ids": torch.tensor([[99]]), "pixel_values": torch.ones(1, 2)}, - Question("test", ("a",), ("First",), "Choose"), - ) - assert not engine.model.model.visual._forward_hooks - assert not engine.model.model.language_model._forward_pre_hooks - assert not engine.model.model.language_model._forward_hooks - - -def test_bundle_roundtrip_and_independent_storage(tmp_path): - torch = pytest.importorskip("torch") - from safetensors.torch import load_file - - module = recipe() - original = torch.tensor([[1, 2]], dtype=torch.int64) - entries = module.save_tensors( - tmp_path / "q.safetensors", {"a": original, "b": original} - ) - loaded = load_file(str(tmp_path / "q.safetensors")) - assert torch.equal(loaded["a"], original) - assert entries["a"] == module.tensor_info(loaded["a"]) - assert entries["b"] == entries["a"] - json.dumps(entries, allow_nan=False) - - -def test_verifier_checks_files_and_rejects_corruption(tmp_path, monkeypatch): - torch = pytest.importorskip("torch") - monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) - import verify_multimodal_reference as verifier - - module = recipe() - bundle = tmp_path / "bundle" - bundle.mkdir() - tensor_path = bundle / "q.safetensors" - metadata = module.save_tensors(tensor_path, {"example": torch.ones(2)}) - manifest = { - "schema": module.SCHEMA, - "files": { - "q.safetensors": { - "size": tensor_path.stat().st_size, - "sha256": module.sha256(tensor_path.read_bytes()), - } - }, - "questions": [{"tensors_file": "q.safetensors", "tensors": metadata}], - } - module.write_json(bundle / "manifest.json", manifest) - # File integrity is independently testable before semantic validation. - verifier.check_files(bundle, manifest) - tensor_path.write_bytes(tensor_path.read_bytes() + b"corrupt") - with pytest.raises(ValueError, match="file checksum"): - verifier.check_files(bundle, manifest) - - -def test_verifier_does_not_accept_missing_tensors(tmp_path, monkeypatch): - pytest.importorskip("torch") - monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) - import verify_multimodal_reference as verifier - - with pytest.raises(ValueError, match="missing tensors"): - verifier.check_tensors({}, {}, {}) - - -def test_readout_allows_cpu_roundoff_but_rejects_drift(monkeypatch): - torch = pytest.importorskip("torch") - monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) - import verify_multimodal_reference as verifier - - module = recipe() - tensors = { - "input_ids": torch.tensor([[1, 99, 99, 2]]), - "attention_mask": torch.ones(1, 4, dtype=torch.int64), - "mm_token_type_ids": torch.tensor([[0, 1, 1, 0]]), - "pixel_values": torch.ones(2, 3), - "image_grid_thw": torch.tensor([[1, 1, 2]]), - "image_features": torch.ones(2, 2, dtype=torch.bfloat16), - "image_token_indices": torch.tensor([1, 2]), - "inputs_embeds": torch.ones(1, 4, 2, dtype=torch.bfloat16), - "position_ids": torch.arange(4).expand(3, 1, -1), - "rope_deltas": torch.tensor([[0]]), - "last_hidden_state": torch.ones(1, 2, dtype=torch.bfloat16), - "candidate_token_ids": torch.tensor([32, 33]), - "candidate_logits": torch.tensor([0, 1], dtype=torch.bfloat16), - "candidate_probabilities": torch.softmax(torch.tensor([0.0, 1.0]), dim=-1), - } - tensors["candidate_probabilities"][0] += 3e-8 - configs = { - "base": { - "text_config": {"hidden_size": 2}, - "vision_config": { - "spatial_merge_size": 1, - "patch_size": 1, - "temporal_patch_size": 1, - }, - "image_token_id": 99, - } - } - entry = { - "option_keys": ["a", "b"], - "tensors": {k: module.tensor_info(v) for k, v in tensors.items()}, - } - verifier.check_tensors(tensors, entry, configs) - tensors["candidate_probabilities"][0] += 1e-4 - entry["tensors"] = {k: module.tensor_info(v) for k, v in tensors.items()} - with pytest.raises(ValueError, match="readout mismatch"): - verifier.check_tensors(tensors, entry, configs) - - -@pytest.mark.parametrize("unsafe", [False, True]) -def test_verifier_rejects_unlisted_files_and_unsafe_paths( - tmp_path, monkeypatch, unsafe -): - pytest.importorskip("torch") - monkeypatch.syspath_prepend(str(ROOT / "recipe/cua_s1")) - import verify_multimodal_reference as verifier - - module = recipe() - folder = tmp_path / "bundle" - folder.mkdir() - extra = (tmp_path if unsafe else folder) / "extra.json" - extra.write_text("{}") - manifest = {"files": {}} - if unsafe: - manifest["files"]["../extra.json"] = {"size": 2, "sha256": module.sha256(b"{}")} - with pytest.raises( - ValueError, match="unsafe file path" if unsafe else "file inventory" - ): - verifier.check_files(folder, manifest)