From 75a80cbdf95f281eb9e29054091f7bdd7c3d2898 Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Thu, 1 Oct 2026 09:59:24 +0800 Subject: [PATCH 1/2] cua_s1: accept adapted image rows and three-axis language positions --- recipe/cua_s1/native.md | 112 +++++- .../native/examples/multimodal_boundary.rs | 184 ++++++++++ src/models/cua_s1/native/src/inputs.rs | 213 ++++++++++++ src/models/cua_s1/native/src/lib.rs | 1 + src/models/cua_s1/native/src/model.rs | 328 ++++++++++++++---- src/models/cua_s1/native/tests/multimodal.rs | 90 +++++ 6 files changed, 867 insertions(+), 61 deletions(-) create mode 100644 src/models/cua_s1/native/examples/multimodal_boundary.rs create mode 100644 src/models/cua_s1/native/src/inputs.rs create mode 100644 src/models/cua_s1/native/tests/multimodal.rs diff --git a/recipe/cua_s1/native.md b/recipe/cua_s1/native.md index 018feff..35dccb4 100644 --- a/recipe/cua_s1/native.md +++ b/recipe/cua_s1/native.md @@ -24,9 +24,11 @@ CUA_S1_MODEL=weights/cua-s1-4b-0.2-text-merged target/release/omni-cua-s1-native ``` For the local CUDA Graph experiment, also set `CUA_S1_GRAPH=1`. The first use of -each exact prompt length warms the GEMM plans and captures the forward pass; -later requests replay it with freshly uploaded token ids. At most eight lengths -are cached. Growing the scratch allocation clears the captures before freeing +each exact prompt length warms the GEMM plans and captures the forward pass. +The first call returns the eager result; later requests replay it after fresh +token embedding. The graph contains the language layers, which update residuals +in place, so a cache miss must not replay those layers over its eager result. +At most eight lengths are cached. Growing the scratch allocation clears the captures before freeing their buffers. Capture adds first-use latency; leave the variable unset to use the eager control. Rebuild both the worker and CUDA library together (ABI 3). If capture fails, the worker returns the completed eager result and disables @@ -41,3 +43,107 @@ cargo test -p omni-cua-s1-native CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ cargo test --release -p omni-cua-s1-native --test kernels -- --ignored ``` + +## Multimodal language boundary + +`Model::forward_multimodal` consumes token IDs, adapted BF16 image features +`[image_tokens, hidden_size]`, their sorted placeholder indices, and the T/H/W +slices of int64 `position_ids [3, 1, sequence]`. It returns the last position's +final-normalized hidden state, like `Model::forward`. + +Inputs are one unpadded prompt. Every image placeholder must have exactly one +feature row; all other rows come from the token embedding table. The caller +calculates positions and runs the image processor, vision tower and vision LoRA. +Positions must be nonnegative and below `max_position_embeddings`. The language +path uses Qwen3.5's interleaved MRoPE sections, not three contiguous rotary blocks. +Text calls and their captured graphs use immutable text-position tables; +multimodal calls use separate device tables, so returning to text requires no +host table rebuild or restoration copy. Multimodal calls execute eagerly even +when `CUA_S1_GRAPH=1`. The extra tables use +`4 * scratch_capacity * rotary_half` bytes (2 MiB at 16,384 rows). + +This is a Rust model API for integrating a vision producer. The HTTP worker +above continues to serve the text adapter. A native vision encoder, image HTTP +requests, padding, video and batching are not implemented by this API. + +### Prepare a matching language checkpoint + +The language weights must contain the **multimodal** adapter, not the `text` +adapter. In the pinned reference environment, with upstream-verified weights, +export just the merged language model (about 7.5 GB) to a new directory: + +```sh +PYTHONPATH=src HF_HUB_OFFLINE=1 .venv/bin/python - <<'PY' +import json +from pathlib import Path +from models.cua_s1.multimodal.model import ( + ADAPTER_REVISION, BASE_REVISION, MultimodalEngine, +) + +out = Path("weights/cua-s1-4b-0.2-multimodal-language-merged") +if out.exists(): + raise FileExistsError(out) +engine = MultimodalEngine( + "weights/Qwen3.5-4B", "weights/cua-s1-4b-0.2/multimodal" +) +merged = engine.model.merge_and_unload() +merged.model.language_model.save_pretrained(out, max_shard_size="5GB") +# Preserve the root image_token_id and text_config for the native input contract. +merged.config.to_json_file(out / "config.json") +(out / "cua_s1_language_export.json").write_text(json.dumps({ + "format": "cua-s1-multimodal-language-merged/1", + "base_revision": BASE_REVISION, + "adapter_revision": ADAPTER_REVISION, +})) +PY +``` + +No `cua_s1_export.json` text-worker marker is created. The low-level `Model` API +does not verify checkpoint provenance; retain the export metadata and use the +matching adapter for the supplied features. Standalone language safetensors +names and the existing full-model prefixes are supported. + +### Replay a reference boundary + +This optional example consumes the `cua-s1-multimodal-reference-v1` format +from [#53](https://github.com/ThinkFlowLab/system1-omni/pull/53), which is still +open. The exporter and checksum verifier are not yet available on `main`. +Use a separate checkout of exporter revision +`1b64fa2ceb0a82b6a66a69ecdc9bc5cc1b1a0b66` to generate and verify the bundle; +the Rust model API itself does not depend on that PR being merged. +Its eight questions include different image grids and question lengths, JPEG, +structured/non-ASCII text, and 1/3/26 candidates. Then run: + +```sh +CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ + cargo run --release --locked -p omni-cua-s1-native \ + --example multimodal_boundary -- \ + weights/cua-s1-4b-0.2-multimodal-language-merged \ + /path/to/verified-reference-bundle /tmp/native-language.json + +CUA_S1_MODEL=$PWD/weights/cua-s1-4b-0.2-multimodal-language-merged \ +CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ + cargo test --release --locked -p omni-cua-s1-native \ + --test multimodal -- --ignored + +# Compare graph misses, hits, eviction and scratch growth with eager hidden states. +CUA_S1_MODEL=$PWD/weights/cua-s1-4b-0.2-multimodal-language-merged \ +CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ + cargo test --release --locked -p omni-cua-s1-native \ + --lib graph_tests::graph_misses_hits_eviction_growth_and_multimodal_match_eager -- --ignored +``` + +The example checks repeated native hidden-state equality and writes last hidden +states, candidate logits and probabilities. It uses the text engine's FP32 +letter-row readout with FP64 accumulation. Output must be a new file. Verify +bundle integrity before invoking the example; it checks tensor shapes and input +contracts but is not the bundle checksum verifier. + +For accuracy validation, compare against an unmerged FP32 **language** control +with TF32 disabled, feeding the same fixed exported embeddings and positions. +Use the [declared native tolerance](../../src/models/cua_s1/README.md#validation): +maximum probability error over the set must be at most twice the BF16 reference +error plus 0.01, and the top option must match for FP32 margins at least 0.05. +The FP32 control starts after the BF16-exported vision boundary; it does not +validate a full FP32 vision pipeline. Native kernel and LoRA-merge rounding can +change hidden states and logits; bitwise equality to Transformers is not claimed. diff --git a/src/models/cua_s1/native/examples/multimodal_boundary.rs b/src/models/cua_s1/native/examples/multimodal_boundary.rs new file mode 100644 index 0000000..7c93fb4 --- /dev/null +++ b/src/models/cua_s1/native/examples/multimodal_boundary.rs @@ -0,0 +1,184 @@ +//! Replay #53's exported language boundary through the native model. +//! The v1 bundle producer/verifier are on open PR #53 at exporter revision +//! 1b64fa2ceb0a82b6a66a69ecdc9bc5cc1b1a0b66; see recipe/cua_s1/native.md. +use std::path::{Path, PathBuf}; + +use anyhow::{Context, Result, ensure}; +use half::bf16; +use omni_cua_s1_native::{inputs::MultimodalInput, model::Model}; +use safetensors::{Dtype, SafeTensors}; +use serde_json::{Value, json}; + +fn integers(st: &SafeTensors<'_>, name: &str, shape: &[usize]) -> Result> { + let v = st.tensor(name)?; + ensure!( + v.dtype() == Dtype::I64 && v.shape() == shape, + "{name}: expected I64 {shape:?}" + ); + Ok(v.data() + .as_chunks::<8>() + .0 + .iter() + .map(|b| i64::from_le_bytes(*b)) + .collect()) +} + +fn features(st: &SafeTensors<'_>, shape: &[usize]) -> Result> { + let v = st.tensor("image_features")?; + ensure!( + v.dtype() == Dtype::BF16 && v.shape() == shape, + "image_features: expected BF16 {shape:?}" + ); + Ok(v.data() + .as_chunks::<2>() + .0 + .iter() + .map(|b| bf16::from_le_bytes(*b)) + .collect()) +} + +fn embedding_file(dir: &Path) -> Result<(PathBuf, String)> { + let names = [ + "model.language_model.embed_tokens.weight", + "model.embed_tokens.weight", + "embed_tokens.weight", + ]; + if dir.join("model.safetensors.index.json").exists() { + let index: Value = + serde_json::from_slice(&std::fs::read(dir.join("model.safetensors.index.json"))?)?; + for name in names { + if let Some(file) = index["weight_map"][name].as_str() { + return Ok((dir.join(file), name.into())); + } + } + } else { + let path = dir.join("model.safetensors"); + let bytes = std::fs::read(&path)?; + let st = SafeTensors::deserialize(&bytes)?; + for name in names { + if st.tensor(name).is_ok() { + return Ok((path, name.into())); + } + } + } + anyhow::bail!("missing embedding weight") +} + +fn main() -> Result<()> { + let args: Vec<_> = std::env::args_os().skip(1).collect(); + ensure!( + args.len() == 3, + "usage: multimodal_boundary MODEL_DIR REFERENCE_BUNDLE OUTPUT_JSON" + ); + let (dir, bundle, out) = ( + Path::new(&args[0]), + Path::new(&args[1]), + Path::new(&args[2]), + ); + ensure!(!out.exists(), "output already exists"); + let library = PathBuf::from(std::env::var_os("CUA_S1_CUDA_LIB").context("CUA_S1_CUDA_LIB")?); + let manifest: Value = serde_json::from_slice(&std::fs::read(bundle.join("manifest.json"))?)?; + ensure!( + manifest["schema"] == "cua-s1-multimodal-reference-v1", + "unsupported reference schema" + ); + let mut model = Model::load(dir, &library)?; + let (file, name) = embedding_file(dir)?; + let file = std::fs::File::open(file)?; + // SAFETY: the checkpoint is immutable while the example runs. + let map = unsafe { memmap2::Mmap::map(&file)? }; + let weights = SafeTensors::deserialize(&map)?; + let embed = weights.tensor(&name)?; + ensure!( + embed.dtype() == Dtype::BF16 + && embed.shape().len() == 2 + && embed.shape()[1] == model.cfg.hidden, + "embedding shape/dtype" + ); + let mut rows = Vec::new(); + for entry in manifest["questions"].as_array().context("questions")? { + let relative = Path::new(entry["tensors_file"].as_str().context("tensors_file")?); + ensure!( + relative + .components() + .all(|c| matches!(c, std::path::Component::Normal(_))), + "unsafe tensor path" + ); + let bytes = std::fs::read(bundle.join(relative))?; + let st = SafeTensors::deserialize(&bytes)?; + let ids = st.tensor("input_ids")?; + ensure!( + ids.shape().len() == 2 && ids.shape()[0] == 1, + "expected batch one" + ); + let t = ids.shape()[1]; + let ids: Vec = integers(&st, "input_ids", &[1, t])? + .into_iter() + .map(u32::try_from) + .collect::>()?; + let indices = st.tensor("image_token_indices")?; + ensure!( + indices.shape().len() == 1, + "image indices must be one-dimensional" + ); + let count = indices.shape()[0]; + let indices: Vec = integers(&st, "image_token_indices", &[count])? + .into_iter() + .map(usize::try_from) + .collect::>()?; + let features = features(&st, &[count, model.cfg.hidden])?; + let positions = integers(&st, "position_ids", &[3, 1, t])?; + let input = MultimodalInput { + token_ids: &ids, + image_token_indices: &indices, + image_embeddings: &features, + position_ids: [&positions[..t], &positions[t..2 * t], &positions[2 * t..]], + }; + let last = model.forward_multimodal(&input)?; + ensure!( + last.iter().all(|x| x.is_finite()), + "non-finite hidden state" + ); + // Repeating the same boundary in the same model checks buffer reuse. + ensure!( + last == model.forward_multimodal(&input)?, + "repeat changed native hidden state" + ); + let n = entry["option_keys"] + .as_array() + .context("option_keys")? + .len(); + ensure!((1..=26).contains(&n), "candidate count"); + let candidates = integers(&st, "candidate_token_ids", &[n])?; + let mut logits = Vec::new(); + for id in candidates { + let id = usize::try_from(id)?; + ensure!(id < embed.shape()[0], "candidate outside vocabulary"); + let row = &embed.data()[id * last.len() * 2..(id + 1) * last.len() * 2]; + let dot: f64 = row + .as_chunks::<2>() + .0 + .iter() + .zip(&last) + .map(|(b, &h)| bf16::from_le_bytes(*b).to_f64() * h as f64) + .sum(); + logits.push(dot as f32); + } + let max = logits.iter().copied().fold(f32::NEG_INFINITY, f32::max) as f64; + let exps: Vec = logits.iter().map(|&l| (l as f64 - max).exp()).collect(); + let total: f64 = exps.iter().sum(); + let probabilities: Vec = exps.iter().map(|e| (e / total) as f32).collect(); + ensure!( + probabilities.iter().all(|x| x.is_finite()), + "non-finite readout" + ); + rows.push(json!({"case": entry["case"], "question": entry["question"], "sequence": t, "image_tokens": count, "last_hidden_state": last, "candidate_logits": logits, "probabilities": probabilities, "repeat_equal": true})); + } + std::fs::write( + out, + serde_json::to_vec_pretty( + &json!({"schema": "cua-s1-native-language-boundary-v1", "questions": rows}), + )?, + )?; + Ok(()) +} diff --git a/src/models/cua_s1/native/src/inputs.rs b/src/models/cua_s1/native/src/inputs.rs new file mode 100644 index 0000000..1f0c980 --- /dev/null +++ b/src/models/cua_s1/native/src/inputs.rs @@ -0,0 +1,213 @@ +//! Batch-one, unpadded inputs at the adapted-vision / language-model boundary. + +use anyhow::{Result, ensure}; +use half::bf16; + +/// Image rows are already adapted to the language hidden size, in placeholder +/// order. Positions are the T/H/W slices of an int64 `[3, 1, sequence]` tensor. +/// The caller owns preprocessing, vision execution and the position calculation. +pub struct MultimodalInput<'a> { + pub token_ids: &'a [u32], + pub image_token_indices: &'a [usize], + pub image_embeddings: &'a [bf16], + pub position_ids: [&'a [i64]; 3], +} + +impl MultimodalInput<'_> { + /// Check the entire boundary before allocating buffers or launching CUDA. + pub fn validate( + &self, + hidden: usize, + vocab: usize, + image_token: u32, + max_position: usize, + ) -> Result<()> { + let t = self.token_ids.len(); + ensure!(t > 0 && t <= max_position, "empty or oversized prompt"); + ensure!( + self.token_ids.iter().all(|&id| (id as usize) < vocab), + "token id outside the vocabulary" + ); + let expected: Vec = self + .token_ids + .iter() + .enumerate() + .filter_map(|(i, &id)| (id == image_token).then_some(i)) + .collect(); + ensure!( + self.image_token_indices == expected, + "image indices must exactly match the ordered placeholders" + ); + ensure!( + Some(self.image_embeddings.len()) == expected.len().checked_mul(hidden), + "image embedding shape mismatch" + ); + ensure!( + self.image_embeddings.iter().all(|x| x.is_finite()), + "non-finite image embedding" + ); + ensure!( + self.position_ids.iter().all(|axis| axis.len() == t), + "position_ids must have shape [3, 1, sequence]" + ); + ensure!( + self.position_ids + .iter() + .flat_map(|axis| axis.iter()) + .all(|&p| p >= 0 && (p as u64) < max_position as u64), + "position outside the configured range" + ); + Ok(()) + } +} + +/// Qwen3.5's interleaved recomposition: overwrite H at 1::3 and W at 2::3 up +/// to section[axis] * 3, retaining T elsewhere. The second rotary half repeats +/// these frequencies, which the existing attention-prep kernel handles. +/// Float32 inverse frequencies/products and host float64 trig preserve the +/// original native text table rounding when all three axes are equal. +pub(crate) fn rotary_tables( + positions: [&[i64]; 3], + half: usize, + theta: f64, + sections: [usize; 3], +) -> (Vec, Vec) { + let inv: Vec = (0..half) + .map(|i| 1f32 / (theta as f32).powf((2 * i) as f32 / (2 * half) as f32)) + .collect(); + let mut cos = Vec::with_capacity(positions[0].len() * half * 2); + let mut sin = Vec::with_capacity(cos.capacity()); + for (t, _) in positions[0].iter().enumerate() { + for (i, &f) in inv.iter().enumerate() { + let axis = if i % 3 == 1 && i < sections[1] * 3 { + 1 + } else if i % 3 == 2 && i < sections[2] * 3 { + 2 + } else { + 0 + }; + let angle = (f * positions[axis][t] as f32) as f64; + cos.extend(bf16::from_f32(angle.cos() as f32).to_le_bytes()); + sin.extend(bf16::from_f32(angle.sin() as f32).to_le_bytes()); + } + } + (cos, sin) +} + +#[cfg(test)] +mod tests { + use super::*; + use half::bf16; + + #[test] + fn valid_input_and_three_distinct_axes() { + let input = MultimodalInput { + token_ids: &[1, 99, 99, 2], + image_token_indices: &[1, 2], + image_embeddings: &[bf16::ONE; 8], + position_ids: [&[0, 1, 1, 3], &[0, 1, 2, 3], &[0, 2, 1, 3]], + }; + input.validate(4, 100, 99, 100).unwrap(); + } + + #[test] + fn rejects_bad_placeholder_inventory_and_feature_rows() { + for indices in [vec![2, 1], vec![1, 1], vec![1], vec![0, 1], vec![1, 4]] { + let input = MultimodalInput { + token_ids: &[1, 99, 99, 2], + image_token_indices: &indices, + image_embeddings: &[bf16::ONE; 8], + position_ids: [&[0, 1, 1, 3]; 3], + }; + assert!(input.validate(4, 100, 99, 100).is_err(), "{indices:?}"); + } + for features in [vec![bf16::ONE; 7], vec![bf16::ONE; 9], vec![bf16::NAN; 8]] { + let input = MultimodalInput { + token_ids: &[1, 99, 99, 2], + image_token_indices: &[1, 2], + image_embeddings: &features, + position_ids: [&[0, 1, 1, 3]; 3], + }; + assert!(input.validate(4, 100, 99, 100).is_err()); + } + } + + #[test] + fn rejects_bad_tokens_positions_and_empty_sequence() { + let features = [bf16::ONE; 4]; + for (ids, positions) in [ + (vec![100, 99], vec![0, 1]), + (vec![1, 99], vec![0]), + (vec![1, 99], vec![0, -1]), + (vec![1, 99], vec![0, 100]), + (vec![], vec![]), + ] { + let input = MultimodalInput { + token_ids: &ids, + image_token_indices: &[1], + image_embeddings: &features, + position_ids: [&positions; 3], + }; + assert!(input.validate(4, 100, 99, 100).is_err()); + } + } + + #[test] + fn image_free_explicit_positions_are_valid() { + MultimodalInput { + token_ids: &[1, 2], + image_token_indices: &[], + image_embeddings: &[], + position_ids: [&[7, 8]; 3], + } + .validate(4, 100, 99, 100) + .unwrap(); + } + + #[test] + fn rotary_interleaves_height_width_and_leaves_temporal_tail() { + // theta=1 makes every inverse frequency 1. Axis values differ so a plain + // text table or a contiguous-section implementation fails this check. + for (sections, tail) in [([12, 10, 10], [0, 0]), ([11, 11, 10], [0, 1])] { + let (cos, sin) = rotary_tables([&[0], &[1], &[2]], 32, 1.0, sections); + // Ten T/H/W triples, then T/T for the synthetic layout or T/H for + // the real checkpoint. In particular, frequency 31 must use H. + let axes = [ + 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, + 1, 2, tail[0], tail[1], + ]; + for (i, axis) in axes.into_iter().enumerate() { + let angle = axis as f32; + assert_eq!( + &cos[i * 2..i * 2 + 2], + &bf16::from_f32(angle.cos()).to_le_bytes() + ); + assert_eq!( + &sin[i * 2..i * 2 + 2], + &bf16::from_f32(angle.sin()).to_le_bytes() + ); + } + } + } + + #[test] + fn equal_axes_reproduce_the_existing_text_tables() { + let pos: Vec = (0..257).collect(); + let (cos, sin) = rotary_tables([&pos; 3], 32, 10_000_000.0, [11, 11, 10]); + for (t, &p) in pos.iter().enumerate() { + for i in 0..32 { + let inv = 1f32 / 10_000_000f32.powf((2 * i) as f32 / 64.0); + let angle = (inv * p as f32) as f64; + let offset = (t * 32 + i) * 2; + assert_eq!( + &cos[offset..offset + 2], + &bf16::from_f32(angle.cos() as f32).to_le_bytes() + ); + assert_eq!( + &sin[offset..offset + 2], + &bf16::from_f32(angle.sin() as f32).to_le_bytes() + ); + } + } + } +} diff --git a/src/models/cua_s1/native/src/lib.rs b/src/models/cua_s1/native/src/lib.rs index 0d3aa5f..3d09d9c 100644 --- a/src/models/cua_s1/native/src/lib.rs +++ b/src/models/cua_s1/native/src/lib.rs @@ -5,5 +5,6 @@ pub mod contract; pub mod cuda; pub mod engine; +pub mod inputs; pub mod json; pub mod model; diff --git a/src/models/cua_s1/native/src/model.rs b/src/models/cua_s1/native/src/model.rs index ae14d55..d6d6597 100644 --- a/src/models/cua_s1/native/src/model.rs +++ b/src/models/cua_s1/native/src/model.rs @@ -6,8 +6,8 @@ //! The order of operations follows `modeling_qwen3_5.py`, and so do the points where //! it rounds to bfloat16, except inside attention and the Gated DeltaNet prefill (see //! their kernels). Text prompts use one position per -//! token, so the multimodal rotary sections all get the same position and the -//! rotary embedding is the plain one. +//! token. The explicit multimodal boundary inserts adapted image rows and supplies +//! the interleaved temporal/height/width rotary positions to the same layer loop. use std::collections::{HashMap, VecDeque}; use std::ffi::c_void; @@ -17,6 +17,7 @@ use anyhow::{Context, Result, bail, ensure}; use serde_json::Value as Json; use crate::cuda::{self, DeviceBuffer, Stream, check}; +use crate::inputs::{MultimodalInput, rotary_tables}; const ALIGN: usize = 256; const BF16: usize = 2; @@ -35,6 +36,9 @@ pub struct Config { /// Half the number of rotary dims (rotate_half pairs dim i with dim i + half). pub rotary_half: usize, pub rope_theta: f64, + pub mrope_section: [usize; 3], + pub max_positions: usize, + pub image_token_id: Option, pub lin_k_heads: usize, pub lin_v_heads: usize, pub lin_k_dim: usize, @@ -69,6 +73,31 @@ impl Config { other => bail!("unknown layer type {other:?}"), }) .collect::>>()?; + let sections = rope + .get("mrope_section") + .cloned() + .unwrap_or(serde_json::json!([11, 11, 10])); + let sections = sections + .as_array() + .context("mrope_section must be an array")? + .iter() + .map(|v| { + v.as_u64() + .and_then(|n| usize::try_from(n).ok()) + .context("mrope_section must contain integers") + }) + .collect::>>()?; + let mrope_section: [usize; 3] = sections + .try_into() + .map_err(|_| anyhow::anyhow!("mrope_section must have three entries"))?; + let image_token_id = root + .get("image_token_id") + .map(|v| { + v.as_u64() + .and_then(|id| u32::try_from(id).ok()) + .context("image_token_id must be a u32") + }) + .transpose()?; let cfg = Config { hidden: int("hidden_size")?, intermediate: int("intermediate_size")?, @@ -81,6 +110,9 @@ impl Config { .as_f64() .or(c["rope_theta"].as_f64()) .context("rope_theta")?, + mrope_section, + max_positions: int("max_position_embeddings")?, + image_token_id, lin_k_heads: int("linear_num_key_heads")?, lin_v_heads: int("linear_num_value_heads")?, lin_k_dim: int("linear_key_head_dim")?, @@ -115,6 +147,25 @@ impl Config { cfg.lin_v_dim ); ensure!(cfg.rotary_half == 32, "{} rotary dims", 2 * cfg.rotary_half); + ensure!( + cfg.rope_theta.is_finite() && cfg.rope_theta > 0.0, + "invalid rope_theta" + ); + ensure!( + rope["mrope_interleaved"].as_bool() != Some(false), + "non-interleaved mrope is unsupported" + ); + ensure!( + cfg.mrope_section + .iter() + .try_fold(0usize, |sum, &x| sum.checked_add(x)) + == Some(cfg.rotary_half), + "mrope sections must sum to rotary_half" + ); + ensure!( + cfg.max_positions > 0 && cfg.max_positions <= (1 << 24), + "unsupported max_position_embeddings" + ); ensure!( cfg.kv_heads > 0 && cfg.heads.is_multiple_of(cfg.kv_heads), "attention heads" @@ -218,7 +269,7 @@ impl Weights { .enumerate() .flat_map(|(i, st)| st.names().into_iter().map(move |n| (i, n.to_string()))) .collect(); - let prefix = ["model.language_model.", "model."] + let prefix = ["model.language_model.", "model.", ""] .into_iter() .find(|p| { names @@ -385,6 +436,8 @@ struct Scratch { act: usize, cos: usize, sin: usize, + custom_cos: usize, + custom_sin: usize, } impl Scratch { @@ -423,6 +476,8 @@ impl Scratch { take(cap * cfg.intermediate * BF16), take(cap * cfg.rotary_half * BF16), take(cap * cfg.rotary_half * BF16), + take(cap * cfg.rotary_half * BF16), + take(cap * cfg.rotary_half * BF16), ]; let buf = DeviceBuffer::new(next)?; let [ @@ -448,24 +503,16 @@ impl Scratch { act, cos, sin, + custom_cos, + custom_sin, ] = offsets; - // Rotary tables close to how Qwen3_5TextRotaryEmbedding builds them: inv_freq and - // freqs = inv_freq * position in float32, cos and sin rounded to bfloat16. Here - // cos and sin are taken in float64 on the host rather than in float32 on the - // GPU, so a few of the rounded values can differ by one bfloat16 step. - let half = cfg.rotary_half; - let inv: Vec = (0..half) - .map(|i| 1.0f32 / (cfg.rope_theta as f32).powf((2 * i) as f32 / (2 * half) as f32)) - .collect(); - let mut cos_t = Vec::with_capacity(cap * half * BF16); - let mut sin_t = Vec::with_capacity(cap * half * BF16); - for pos in 0..cap { - for &f in &inv { - let freq = (f * pos as f32) as f64; - cos_t.extend(half::bf16::from_f32(freq.cos() as f32).to_le_bytes()); - sin_t.extend(half::bf16::from_f32(freq.sin() as f32).to_le_bytes()); - } - } + let positions: Vec = (0..cap as i64).collect(); + let (cos_t, sin_t) = rotary_tables( + [&positions; 3], + cfg.rotary_half, + cfg.rope_theta, + cfg.mrope_section, + ); // SAFETY: both tables were laid out for cap * rotary_half bfloat16 values. unsafe { cuda::upload(buf.at(cos), &cos_t, stream)?; @@ -496,6 +543,8 @@ impl Scratch { act, cos, sin, + custom_cos, + custom_sin, }) } @@ -632,15 +681,7 @@ impl Model { ) } - /// The final-norm hidden state at the last position, as float32. - pub fn forward(&mut self, ids: &[u32]) -> Result> { - let t = ids.len(); - ensure!(t > 0, "empty prompt"); - let (vocab, h) = (self.embed.shape[0], self.cfg.hidden); - ensure!( - ids.iter().all(|&i| (i as usize) < vocab), - "token id outside the vocabulary" - ); + fn prepare_scratch(&mut self, t: usize) -> Result<()> { cuda::set_device(0)?; if self.scratch.as_ref().is_none_or(|s| t > s.cap) { self.graphs.clear(); @@ -651,16 +692,33 @@ impl Model { self.stream, )?); } + Ok(()) + } + + /// The final-norm hidden state at the last position, as float32. + pub fn forward(&mut self, ids: &[u32]) -> Result> { + let t = ids.len(); + ensure!( + t > 0 && t <= self.cfg.max_positions, + "empty or oversized prompt" + ); + ensure!( + ids.iter().all(|&i| (i as usize) < self.embed.shape[0]), + "token id outside the vocabulary" + ); + self.prepare_scratch(t)?; let s = self.scratch.as_ref().unwrap(); - let ids32: Vec = ids.iter().flat_map(|&i| (i as i32).to_le_bytes()).collect(); - // SAFETY: the ids buffer holds at least t int32 values. - unsafe { cuda::upload(s.at(s.ids), &ids32, self.stream)? }; + self.embed_tokens(s, ids)?; if self.graph_enabled { - if !self.graphs.iter().any(|(length, _)| *length == t) { - // Initialize every cuBLASLt plan before stream capture. - self.run(s, t)?; + if let Some((_, graph)) = self.graphs.iter().find(|(length, _)| *length == t) { + graph.launch(self.stream)?; + } else { + // Warm GEMM plans and keep this eager result for the cache miss. + // run() advances s.res in place and no longer embeds tokens, so + // launching the new graph here would advance the residual twice. + self.run(s, t, false)?; cuda::synchronize(self.stream)?; - match cuda::Graph::capture(self.stream, || self.run(s, t)) { + match cuda::Graph::capture(self.stream, || self.run(s, t, false)) { Ok(graph) => { if self.graphs.len() == 8 { self.graphs.pop_front(); @@ -669,27 +727,111 @@ impl Model { } Err(error) => { // Capture records without executing: the eager result is valid. - // Disable graphs for this worker rather than retrying failures. eprintln!("CUDA Graph capture failed; using eager execution: {error:#}"); self.graph_enabled = false; self.graphs.clear(); } } } - if self.graph_enabled { - self.graphs - .iter() - .find(|(length, _)| *length == t) - .unwrap() - .1 - .launch(self.stream)?; - } } else { - self.run(s, t)?; + self.run(s, t, false)?; + } + self.last_hidden(s, t) + } + + /// Prefill one unpadded prompt with already-adapted BF16 image embeddings and + /// explicit `[3, 1, sequence]` T/H/W positions. No vision tower runs here. + /// Load a checkpoint with the matching multimodal language adapter merged. + pub fn forward_multimodal(&mut self, input: &MultimodalInput<'_>) -> Result> { + let image_token = self + .cfg + .image_token_id + .context("checkpoint has no image_token_id")?; + input.validate( + self.cfg.hidden, + self.embed.shape[0], + image_token, + self.cfg.max_positions, + )?; + let t = input.token_ids.len(); + self.prepare_scratch(t)?; + let s = self.scratch.as_ref().unwrap(); + self.upload_positions(s, input.position_ids)?; + self.embed_tokens(s, input.token_ids)?; + let bytes: Vec = input + .image_embeddings + .iter() + .flat_map(|x| x.to_le_bytes()) + .collect(); + // Coalesce adjacent placeholders. Text rows remain those of embed_tokens. + let indices = input.image_token_indices; + let mut begin = 0; + while begin < indices.len() { + let mut end = begin + 1; + while end < indices.len() && indices[end] == indices[end - 1] + 1 { + end += 1; + } + let row_bytes = self.cfg.hidden * BF16; + // SAFETY: validated indices lie in the t-row residual buffer, and + // features contain exactly one hidden-size BF16 row per placeholder. + unsafe { + cuda::upload( + s.at(s.res + indices[begin] * row_bytes), + &bytes[begin * row_bytes..end * row_bytes], + self.stream, + )?; + } + begin = end; + } + self.run(s, t, true)?; + self.last_hidden(s, t) + } + + fn upload_positions(&self, s: &Scratch, positions: [&[i64]; 3]) -> Result<()> { + let (cos, sin) = rotary_tables( + positions, + self.cfg.rotary_half, + self.cfg.rope_theta, + self.cfg.mrope_section, + ); + // SAFETY: tables contain at most s.cap rows of rotary_half BF16 values. + unsafe { + cuda::upload(s.at(s.custom_cos), &cos, self.stream)?; + cuda::upload(s.at(s.custom_sin), &sin, self.stream)?; + } + Ok(()) + } + + fn embed_tokens(&self, s: &Scratch, ids: &[u32]) -> Result<()> { + let ids32: Vec = ids.iter().flat_map(|&i| (i as i32).to_le_bytes()).collect(); + // SAFETY: IDs were checked against the vocabulary; scratch holds t rows. + unsafe { + cuda::upload(s.at(s.ids), &ids32, self.stream)?; + check( + (cuda::api().cs1_embed)( + s.at(s.ids).cast(), + self.embed.ptr, + s.at(s.res), + ids.len() as i32, + self.cfg.hidden as i32, + self.stream, + ), + "embed", + )?; } - let mut last = vec![0u8; h * BF16]; + Ok(()) + } + + fn last_hidden(&self, s: &Scratch, t: usize) -> Result> { + let mut last = vec![0u8; self.cfg.hidden * BF16]; // SAFETY: x holds at least t rows of the hidden size. - unsafe { cuda::download(&mut last, s.at(s.x + (t - 1) * h * BF16), self.stream)? }; + unsafe { + cuda::download( + &mut last, + s.at(s.x + (t - 1) * self.cfg.hidden * BF16), + self.stream, + )?; + } let (pairs, _) = last.as_chunks::<2>(); Ok(pairs .iter() @@ -697,9 +839,10 @@ impl Model { .collect()) } - /// Queue one forward pass over the first `t` ids in `s`. The final-norm hidden - /// states end up in `s.x`. - fn run(&self, s: &Scratch, t: usize) -> Result<()> { + /// Queue language layers over prepared embeddings in `s.res`, with rotary + /// tables in immutable text buffers or separate explicit-position buffers. + /// Final-norm hidden states end up in `s.x`. + fn run(&self, s: &Scratch, t: usize, custom_positions: bool) -> Result<()> { let cfg = &self.cfg; let st = self.stream; let (ti, hi, eps) = (t as i32, cfg.hidden as i32, cfg.eps); @@ -707,13 +850,14 @@ impl Model { let (hq, hk, hd) = (cfg.heads as i32, cfg.kv_heads as i32, cfg.head_dim as i32); let w = Widths::of(cfg); let p = |off: usize| s.at(off); - // SAFETY (every kernel call below): the pointers are weights in the arena or + let (cos, sin) = if custom_positions { + (s.custom_cos, s.custom_sin) + } else { + (s.cos, s.sin) + }; + // SAFETY (every kernel call below): pointers are weights in the arena or // scratch buffers laid out for at least t tokens with the widths used here. unsafe { - check( - (cuda::api().cs1_embed)(p(s.ids).cast(), self.embed.ptr, p(s.res), ti, hi, st), - "embed", - )?; check( (cuda::api().cs1_rms_norm)( p(s.res), @@ -814,8 +958,8 @@ impl Model { ld, fa.q_norm.ptr, fa.k_norm.ptr, - p(s.cos), - p(s.sin), + p(cos), + p(sin), p(s.aq), p(s.agate), p(s.ak), @@ -911,3 +1055,71 @@ impl Model { Ok(()) } } + +#[cfg(test)] +mod graph_tests { + use super::*; + + #[test] + #[ignore = "needs CUA_S1_MODEL and ABI-3 CUA_S1_CUDA_LIB on a GPU"] + fn graph_misses_hits_eviction_growth_and_multimodal_match_eager() { + let dir = std::path::PathBuf::from(std::env::var_os("CUA_S1_MODEL").unwrap()); + let lib = std::path::PathBuf::from(std::env::var_os("CUA_S1_CUDA_LIB").unwrap()); + let prompts: Vec> = (4..=13) + .chain([1025]) + .map(|t| vec![32 + (t % 3) as u32; t]) + .collect(); + let changed_ids = vec![35; 4]; + let mut eager = Model::load(&dir, &lib).unwrap(); + eager.graph_enabled = false; + let expected: Vec<_> = prompts + .iter() + .map(|ids| eager.forward(ids).unwrap()) + .collect(); + let changed_expected = eager.forward(&changed_ids).unwrap(); + let image_ids = [32, eager.cfg.image_token_id.unwrap(), 33, 34]; + let features = vec![half::bf16::ONE; eager.cfg.hidden]; + let boundary = MultimodalInput { + token_ids: &image_ids, + image_token_indices: &[1], + image_embeddings: &features, + position_ids: [&[0, 1, 2, 3], &[0, 7, 8, 9], &[0, 3, 4, 5]], + }; + let multimodal_expected = eager.forward_multimodal(&boundary).unwrap(); + drop(eager); + + let mut model = Model::load(&dir, &lib).unwrap(); + model.graph_enabled = true; + for (ids, expected) in prompts[..10].iter().zip(&expected) { + // First use returns the eager result, then the graph is a cache hit. + assert_eq!(expected, &model.forward(ids).unwrap()); + assert!( + model.graph_enabled, + "capture unexpectedly fell back to eager" + ); + assert!(model.graphs.iter().any(|(t, _)| *t == ids.len())); + assert_eq!(expected, &model.forward(ids).unwrap()); + } + assert_eq!(model.graphs.len(), 8); + assert!(!model.graphs.iter().any(|(t, _)| *t == 4)); + // Evicted length is another miss; then change token IDs on a warm hit. + assert_eq!(expected[0], model.forward(&prompts[0]).unwrap()); + assert_eq!(changed_expected, model.forward(&changed_ids).unwrap()); + + // The same cached length must still use eager multimodal execution: + // reusing the text graph would read text positions instead of T/H/W. + assert!(model.graphs.iter().any(|(t, _)| *t == image_ids.len())); + assert_eq!( + multimodal_expected, + model.forward_multimodal(&boundary).unwrap() + ); + assert_eq!(expected[0], model.forward(&prompts[0]).unwrap()); + + // Growth invalidates all captures before freeing their device buffers. + assert_eq!(expected[10], model.forward(&prompts[10]).unwrap()); + assert_eq!(model.graphs.len(), 1); + assert_eq!(model.graphs[0].0, 1025); + assert_eq!(expected[10], model.forward(&prompts[10]).unwrap()); + assert_eq!(expected[0], model.forward(&prompts[0]).unwrap()); + } +} diff --git a/src/models/cua_s1/native/tests/multimodal.rs b/src/models/cua_s1/native/tests/multimodal.rs new file mode 100644 index 0000000..20afda5 --- /dev/null +++ b/src/models/cua_s1/native/tests/multimodal.rs @@ -0,0 +1,90 @@ +//! Real CUDA regression: explicit positions must not contaminate later text calls. +use std::path::PathBuf; + +use half::bf16; +use omni_cua_s1_native::{inputs::MultimodalInput, model::Model}; + +#[test] +#[ignore = "needs CUA_S1_MODEL and CUA_S1_CUDA_LIB on a GPU"] +fn text_multimodal_text_keeps_text_positions_and_overwrites_image_rows() { + let dir = PathBuf::from(std::env::var_os("CUA_S1_MODEL").expect("CUA_S1_MODEL")); + let lib = PathBuf::from(std::env::var_os("CUA_S1_CUDA_LIB").expect("CUA_S1_CUDA_LIB")); + let mut model = Model::load(&dir, &lib).unwrap(); + let ids = [32, 33, 34, 35]; + let positions = [0, 1, 2, 3]; + let baseline = model.forward(&ids).unwrap(); + let explicit = model + .forward_multimodal(&MultimodalInput { + token_ids: &ids, + image_token_indices: &[], + image_embeddings: &[], + position_ids: [&positions; 3], + }) + .unwrap(); + assert_eq!(baseline, explicit); + + let image_token = model.cfg.image_token_id.unwrap(); + let image_ids = [32, image_token, image_token, 35]; + let features = vec![bf16::ONE; 2 * model.cfg.hidden]; + let different = model + .forward_multimodal(&MultimodalInput { + token_ids: &image_ids, + image_token_indices: &[1, 2], + image_embeddings: &features, + position_ids: [&[0, 1, 1, 2], &[0, 1, 2, 2], &[0, 2, 1, 2]], + }) + .unwrap(); + assert!(different.iter().all(|x| x.is_finite())); + assert_ne!(baseline, different); + assert_eq!(baseline, model.forward(&ids).unwrap()); + + // Reusing the same layout with changed features must not retain old rows. + let features = vec![bf16::from_f32(-1.0); 2 * model.cfg.hidden]; + let changed = model + .forward_multimodal(&MultimodalInput { + token_ids: &image_ids, + image_token_indices: &[1, 2], + image_embeddings: &features, + position_ids: [&[0, 1, 1, 2], &[0, 1, 2, 2], &[0, 2, 1, 2]], + }) + .unwrap(); + assert_ne!(different, changed); + assert_eq!(baseline, model.forward(&ids).unwrap()); + + // Non-adjacent image spans exercise separate uploads and untouched text rows. + let disjoint_ids = [image_token, 33, image_token, 35]; + let disjoint = MultimodalInput { + token_ids: &disjoint_ids, + image_token_indices: &[0, 2], + image_embeddings: &features, + position_ids: [&positions; 3], + }; + let last = model.forward_multimodal(&disjoint).unwrap(); + assert_eq!(last, model.forward_multimodal(&disjoint).unwrap()); + + // A rejected boundary must not interfere with restoring the next text call. + assert!( + model + .forward_multimodal(&MultimodalInput { + image_token_indices: &[2, 0], + ..disjoint + }) + .is_err() + ); + assert_eq!(baseline, model.forward(&ids).unwrap()); + + // Cross the 1024-row allocation boundary, then use a shorter custom layout + // before reusing the immutable text tables in the larger scratch allocation. + let long_ids = vec![32; 1025]; + let long_text = model.forward(&long_ids).unwrap(); + model + .forward_multimodal(&MultimodalInput { + token_ids: &image_ids, + image_token_indices: &[1, 2], + image_embeddings: &features, + position_ids: [&[0, 1, 1, 2], &[0, 1, 2, 2], &[0, 2, 1, 2]], + }) + .unwrap(); + assert_eq!(long_text, model.forward(&long_ids).unwrap()); + assert_eq!(baseline, model.forward(&ids).unwrap()); +} From f73ce6ade5372a63f8bb85bef3997e1feadef72e Mon Sep 17 00:00:00 2001 From: levius <2114377220@qq.com> Date: Fri, 2 Oct 2026 13:18:06 +0800 Subject: [PATCH 2/2] chore(cua_s1): limit PR diff to core implementation --- recipe/cua_s1/native.md | 112 +---------- .../native/examples/multimodal_boundary.rs | 184 ------------------ src/models/cua_s1/native/src/inputs.rs | 118 ----------- src/models/cua_s1/native/src/model.rs | 68 ------- src/models/cua_s1/native/tests/multimodal.rs | 90 --------- 5 files changed, 3 insertions(+), 569 deletions(-) delete mode 100644 src/models/cua_s1/native/examples/multimodal_boundary.rs delete mode 100644 src/models/cua_s1/native/tests/multimodal.rs diff --git a/recipe/cua_s1/native.md b/recipe/cua_s1/native.md index 35dccb4..018feff 100644 --- a/recipe/cua_s1/native.md +++ b/recipe/cua_s1/native.md @@ -24,11 +24,9 @@ CUA_S1_MODEL=weights/cua-s1-4b-0.2-text-merged target/release/omni-cua-s1-native ``` For the local CUDA Graph experiment, also set `CUA_S1_GRAPH=1`. The first use of -each exact prompt length warms the GEMM plans and captures the forward pass. -The first call returns the eager result; later requests replay it after fresh -token embedding. The graph contains the language layers, which update residuals -in place, so a cache miss must not replay those layers over its eager result. -At most eight lengths are cached. Growing the scratch allocation clears the captures before freeing +each exact prompt length warms the GEMM plans and captures the forward pass; +later requests replay it with freshly uploaded token ids. At most eight lengths +are cached. Growing the scratch allocation clears the captures before freeing their buffers. Capture adds first-use latency; leave the variable unset to use the eager control. Rebuild both the worker and CUDA library together (ABI 3). If capture fails, the worker returns the completed eager result and disables @@ -43,107 +41,3 @@ cargo test -p omni-cua-s1-native CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ cargo test --release -p omni-cua-s1-native --test kernels -- --ignored ``` - -## Multimodal language boundary - -`Model::forward_multimodal` consumes token IDs, adapted BF16 image features -`[image_tokens, hidden_size]`, their sorted placeholder indices, and the T/H/W -slices of int64 `position_ids [3, 1, sequence]`. It returns the last position's -final-normalized hidden state, like `Model::forward`. - -Inputs are one unpadded prompt. Every image placeholder must have exactly one -feature row; all other rows come from the token embedding table. The caller -calculates positions and runs the image processor, vision tower and vision LoRA. -Positions must be nonnegative and below `max_position_embeddings`. The language -path uses Qwen3.5's interleaved MRoPE sections, not three contiguous rotary blocks. -Text calls and their captured graphs use immutable text-position tables; -multimodal calls use separate device tables, so returning to text requires no -host table rebuild or restoration copy. Multimodal calls execute eagerly even -when `CUA_S1_GRAPH=1`. The extra tables use -`4 * scratch_capacity * rotary_half` bytes (2 MiB at 16,384 rows). - -This is a Rust model API for integrating a vision producer. The HTTP worker -above continues to serve the text adapter. A native vision encoder, image HTTP -requests, padding, video and batching are not implemented by this API. - -### Prepare a matching language checkpoint - -The language weights must contain the **multimodal** adapter, not the `text` -adapter. In the pinned reference environment, with upstream-verified weights, -export just the merged language model (about 7.5 GB) to a new directory: - -```sh -PYTHONPATH=src HF_HUB_OFFLINE=1 .venv/bin/python - <<'PY' -import json -from pathlib import Path -from models.cua_s1.multimodal.model import ( - ADAPTER_REVISION, BASE_REVISION, MultimodalEngine, -) - -out = Path("weights/cua-s1-4b-0.2-multimodal-language-merged") -if out.exists(): - raise FileExistsError(out) -engine = MultimodalEngine( - "weights/Qwen3.5-4B", "weights/cua-s1-4b-0.2/multimodal" -) -merged = engine.model.merge_and_unload() -merged.model.language_model.save_pretrained(out, max_shard_size="5GB") -# Preserve the root image_token_id and text_config for the native input contract. -merged.config.to_json_file(out / "config.json") -(out / "cua_s1_language_export.json").write_text(json.dumps({ - "format": "cua-s1-multimodal-language-merged/1", - "base_revision": BASE_REVISION, - "adapter_revision": ADAPTER_REVISION, -})) -PY -``` - -No `cua_s1_export.json` text-worker marker is created. The low-level `Model` API -does not verify checkpoint provenance; retain the export metadata and use the -matching adapter for the supplied features. Standalone language safetensors -names and the existing full-model prefixes are supported. - -### Replay a reference boundary - -This optional example consumes the `cua-s1-multimodal-reference-v1` format -from [#53](https://github.com/ThinkFlowLab/system1-omni/pull/53), which is still -open. The exporter and checksum verifier are not yet available on `main`. -Use a separate checkout of exporter revision -`1b64fa2ceb0a82b6a66a69ecdc9bc5cc1b1a0b66` to generate and verify the bundle; -the Rust model API itself does not depend on that PR being merged. -Its eight questions include different image grids and question lengths, JPEG, -structured/non-ASCII text, and 1/3/26 candidates. Then run: - -```sh -CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ - cargo run --release --locked -p omni-cua-s1-native \ - --example multimodal_boundary -- \ - weights/cua-s1-4b-0.2-multimodal-language-merged \ - /path/to/verified-reference-bundle /tmp/native-language.json - -CUA_S1_MODEL=$PWD/weights/cua-s1-4b-0.2-multimodal-language-merged \ -CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ - cargo test --release --locked -p omni-cua-s1-native \ - --test multimodal -- --ignored - -# Compare graph misses, hits, eviction and scratch growth with eager hidden states. -CUA_S1_MODEL=$PWD/weights/cua-s1-4b-0.2-multimodal-language-merged \ -CUA_S1_CUDA_LIB=$PWD/target/release/libqwen3_5_cuda.so \ - cargo test --release --locked -p omni-cua-s1-native \ - --lib graph_tests::graph_misses_hits_eviction_growth_and_multimodal_match_eager -- --ignored -``` - -The example checks repeated native hidden-state equality and writes last hidden -states, candidate logits and probabilities. It uses the text engine's FP32 -letter-row readout with FP64 accumulation. Output must be a new file. Verify -bundle integrity before invoking the example; it checks tensor shapes and input -contracts but is not the bundle checksum verifier. - -For accuracy validation, compare against an unmerged FP32 **language** control -with TF32 disabled, feeding the same fixed exported embeddings and positions. -Use the [declared native tolerance](../../src/models/cua_s1/README.md#validation): -maximum probability error over the set must be at most twice the BF16 reference -error plus 0.01, and the top option must match for FP32 margins at least 0.05. -The FP32 control starts after the BF16-exported vision boundary; it does not -validate a full FP32 vision pipeline. Native kernel and LoRA-merge rounding can -change hidden states and logits; bitwise equality to Transformers is not claimed. diff --git a/src/models/cua_s1/native/examples/multimodal_boundary.rs b/src/models/cua_s1/native/examples/multimodal_boundary.rs deleted file mode 100644 index 7c93fb4..0000000 --- a/src/models/cua_s1/native/examples/multimodal_boundary.rs +++ /dev/null @@ -1,184 +0,0 @@ -//! Replay #53's exported language boundary through the native model. -//! The v1 bundle producer/verifier are on open PR #53 at exporter revision -//! 1b64fa2ceb0a82b6a66a69ecdc9bc5cc1b1a0b66; see recipe/cua_s1/native.md. -use std::path::{Path, PathBuf}; - -use anyhow::{Context, Result, ensure}; -use half::bf16; -use omni_cua_s1_native::{inputs::MultimodalInput, model::Model}; -use safetensors::{Dtype, SafeTensors}; -use serde_json::{Value, json}; - -fn integers(st: &SafeTensors<'_>, name: &str, shape: &[usize]) -> Result> { - let v = st.tensor(name)?; - ensure!( - v.dtype() == Dtype::I64 && v.shape() == shape, - "{name}: expected I64 {shape:?}" - ); - Ok(v.data() - .as_chunks::<8>() - .0 - .iter() - .map(|b| i64::from_le_bytes(*b)) - .collect()) -} - -fn features(st: &SafeTensors<'_>, shape: &[usize]) -> Result> { - let v = st.tensor("image_features")?; - ensure!( - v.dtype() == Dtype::BF16 && v.shape() == shape, - "image_features: expected BF16 {shape:?}" - ); - Ok(v.data() - .as_chunks::<2>() - .0 - .iter() - .map(|b| bf16::from_le_bytes(*b)) - .collect()) -} - -fn embedding_file(dir: &Path) -> Result<(PathBuf, String)> { - let names = [ - "model.language_model.embed_tokens.weight", - "model.embed_tokens.weight", - "embed_tokens.weight", - ]; - if dir.join("model.safetensors.index.json").exists() { - let index: Value = - serde_json::from_slice(&std::fs::read(dir.join("model.safetensors.index.json"))?)?; - for name in names { - if let Some(file) = index["weight_map"][name].as_str() { - return Ok((dir.join(file), name.into())); - } - } - } else { - let path = dir.join("model.safetensors"); - let bytes = std::fs::read(&path)?; - let st = SafeTensors::deserialize(&bytes)?; - for name in names { - if st.tensor(name).is_ok() { - return Ok((path, name.into())); - } - } - } - anyhow::bail!("missing embedding weight") -} - -fn main() -> Result<()> { - let args: Vec<_> = std::env::args_os().skip(1).collect(); - ensure!( - args.len() == 3, - "usage: multimodal_boundary MODEL_DIR REFERENCE_BUNDLE OUTPUT_JSON" - ); - let (dir, bundle, out) = ( - Path::new(&args[0]), - Path::new(&args[1]), - Path::new(&args[2]), - ); - ensure!(!out.exists(), "output already exists"); - let library = PathBuf::from(std::env::var_os("CUA_S1_CUDA_LIB").context("CUA_S1_CUDA_LIB")?); - let manifest: Value = serde_json::from_slice(&std::fs::read(bundle.join("manifest.json"))?)?; - ensure!( - manifest["schema"] == "cua-s1-multimodal-reference-v1", - "unsupported reference schema" - ); - let mut model = Model::load(dir, &library)?; - let (file, name) = embedding_file(dir)?; - let file = std::fs::File::open(file)?; - // SAFETY: the checkpoint is immutable while the example runs. - let map = unsafe { memmap2::Mmap::map(&file)? }; - let weights = SafeTensors::deserialize(&map)?; - let embed = weights.tensor(&name)?; - ensure!( - embed.dtype() == Dtype::BF16 - && embed.shape().len() == 2 - && embed.shape()[1] == model.cfg.hidden, - "embedding shape/dtype" - ); - let mut rows = Vec::new(); - for entry in manifest["questions"].as_array().context("questions")? { - let relative = Path::new(entry["tensors_file"].as_str().context("tensors_file")?); - ensure!( - relative - .components() - .all(|c| matches!(c, std::path::Component::Normal(_))), - "unsafe tensor path" - ); - let bytes = std::fs::read(bundle.join(relative))?; - let st = SafeTensors::deserialize(&bytes)?; - let ids = st.tensor("input_ids")?; - ensure!( - ids.shape().len() == 2 && ids.shape()[0] == 1, - "expected batch one" - ); - let t = ids.shape()[1]; - let ids: Vec = integers(&st, "input_ids", &[1, t])? - .into_iter() - .map(u32::try_from) - .collect::>()?; - let indices = st.tensor("image_token_indices")?; - ensure!( - indices.shape().len() == 1, - "image indices must be one-dimensional" - ); - let count = indices.shape()[0]; - let indices: Vec = integers(&st, "image_token_indices", &[count])? - .into_iter() - .map(usize::try_from) - .collect::>()?; - let features = features(&st, &[count, model.cfg.hidden])?; - let positions = integers(&st, "position_ids", &[3, 1, t])?; - let input = MultimodalInput { - token_ids: &ids, - image_token_indices: &indices, - image_embeddings: &features, - position_ids: [&positions[..t], &positions[t..2 * t], &positions[2 * t..]], - }; - let last = model.forward_multimodal(&input)?; - ensure!( - last.iter().all(|x| x.is_finite()), - "non-finite hidden state" - ); - // Repeating the same boundary in the same model checks buffer reuse. - ensure!( - last == model.forward_multimodal(&input)?, - "repeat changed native hidden state" - ); - let n = entry["option_keys"] - .as_array() - .context("option_keys")? - .len(); - ensure!((1..=26).contains(&n), "candidate count"); - let candidates = integers(&st, "candidate_token_ids", &[n])?; - let mut logits = Vec::new(); - for id in candidates { - let id = usize::try_from(id)?; - ensure!(id < embed.shape()[0], "candidate outside vocabulary"); - let row = &embed.data()[id * last.len() * 2..(id + 1) * last.len() * 2]; - let dot: f64 = row - .as_chunks::<2>() - .0 - .iter() - .zip(&last) - .map(|(b, &h)| bf16::from_le_bytes(*b).to_f64() * h as f64) - .sum(); - logits.push(dot as f32); - } - let max = logits.iter().copied().fold(f32::NEG_INFINITY, f32::max) as f64; - let exps: Vec = logits.iter().map(|&l| (l as f64 - max).exp()).collect(); - let total: f64 = exps.iter().sum(); - let probabilities: Vec = exps.iter().map(|e| (e / total) as f32).collect(); - ensure!( - probabilities.iter().all(|x| x.is_finite()), - "non-finite readout" - ); - rows.push(json!({"case": entry["case"], "question": entry["question"], "sequence": t, "image_tokens": count, "last_hidden_state": last, "candidate_logits": logits, "probabilities": probabilities, "repeat_equal": true})); - } - std::fs::write( - out, - serde_json::to_vec_pretty( - &json!({"schema": "cua-s1-native-language-boundary-v1", "questions": rows}), - )?, - )?; - Ok(()) -} diff --git a/src/models/cua_s1/native/src/inputs.rs b/src/models/cua_s1/native/src/inputs.rs index 1f0c980..bf719ec 100644 --- a/src/models/cua_s1/native/src/inputs.rs +++ b/src/models/cua_s1/native/src/inputs.rs @@ -93,121 +93,3 @@ pub(crate) fn rotary_tables( } (cos, sin) } - -#[cfg(test)] -mod tests { - use super::*; - use half::bf16; - - #[test] - fn valid_input_and_three_distinct_axes() { - let input = MultimodalInput { - token_ids: &[1, 99, 99, 2], - image_token_indices: &[1, 2], - image_embeddings: &[bf16::ONE; 8], - position_ids: [&[0, 1, 1, 3], &[0, 1, 2, 3], &[0, 2, 1, 3]], - }; - input.validate(4, 100, 99, 100).unwrap(); - } - - #[test] - fn rejects_bad_placeholder_inventory_and_feature_rows() { - for indices in [vec![2, 1], vec![1, 1], vec![1], vec![0, 1], vec![1, 4]] { - let input = MultimodalInput { - token_ids: &[1, 99, 99, 2], - image_token_indices: &indices, - image_embeddings: &[bf16::ONE; 8], - position_ids: [&[0, 1, 1, 3]; 3], - }; - assert!(input.validate(4, 100, 99, 100).is_err(), "{indices:?}"); - } - for features in [vec![bf16::ONE; 7], vec![bf16::ONE; 9], vec![bf16::NAN; 8]] { - let input = MultimodalInput { - token_ids: &[1, 99, 99, 2], - image_token_indices: &[1, 2], - image_embeddings: &features, - position_ids: [&[0, 1, 1, 3]; 3], - }; - assert!(input.validate(4, 100, 99, 100).is_err()); - } - } - - #[test] - fn rejects_bad_tokens_positions_and_empty_sequence() { - let features = [bf16::ONE; 4]; - for (ids, positions) in [ - (vec![100, 99], vec![0, 1]), - (vec![1, 99], vec![0]), - (vec![1, 99], vec![0, -1]), - (vec![1, 99], vec![0, 100]), - (vec![], vec![]), - ] { - let input = MultimodalInput { - token_ids: &ids, - image_token_indices: &[1], - image_embeddings: &features, - position_ids: [&positions; 3], - }; - assert!(input.validate(4, 100, 99, 100).is_err()); - } - } - - #[test] - fn image_free_explicit_positions_are_valid() { - MultimodalInput { - token_ids: &[1, 2], - image_token_indices: &[], - image_embeddings: &[], - position_ids: [&[7, 8]; 3], - } - .validate(4, 100, 99, 100) - .unwrap(); - } - - #[test] - fn rotary_interleaves_height_width_and_leaves_temporal_tail() { - // theta=1 makes every inverse frequency 1. Axis values differ so a plain - // text table or a contiguous-section implementation fails this check. - for (sections, tail) in [([12, 10, 10], [0, 0]), ([11, 11, 10], [0, 1])] { - let (cos, sin) = rotary_tables([&[0], &[1], &[2]], 32, 1.0, sections); - // Ten T/H/W triples, then T/T for the synthetic layout or T/H for - // the real checkpoint. In particular, frequency 31 must use H. - let axes = [ - 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, 1, 2, 0, - 1, 2, tail[0], tail[1], - ]; - for (i, axis) in axes.into_iter().enumerate() { - let angle = axis as f32; - assert_eq!( - &cos[i * 2..i * 2 + 2], - &bf16::from_f32(angle.cos()).to_le_bytes() - ); - assert_eq!( - &sin[i * 2..i * 2 + 2], - &bf16::from_f32(angle.sin()).to_le_bytes() - ); - } - } - } - - #[test] - fn equal_axes_reproduce_the_existing_text_tables() { - let pos: Vec = (0..257).collect(); - let (cos, sin) = rotary_tables([&pos; 3], 32, 10_000_000.0, [11, 11, 10]); - for (t, &p) in pos.iter().enumerate() { - for i in 0..32 { - let inv = 1f32 / 10_000_000f32.powf((2 * i) as f32 / 64.0); - let angle = (inv * p as f32) as f64; - let offset = (t * 32 + i) * 2; - assert_eq!( - &cos[offset..offset + 2], - &bf16::from_f32(angle.cos() as f32).to_le_bytes() - ); - assert_eq!( - &sin[offset..offset + 2], - &bf16::from_f32(angle.sin() as f32).to_le_bytes() - ); - } - } - } -} diff --git a/src/models/cua_s1/native/src/model.rs b/src/models/cua_s1/native/src/model.rs index d6d6597..ab4439a 100644 --- a/src/models/cua_s1/native/src/model.rs +++ b/src/models/cua_s1/native/src/model.rs @@ -1055,71 +1055,3 @@ impl Model { Ok(()) } } - -#[cfg(test)] -mod graph_tests { - use super::*; - - #[test] - #[ignore = "needs CUA_S1_MODEL and ABI-3 CUA_S1_CUDA_LIB on a GPU"] - fn graph_misses_hits_eviction_growth_and_multimodal_match_eager() { - let dir = std::path::PathBuf::from(std::env::var_os("CUA_S1_MODEL").unwrap()); - let lib = std::path::PathBuf::from(std::env::var_os("CUA_S1_CUDA_LIB").unwrap()); - let prompts: Vec> = (4..=13) - .chain([1025]) - .map(|t| vec![32 + (t % 3) as u32; t]) - .collect(); - let changed_ids = vec![35; 4]; - let mut eager = Model::load(&dir, &lib).unwrap(); - eager.graph_enabled = false; - let expected: Vec<_> = prompts - .iter() - .map(|ids| eager.forward(ids).unwrap()) - .collect(); - let changed_expected = eager.forward(&changed_ids).unwrap(); - let image_ids = [32, eager.cfg.image_token_id.unwrap(), 33, 34]; - let features = vec![half::bf16::ONE; eager.cfg.hidden]; - let boundary = MultimodalInput { - token_ids: &image_ids, - image_token_indices: &[1], - image_embeddings: &features, - position_ids: [&[0, 1, 2, 3], &[0, 7, 8, 9], &[0, 3, 4, 5]], - }; - let multimodal_expected = eager.forward_multimodal(&boundary).unwrap(); - drop(eager); - - let mut model = Model::load(&dir, &lib).unwrap(); - model.graph_enabled = true; - for (ids, expected) in prompts[..10].iter().zip(&expected) { - // First use returns the eager result, then the graph is a cache hit. - assert_eq!(expected, &model.forward(ids).unwrap()); - assert!( - model.graph_enabled, - "capture unexpectedly fell back to eager" - ); - assert!(model.graphs.iter().any(|(t, _)| *t == ids.len())); - assert_eq!(expected, &model.forward(ids).unwrap()); - } - assert_eq!(model.graphs.len(), 8); - assert!(!model.graphs.iter().any(|(t, _)| *t == 4)); - // Evicted length is another miss; then change token IDs on a warm hit. - assert_eq!(expected[0], model.forward(&prompts[0]).unwrap()); - assert_eq!(changed_expected, model.forward(&changed_ids).unwrap()); - - // The same cached length must still use eager multimodal execution: - // reusing the text graph would read text positions instead of T/H/W. - assert!(model.graphs.iter().any(|(t, _)| *t == image_ids.len())); - assert_eq!( - multimodal_expected, - model.forward_multimodal(&boundary).unwrap() - ); - assert_eq!(expected[0], model.forward(&prompts[0]).unwrap()); - - // Growth invalidates all captures before freeing their device buffers. - assert_eq!(expected[10], model.forward(&prompts[10]).unwrap()); - assert_eq!(model.graphs.len(), 1); - assert_eq!(model.graphs[0].0, 1025); - assert_eq!(expected[10], model.forward(&prompts[10]).unwrap()); - assert_eq!(expected[0], model.forward(&prompts[0]).unwrap()); - } -} diff --git a/src/models/cua_s1/native/tests/multimodal.rs b/src/models/cua_s1/native/tests/multimodal.rs deleted file mode 100644 index 20afda5..0000000 --- a/src/models/cua_s1/native/tests/multimodal.rs +++ /dev/null @@ -1,90 +0,0 @@ -//! Real CUDA regression: explicit positions must not contaminate later text calls. -use std::path::PathBuf; - -use half::bf16; -use omni_cua_s1_native::{inputs::MultimodalInput, model::Model}; - -#[test] -#[ignore = "needs CUA_S1_MODEL and CUA_S1_CUDA_LIB on a GPU"] -fn text_multimodal_text_keeps_text_positions_and_overwrites_image_rows() { - let dir = PathBuf::from(std::env::var_os("CUA_S1_MODEL").expect("CUA_S1_MODEL")); - let lib = PathBuf::from(std::env::var_os("CUA_S1_CUDA_LIB").expect("CUA_S1_CUDA_LIB")); - let mut model = Model::load(&dir, &lib).unwrap(); - let ids = [32, 33, 34, 35]; - let positions = [0, 1, 2, 3]; - let baseline = model.forward(&ids).unwrap(); - let explicit = model - .forward_multimodal(&MultimodalInput { - token_ids: &ids, - image_token_indices: &[], - image_embeddings: &[], - position_ids: [&positions; 3], - }) - .unwrap(); - assert_eq!(baseline, explicit); - - let image_token = model.cfg.image_token_id.unwrap(); - let image_ids = [32, image_token, image_token, 35]; - let features = vec![bf16::ONE; 2 * model.cfg.hidden]; - let different = model - .forward_multimodal(&MultimodalInput { - token_ids: &image_ids, - image_token_indices: &[1, 2], - image_embeddings: &features, - position_ids: [&[0, 1, 1, 2], &[0, 1, 2, 2], &[0, 2, 1, 2]], - }) - .unwrap(); - assert!(different.iter().all(|x| x.is_finite())); - assert_ne!(baseline, different); - assert_eq!(baseline, model.forward(&ids).unwrap()); - - // Reusing the same layout with changed features must not retain old rows. - let features = vec![bf16::from_f32(-1.0); 2 * model.cfg.hidden]; - let changed = model - .forward_multimodal(&MultimodalInput { - token_ids: &image_ids, - image_token_indices: &[1, 2], - image_embeddings: &features, - position_ids: [&[0, 1, 1, 2], &[0, 1, 2, 2], &[0, 2, 1, 2]], - }) - .unwrap(); - assert_ne!(different, changed); - assert_eq!(baseline, model.forward(&ids).unwrap()); - - // Non-adjacent image spans exercise separate uploads and untouched text rows. - let disjoint_ids = [image_token, 33, image_token, 35]; - let disjoint = MultimodalInput { - token_ids: &disjoint_ids, - image_token_indices: &[0, 2], - image_embeddings: &features, - position_ids: [&positions; 3], - }; - let last = model.forward_multimodal(&disjoint).unwrap(); - assert_eq!(last, model.forward_multimodal(&disjoint).unwrap()); - - // A rejected boundary must not interfere with restoring the next text call. - assert!( - model - .forward_multimodal(&MultimodalInput { - image_token_indices: &[2, 0], - ..disjoint - }) - .is_err() - ); - assert_eq!(baseline, model.forward(&ids).unwrap()); - - // Cross the 1024-row allocation boundary, then use a shorter custom layout - // before reusing the immutable text tables in the larger scratch allocation. - let long_ids = vec![32; 1025]; - let long_text = model.forward(&long_ids).unwrap(); - model - .forward_multimodal(&MultimodalInput { - token_ids: &image_ids, - image_token_indices: &[1, 2], - image_embeddings: &features, - position_ids: [&[0, 1, 1, 2], &[0, 1, 2, 2], &[0, 2, 1, 2]], - }) - .unwrap(); - assert_eq!(long_text, model.forward(&long_ids).unwrap()); - assert_eq!(baseline, model.forward(&ids).unwrap()); -}