Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
44 commits
Select commit Hold shift + click to select a range
74c95e5
Add Cua-S1 multimodal CUDA worker and parity recipe
Levius-Fubuki Sep 27, 2026
ee970c2
Merge main and align multimodal worker error contract
Levius-Fubuki Sep 28, 2026
767e211
Add reproducible multimodal profiling matrix
Levius-Fubuki Sep 28, 2026
ebd7adc
Distinguish profiler device annotations and support trace-only runs
Levius-Fubuki Sep 28, 2026
4c605e3
Record RTX 4090 profiling results and trace recovery
Levius-Fubuki Sep 28, 2026
159789a
docs: plan request-local image reuse experiment
Levius-Fubuki Sep 28, 2026
8c3aa20
test: add paired image reuse experiment and correctness checks
Levius-Fubuki Sep 28, 2026
7e7ed34
test: preserve unoptimized profiling baseline after reuse
Levius-Fubuki Sep 28, 2026
7874c6c
style: simplify profiling test environment stub
Levius-Fubuki Sep 28, 2026
cf899ff
Reuse image preprocessing and adapted vision features within requests
Levius-Fubuki Sep 28, 2026
c18a21b
test: cover multi-question JPEG reuse and document tensor contract
Levius-Fubuki Sep 28, 2026
4bb4035
Test image reuse cleanup after inference failures
Levius-Fubuki Sep 28, 2026
0abf31d
Avoid body-write race in transport rejection tests
Levius-Fubuki Sep 28, 2026
d0c4781
docs: link request reuse behavior and record completed reviews
Levius-Fubuki Sep 28, 2026
209a660
test: add reproducible result audit and CUDA HTTP postflight
Levius-Fubuki Sep 28, 2026
9f4a4fe
test: audit exact coverage of the paired workload matrix
Levius-Fubuki Sep 28, 2026
ccaa4fb
docs: record paired RTX 4090 image reuse results and exact parity
Levius-Fubuki Sep 28, 2026
95713a5
docs: record PR publication and verified experiment server shutdown
Levius-Fubuki Sep 28, 2026
224d88b
perf(cua-s1): project only final logits in reused path
Levius-Fubuki Sep 28, 2026
89e35e9
docs(cua-s1): publish RTX 4090 final-logits experiment
Levius-Fubuki Sep 28, 2026
c522e00
docs(cua-s1): record experiment publication and server shutdown
Levius-Fubuki Sep 28, 2026
a2abe7e
bench(cua-s1): measure multimodal CUDA Graph language forward
Levius-Fubuki Sep 28, 2026
809cacc
bench(cua-s1): validate graph replay with changed question text
Levius-Fubuki Sep 28, 2026
74b8c9e
docs(cua-s1): publish multimodal CUDA Graph evidence
Levius-Fubuki Sep 28, 2026
e6cf582
Merge main and fix multimodal validation review findings
Levius-Fubuki Sep 29, 2026
04acf99
Merge reviewed multimodal validation fixes into profiling
Levius-Fubuki Sep 29, 2026
642d14c
Merge reviewed validation and cleanup tests into image reuse
Levius-Fubuki Sep 29, 2026
068eb52
Sync reviewed validation fixes into post-reuse profiling
Levius-Fubuki Sep 29, 2026
8d20c44
Sync reviewed validation fixes into graph experiment
Levius-Fubuki Sep 29, 2026
63edd9b
chore: remove local experiment artifacts from PR
Levius-Fubuki Sep 29, 2026
d608d39
chore: remove local experiment artifacts from PR
Levius-Fubuki Sep 29, 2026
88034eb
chore: remove local experiment artifacts from PR
Levius-Fubuki Sep 29, 2026
0ed9563
chore: remove local experiment artifacts from PR
Levius-Fubuki Sep 29, 2026
1e3df30
chore: remove local experiment artifacts from PR
Levius-Fubuki Sep 29, 2026
9093fd8
Move Cua-S1 HTTP serving out of models and reuse upstream weight mani…
Levius-Fubuki Sep 29, 2026
138c4d1
Sync reviewed worker layout and weight manifest cleanup
Levius-Fubuki Sep 29, 2026
66d7cf5
Sync reviewed worker layout and weight manifest cleanup
Levius-Fubuki Sep 29, 2026
70dad54
Sync reviewed worker layout and weight manifest cleanup
Levius-Fubuki Sep 29, 2026
baaf32f
Sync reviewed worker layout and weight manifest cleanup
Levius-Fubuki Sep 29, 2026
934e1ad
chore: limit PR to multimodal runtime code
Levius-Fubuki Sep 30, 2026
f298bbd
chore: limit PR to multimodal runtime code
Levius-Fubuki Sep 30, 2026
1a339e1
chore: limit PR to multimodal runtime code
Levius-Fubuki Sep 30, 2026
378d101
chore: limit PR to multimodal runtime code
Levius-Fubuki Sep 30, 2026
75aa051
chore: limit PR to multimodal runtime code
Levius-Fubuki Sep 30, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
136 changes: 136 additions & 0 deletions src/frontend/cua_s1.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,136 @@
"""Small loopback HTTP worker; the Rust frontend remains the public serving layer."""

from __future__ import annotations

import argparse
import json
import logging
import socket
import threading
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer

from models.cua_s1.multimodal.protocol import (
MAX_BODY,
InvalidRequest,
MalformedJSON,
decode_request,
parse_request,
)

LOG = logging.getLogger(__name__)


class WorkerServer(ThreadingHTTPServer):
daemon_threads = True

def __init__(self, address, engine):
self.engine = engine
self.inference_lock = threading.Lock()
super().__init__(address, Handler)


class Handler(BaseHTTPRequestHandler):
def setup(self):
super().setup()
self.connection.settimeout(15)

def log_message(self, format, *args):
# Do not log paths, input images, instructions or arbitrary request headers.
pass

def send_json(self, status, value):
raw = json.dumps(value, ensure_ascii=False, allow_nan=False).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(raw)))
self.end_headers()
try:
self.wfile.write(raw)
except (BrokenPipeError, ConnectionResetError):
pass

def do_GET(self):
if self.path == "/health":
self.send_json(200, {"status": "ready", "modality": "multimodal"})
else:
self.send_json(404, {"detail": "unknown route"})

def do_POST(self):
if self.path != "/v1/systemone":
self.send_json(404, {"detail": "unknown route"})
return
if self.headers.get("Transfer-Encoding"):
self.send_json(
411,
{
"detail": "Content-Length is required; chunked requests are unsupported"
},
)
return
lengths = self.headers.get_all("Content-Length", [])
if len(lengths) != 1:
self.send_json(411, {"detail": "one Content-Length is required"})
return
try:
length = int(lengths[0])
except ValueError:
self.send_json(400, {"detail": "invalid Content-Length"})
return
if length < 0 or length > MAX_BODY:
self.send_json(413, {"detail": "request exceeds body limit"})
return
if self.headers.get_content_type() != "application/json":
self.send_json(415, {"detail": "Content-Type must be application/json"})
return
if not self.server.inference_lock.acquire(blocking=False):
self.send_json(503, {"detail": "worker busy"})
return
try:
raw = self.rfile.read(length)
if len(raw) != length:
self.send_json(400, {"detail": "incomplete body"})
return
parsed = parse_request(decode_request(raw))
result = self.server.engine.predict(parsed)
self.send_json(200, result)
except MalformedJSON as exc:
self.send_json(400, {"detail": str(exc)})
except InvalidRequest as exc:
self.send_json(422, {"detail": str(exc)})
except (TimeoutError, socket.timeout):
self.send_json(408, {"detail": "request body timed out"})
except Exception as exc:
LOG.error("inference failed: %s", type(exc).__name__)
self.send_json(500, {"detail": "inference failed"})
finally:
self.server.inference_lock.release()


def main():
from models.cua_s1.multimodal.model import MultimodalEngine

p = argparse.ArgumentParser(description=__doc__)
p.add_argument(
"--base", required=True, help="verified local base checkpoint directory"
)
p.add_argument(
"--adapter", required=True, help="verified local multimodal adapter directory"
)
p.add_argument("--port", type=int, default=8000)
args = p.parse_args()
logging.basicConfig(level=logging.INFO)
engine = MultimodalEngine(args.base, args.adapter)
engine.warmup()
# Bind only after model loading and a representative inference succeed.
server = WorkerServer(("127.0.0.1", args.port), engine)
LOG.info("multimodal worker ready on 127.0.0.1:%s", args.port)
try:
server.serve_forever()
except KeyboardInterrupt:
pass
finally:
server.server_close()


if __name__ == "__main__":
main()
Loading