Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,8 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver
- Windows development checks: the smoke script accepts CRLF output, shell scripts and Git hooks retain LF
line endings, and tests check socket closure and invalid output directories without Unix-specific behavior.
The core CI matrix now covers Windows with Python 3.11.
- Rail, tool, and browser evaluations charge Jev-rate input tokens only when the decision backend declares
them billable. Local model token usage remains recorded without Jev API charges.

### Added

Expand Down
5 changes: 3 additions & 2 deletions s1a/browser/decision_model.py
Original file line number Diff line number Diff line change
Expand Up @@ -773,9 +773,10 @@ def report(self) -> dict[str, Any]:
"interactions": len([h for h in run.history if h["kind"] != "wait"]),
"waits": len([h for h in run.history if h["kind"] == "wait"]),
"median_decision_ms": int(statistics.median(jev)) if jev else 0,
# Laya and Cua run in process: their tokens are free and unpriced (s1a/tool/loop.py does the same).
"jev_input_tokens": (
sum(int(t.get("input_tokens") or 0) for t in run.ticks) if self._decision_model.name == "jev" else 0
sum(int(t.get("input_tokens") or 0) for t in run.ticks)
if self._decision_model.bills_input_tokens
else 0
),
"median_probe_ms": int(statistics.median(t["probe_ms"] for t in run.ticks)) if run.ticks else 0,
"settle_probes": sum(t.get("settle_probes", 0) for t in run.ticks),
Expand Down
1 change: 1 addition & 0 deletions s1a/decision_models/base.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,6 +41,7 @@ class DecisionModel(ABC):
"""A model that reads an observation and a discrete action space and returns a distribution over it."""

name: str = "decision_model" # the ``--model`` value; lands in every tick's ``source``
bills_input_tokens: bool # True when input tokens are priced at JEV_USD_PER_INPUT_TOKEN
supports_images: bool = False
question_types: frozenset[str] = frozenset({"choice", "noul"})
deterministic: bool = False # the same request always gets the same answer, so a re-ask is a wasted call
Expand Down
2 changes: 2 additions & 0 deletions s1a/decision_models/baselines.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@ class RandomModel(DecisionModel):
"""Uniform over the offered keys: the loop-overhead arm. Answers choice questions only."""

name = "random"
bills_input_tokens = False
question_types = frozenset({"choice"})

def __init__(self, seed: int) -> None:
Expand Down Expand Up @@ -43,6 +44,7 @@ class RuleModel(DecisionModel):
decision failure.
"""

bills_input_tokens = False
question_types = frozenset({"choice"})
deterministic = True

Expand Down
1 change: 1 addition & 0 deletions s1a/decision_models/cua.py
Original file line number Diff line number Diff line change
Expand Up @@ -70,6 +70,7 @@ class CuaS1Model(DecisionModel):
"""Cua-S1 Nano's ``NanoScorer`` behind the interface: choice questions only, text only, deterministic."""

name = "cua"
bills_input_tokens = False
question_types = frozenset({"choice"})
deterministic = True

Expand Down
1 change: 1 addition & 0 deletions s1a/decision_models/fakes.py
Original file line number Diff line number Diff line change
Expand Up @@ -64,6 +64,7 @@ class ScriptedModel(DecisionModel):
images, so a pass-through can be tested."""

name = "scripted"
bills_input_tokens = False
supports_images = True

def __init__(
Expand Down
1 change: 1 addition & 0 deletions s1a/decision_models/jev.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@ class JevModel(DecisionModel):
"""TypeSafe Jev over HTTP; the transport owns the connection, its retries and the round-trip clock."""

name = "jev"
bills_input_tokens = True

def __init__(self, transport: JevTransport) -> None:
self._transport = transport
Expand Down
1 change: 1 addition & 0 deletions s1a/decision_models/laya.py
Original file line number Diff line number Diff line change
Expand Up @@ -51,6 +51,7 @@ class LayaModel(DecisionModel):
"""Laya's ``Agent`` (or anything with ``system_one(state, questions)`` and a ``cfg``) behind the interface."""

name = "laya"
bills_input_tokens = False
deterministic = True

def __init__(self, agent: Any, *, model: str) -> None:
Expand Down
2 changes: 1 addition & 1 deletion s1a/rails.py
Original file line number Diff line number Diff line change
Expand Up @@ -120,7 +120,7 @@ async def evaluate(
tp = sum(a and label for a, label in zip(acted, labels))
fp = sum(a and not label for a, label in zip(acted, labels))
fn = sum(label and not a for a, label in zip(acted, labels))
jev_input_tokens = sum(verdict.input_tokens for verdict in verdicts)
jev_input_tokens = sum(verdict.input_tokens for verdict in verdicts) if decision_model.bills_input_tokens else 0
summary = {
"rail": spec.name,
"records": len(records),
Expand Down
7 changes: 5 additions & 2 deletions s1a/tool/loop.py
Original file line number Diff line number Diff line change
Expand Up @@ -273,8 +273,11 @@ async def run_episode(
for event in state.rethinks:
print(f" rethink {event}", file=sys.stderr)
policy = model.name if isinstance(model, ToolDecisionModel) else "llm"
# Laya and Cua run in process: their tokens are free and unpriced
jev_input_tokens = sum(tick["input_tokens"] for tick in state.ticks if tick["source"] == "jev")
jev_input_tokens = (
sum(tick["input_tokens"] for tick in state.ticks if tick["source"] != "llm")
if isinstance(model, ToolDecisionModel) and model.bills_input_tokens
else 0
)
chat_input_tokens = sum(call["input_tokens"] for call in state.chat)
chat_output_tokens = sum(call["output_tokens"] for call in state.chat)
chat_cache_tokens = sum(call["cache_tokens"] for call in state.chat)
Expand Down
1 change: 1 addition & 0 deletions s1a/tool/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,6 +80,7 @@ def __init__(
self._fallback = fallback
self._act_name = ACT_TOOL
self.name = decision_model.name # lands in every tick's ``source`` and in ``Episode.policy``
self.bills_input_tokens = decision_model.bills_input_tokens

async def invoke(self, messages: Any, *, tools: Any = None, **kwargs: Any) -> AssistantMessage:
act_name = tool_name(tools, ACT_TOOL)
Expand Down
6 changes: 4 additions & 2 deletions tests/test_browser_policy.py
Original file line number Diff line number Diff line change
Expand Up @@ -433,8 +433,8 @@ async def test_the_tick_records_the_decisions_latency_tokens_and_the_answering_m
report = slot_model.report()
self.assertEqual((report["decisions"], report["median_decision_ms"], report["jev_input_tokens"]), (1, 7, 315))

async def test_jev_input_tokens_are_zero_for_a_non_jev_decision_model(self) -> None:
"""Laya and Cua run in process for free; only Jev-over-HTTP tokens are priced (s1a/tool/loop.py does the same)."""
async def test_report_prices_input_tokens_by_backend_flag(self) -> None:
"""The scripted backend's token usage is priced only when it opts into the Jev input rate."""
from s1a.decision_models import ScriptedModel

decision_model = ScriptedModel(latency_ms=3, usage=Usage(11, 0), model="laya-rl-agent")
Expand All @@ -444,6 +444,8 @@ async def test_jev_input_tokens_are_zero_for_a_non_jev_decision_model(self) -> N

self.assertEqual(slot_model.ticks[0]["input_tokens"], 11, "the tick itself still records what the model spent")
self.assertEqual(slot_model.report()["jev_input_tokens"], 0)
decision_model.bills_input_tokens = True
self.assertEqual(slot_model.report()["jev_input_tokens"], 11)

async def test_a_laya_shaped_model_fills_the_slot_the_same_way(self) -> None:
"""The policy asks any decision_model: a scripted one at the interface, no wire at all, decides a tick."""
Expand Down
27 changes: 26 additions & 1 deletion tests/test_decision_models_base.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,23 +3,48 @@

from __future__ import annotations

from unittest import IsolatedAsyncioTestCase
from unittest import IsolatedAsyncioTestCase, TestCase

from openjiuwen.core.common.exception.codes import StatusCode
from openjiuwen.core.common.exception.errors import BaseError, build_error

from decision_model_contract import CHECK, OBSERVATION, PICK, DecisionModelContract
from s1a.decision_models import (
CuaS1Model,
DecisionModel,
JevModel,
LayaModel,
Choice,
Image,
Json,
Noul,
Observation,
RandomModel,
RuleModel,
ScriptedModel,
)


class TestBillingContract(TestCase):
def test_every_backend_declares_whether_input_tokens_use_jev_pricing(self) -> None:
expected = {
JevModel: True,
LayaModel: False,
CuaS1Model: False,
RandomModel: False,
RuleModel: False,
ScriptedModel: False,
}
backends = set(DecisionModel.__subclasses__())
self.assertLessEqual(set(expected), backends)
for backend in backends:
with self.subTest(backend=backend.__name__):
declared = vars(backend).get("bills_input_tokens")
self.assertIsInstance(declared, bool)
if backend in expected:
self.assertIs(declared, expected[backend])


class TestScriptedContract(DecisionModelContract, IsolatedAsyncioTestCase):
def make(self) -> DecisionModel:
return ScriptedModel()
Expand Down
42 changes: 39 additions & 3 deletions tests/test_rails.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,7 +18,7 @@

from s1a import rails
from s1a.agents.injection_guard import SPEC as GUARD
from s1a.decision_models import JevModel, ScriptedModel, ScriptedTransport, Usage
from s1a.decision_models import JevModel, LayaModel, ScriptedModel, ScriptedTransport, Usage
from s1a.spec import RailSpec, Thresholds

INJECTED = (
Expand Down Expand Up @@ -204,7 +204,12 @@ async def test_precision_and_recall_of_the_act_band_against_the_labels(self) ->
labelled = Path(tmp) / "set.jsonl"
labelled.write_text("".join(json.dumps(r) + "\n" for r in records), encoding="utf-8")
summary = await rails.evaluate(
GUARD, labelled, decision_model=_noul([0.1, 0.9, 0.95, 0.2, 0.5]), results_dir=Path(tmp) / "results"
GUARD,
labelled,
decision_model=JevModel(
ScriptedTransport(noul=[0.1, 0.9, 0.95, 0.2, 0.5], usage={"input_tokens": 300}, latency_ms=7)
),
results_dir=Path(tmp) / "results",
)
job_dir = Path(summary["job_dir"])
verdicts = [json.loads(line) for line in (job_dir / "verdicts.jsonl").read_text().splitlines()]
Expand All @@ -217,7 +222,38 @@ async def test_precision_and_recall_of_the_act_band_against_the_labels(self) ->
self.assertEqual([v["band"] for v in verdicts], ["allow", "act", "act", "allow", "uncertain"])
self.assertEqual(written["rail"], "injection_guard")
self.assertEqual(job_dir.parent, Path(tmp) / "results" / "injection_guard")
self.assertTrue(job_dir.name.endswith("__scripted")) # the model's name, jev or laya on a real run
self.assertTrue(job_dir.name.endswith("__jev"))

async def test_laya_usage_is_not_charged_as_jev_in_the_returned_or_saved_summary(self) -> None:
agent = SimpleNamespace(
cfg={"max_len": 512},
system_one=lambda state, questions: {
"answers": {"check": {"noul": 0.9}},
"usage": {"input_tokens": 300},
},
)
decision_model = LayaModel(agent, model="laya-test")
verdict = await rails.ask(GUARD, {"text": INJECTED}, decision_model)
self.assertEqual(verdict.input_tokens, 300)
with tempfile.TemporaryDirectory() as tmp:
labelled = Path(tmp) / "set.jsonl"
labelled.write_text(json.dumps({"state": {"text": INJECTED}, "label": True}) + "\n", encoding="utf-8")
summary = await rails.evaluate(GUARD, labelled, decision_model=decision_model, results_dir=Path(tmp))
job_dir = Path(summary["job_dir"])
written = json.loads((job_dir / "summary.json").read_text(encoding="utf-8"))
for result in (summary, written):
self.assertEqual((result["jev_input_tokens"], result["cost_usd"]), (0, 0.0))
self.assertEqual((result["records"], result["accuracy"]), (1, 1.0))
self.assertTrue(job_dir.name.endswith("__laya"))

async def test_a_priced_backend_counts_tokens_without_relying_on_its_name(self) -> None:
decision_model = _noul([0.9])
decision_model.bills_input_tokens = True
with tempfile.TemporaryDirectory() as tmp:
labelled = Path(tmp) / "set.jsonl"
labelled.write_text(json.dumps({"state": {"text": INJECTED}, "label": True}) + "\n", encoding="utf-8")
summary = await rails.evaluate(GUARD, labelled, decision_model=decision_model, results_dir=Path(tmp))
self.assertEqual((summary["jev_input_tokens"], summary["cost_usd"]), (300, 0.000013))

def test_a_record_without_a_boolean_label_is_rejected(self) -> None:
with tempfile.TemporaryDirectory() as tmp:
Expand Down
11 changes: 11 additions & 0 deletions tests/test_tool_loop.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,9 @@
JevModel,
RandomModel,
RuleModel,
ScriptedModel,
ScriptedTransport,
Usage,
)
from s1a.spec import Budget, ToolAgentSpec
from s1a.tool import loop as agent
Expand Down Expand Up @@ -254,6 +256,15 @@ def _refusing() -> JevModel:
class TestEpisodeThroughTheAgent(IsolatedAsyncioTestCase):
"""Episodes through ``create_deep_agent`` and the Runner, offline: a rule in the slot, no chat model."""

async def test_a_priced_backend_counts_its_tokens_without_relying_on_its_name(self) -> None:
decision_model = ScriptedModel(choose="inc", usage=Usage(input_tokens=300))
decision_model.bills_input_tokens = True
episode = await _play(
CountingEnv(), max_acts=1, timeout_s=60.0, model_name="random", decision_model=decision_model
)
self.assertEqual((episode.decisions[0]["source"], episode.decisions[0]["input_tokens"]), ("scripted", 300))
self.assertEqual((episode.jev_input_tokens, episode.cost_usd), (300, 0.000013))

async def test_a_refused_decision_is_the_episodes_error_with_no_decisions(self) -> None:
episode = await _play(CountingEnv(), max_acts=10, timeout_s=60.0, model_name="jev", decision_model=_refusing())
self.assertTrue(episode.error.startswith("decision failed: "), episode.error)
Expand Down
Loading