From 91185462d9d1f09fe97595eb1f4b6ffe6512f88d Mon Sep 17 00:00:00 2001 From: onion-111 Date: Sat, 26 Sep 2026 23:41:27 +0800 Subject: [PATCH 1/2] [Fix] Avoid charging Jev API prices for local rail models --- CHANGELOG.md | 2 ++ s1a/rails.py | 3 ++- tests/test_rails.py | 33 ++++++++++++++++++++++++++++++--- 3 files changed, 34 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7b1eca6..7e6baab 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,8 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver - Windows development checks: the smoke script accepts CRLF output, shell scripts and Git hooks retain LF line endings, and tests check socket closure and invalid output directories without Unix-specific behavior. The core CI matrix now covers Windows with Python 3.11. +- Rail evaluations charge Jev input tokens only for the Jev backend. Local Laya evaluations keep + `jev_input_tokens` and `cost_usd` at zero, matching the tool and browser fronts. ### Added diff --git a/s1a/rails.py b/s1a/rails.py index b8941da..2935f18 100644 --- a/s1a/rails.py +++ b/s1a/rails.py @@ -120,7 +120,8 @@ async def evaluate( tp = sum(a and label for a, label in zip(acted, labels)) fp = sum(a and not label for a, label in zip(acted, labels)) fn = sum(label and not a for a, label in zip(acted, labels)) - jev_input_tokens = sum(verdict.input_tokens for verdict in verdicts) + # Local models have no Jev API charges, as on the tool and browser fronts. + jev_input_tokens = sum(verdict.input_tokens for verdict in verdicts) if decision_model.name == "jev" else 0 summary = { "rail": spec.name, "records": len(records), diff --git a/tests/test_rails.py b/tests/test_rails.py index 2fa1fff..a5f10dc 100644 --- a/tests/test_rails.py +++ b/tests/test_rails.py @@ -18,7 +18,7 @@ from s1a import rails from s1a.agents.injection_guard import SPEC as GUARD -from s1a.decision_models import JevModel, ScriptedModel, ScriptedTransport, Usage +from s1a.decision_models import JevModel, LayaModel, ScriptedModel, ScriptedTransport, Usage from s1a.spec import RailSpec, Thresholds INJECTED = ( @@ -204,7 +204,12 @@ async def test_precision_and_recall_of_the_act_band_against_the_labels(self) -> labelled = Path(tmp) / "set.jsonl" labelled.write_text("".join(json.dumps(r) + "\n" for r in records), encoding="utf-8") summary = await rails.evaluate( - GUARD, labelled, decision_model=_noul([0.1, 0.9, 0.95, 0.2, 0.5]), results_dir=Path(tmp) / "results" + GUARD, + labelled, + decision_model=JevModel( + ScriptedTransport(noul=[0.1, 0.9, 0.95, 0.2, 0.5], usage={"input_tokens": 300}, latency_ms=7) + ), + results_dir=Path(tmp) / "results", ) job_dir = Path(summary["job_dir"]) verdicts = [json.loads(line) for line in (job_dir / "verdicts.jsonl").read_text().splitlines()] @@ -217,7 +222,29 @@ async def test_precision_and_recall_of_the_act_band_against_the_labels(self) -> self.assertEqual([v["band"] for v in verdicts], ["allow", "act", "act", "allow", "uncertain"]) self.assertEqual(written["rail"], "injection_guard") self.assertEqual(job_dir.parent, Path(tmp) / "results" / "injection_guard") - self.assertTrue(job_dir.name.endswith("__scripted")) # the model's name, jev or laya on a real run + self.assertTrue(job_dir.name.endswith("__jev")) + + async def test_laya_usage_is_not_charged_as_jev_in_the_returned_or_saved_summary(self) -> None: + agent = SimpleNamespace( + cfg={"max_len": 512}, + system_one=lambda state, questions: { + "answers": {"check": {"noul": 0.9}}, + "usage": {"input_tokens": 300}, + }, + ) + decision_model = LayaModel(agent, model="laya-test") + verdict = await rails.ask(GUARD, {"text": INJECTED}, decision_model) + self.assertEqual(verdict.input_tokens, 300) + with tempfile.TemporaryDirectory() as tmp: + labelled = Path(tmp) / "set.jsonl" + labelled.write_text(json.dumps({"state": {"text": INJECTED}, "label": True}) + "\n", encoding="utf-8") + summary = await rails.evaluate(GUARD, labelled, decision_model=decision_model, results_dir=Path(tmp)) + job_dir = Path(summary["job_dir"]) + written = json.loads((job_dir / "summary.json").read_text(encoding="utf-8")) + for result in (summary, written): + self.assertEqual((result["jev_input_tokens"], result["cost_usd"]), (0, 0.0)) + self.assertEqual((result["records"], result["accuracy"]), (1, 1.0)) + self.assertTrue(job_dir.name.endswith("__laya")) def test_a_record_without_a_boolean_label_is_rejected(self) -> None: with tempfile.TemporaryDirectory() as tmp: From 68f29f4205493f7823c6c6b19c1e4decd368c1b7 Mon Sep 17 00:00:00 2001 From: onion-111 Date: Mon, 28 Sep 2026 18:27:57 +0800 Subject: [PATCH 2/2] [Fix] Let decision backends declare billable input tokens --- CHANGELOG.md | 4 ++-- s1a/browser/decision_model.py | 5 +++-- s1a/decision_models/base.py | 1 + s1a/decision_models/baselines.py | 2 ++ s1a/decision_models/cua.py | 1 + s1a/decision_models/fakes.py | 1 + s1a/decision_models/jev.py | 1 + s1a/decision_models/laya.py | 1 + s1a/rails.py | 3 +-- s1a/tool/loop.py | 7 +++++-- s1a/tool/models.py | 1 + tests/test_browser_policy.py | 6 ++++-- tests/test_decision_models_base.py | 27 ++++++++++++++++++++++++++- tests/test_rails.py | 9 +++++++++ tests/test_tool_loop.py | 11 +++++++++++ 15 files changed, 69 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7e6baab..1a217f2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,8 +9,8 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver - Windows development checks: the smoke script accepts CRLF output, shell scripts and Git hooks retain LF line endings, and tests check socket closure and invalid output directories without Unix-specific behavior. The core CI matrix now covers Windows with Python 3.11. -- Rail evaluations charge Jev input tokens only for the Jev backend. Local Laya evaluations keep - `jev_input_tokens` and `cost_usd` at zero, matching the tool and browser fronts. +- Rail, tool, and browser evaluations charge Jev-rate input tokens only when the decision backend declares + them billable. Local model token usage remains recorded without Jev API charges. ### Added diff --git a/s1a/browser/decision_model.py b/s1a/browser/decision_model.py index 001478b..cf94675 100644 --- a/s1a/browser/decision_model.py +++ b/s1a/browser/decision_model.py @@ -773,9 +773,10 @@ def report(self) -> dict[str, Any]: "interactions": len([h for h in run.history if h["kind"] != "wait"]), "waits": len([h for h in run.history if h["kind"] == "wait"]), "median_decision_ms": int(statistics.median(jev)) if jev else 0, - # Laya and Cua run in process: their tokens are free and unpriced (s1a/tool/loop.py does the same). "jev_input_tokens": ( - sum(int(t.get("input_tokens") or 0) for t in run.ticks) if self._decision_model.name == "jev" else 0 + sum(int(t.get("input_tokens") or 0) for t in run.ticks) + if self._decision_model.bills_input_tokens + else 0 ), "median_probe_ms": int(statistics.median(t["probe_ms"] for t in run.ticks)) if run.ticks else 0, "settle_probes": sum(t.get("settle_probes", 0) for t in run.ticks), diff --git a/s1a/decision_models/base.py b/s1a/decision_models/base.py index b120ae0..2cc7ae8 100644 --- a/s1a/decision_models/base.py +++ b/s1a/decision_models/base.py @@ -41,6 +41,7 @@ class DecisionModel(ABC): """A model that reads an observation and a discrete action space and returns a distribution over it.""" name: str = "decision_model" # the ``--model`` value; lands in every tick's ``source`` + bills_input_tokens: bool # True when input tokens are priced at JEV_USD_PER_INPUT_TOKEN supports_images: bool = False question_types: frozenset[str] = frozenset({"choice", "noul"}) deterministic: bool = False # the same request always gets the same answer, so a re-ask is a wasted call diff --git a/s1a/decision_models/baselines.py b/s1a/decision_models/baselines.py index 741ef31..b823179 100644 --- a/s1a/decision_models/baselines.py +++ b/s1a/decision_models/baselines.py @@ -16,6 +16,7 @@ class RandomModel(DecisionModel): """Uniform over the offered keys: the loop-overhead arm. Answers choice questions only.""" name = "random" + bills_input_tokens = False question_types = frozenset({"choice"}) def __init__(self, seed: int) -> None: @@ -43,6 +44,7 @@ class RuleModel(DecisionModel): decision failure. """ + bills_input_tokens = False question_types = frozenset({"choice"}) deterministic = True diff --git a/s1a/decision_models/cua.py b/s1a/decision_models/cua.py index cc7d2e7..194ffbb 100644 --- a/s1a/decision_models/cua.py +++ b/s1a/decision_models/cua.py @@ -70,6 +70,7 @@ class CuaS1Model(DecisionModel): """Cua-S1 Nano's ``NanoScorer`` behind the interface: choice questions only, text only, deterministic.""" name = "cua" + bills_input_tokens = False question_types = frozenset({"choice"}) deterministic = True diff --git a/s1a/decision_models/fakes.py b/s1a/decision_models/fakes.py index 1f5126b..e8775ee 100644 --- a/s1a/decision_models/fakes.py +++ b/s1a/decision_models/fakes.py @@ -64,6 +64,7 @@ class ScriptedModel(DecisionModel): images, so a pass-through can be tested.""" name = "scripted" + bills_input_tokens = False supports_images = True def __init__( diff --git a/s1a/decision_models/jev.py b/s1a/decision_models/jev.py index fce6e39..84a90ec 100644 --- a/s1a/decision_models/jev.py +++ b/s1a/decision_models/jev.py @@ -34,6 +34,7 @@ class JevModel(DecisionModel): """TypeSafe Jev over HTTP; the transport owns the connection, its retries and the round-trip clock.""" name = "jev" + bills_input_tokens = True def __init__(self, transport: JevTransport) -> None: self._transport = transport diff --git a/s1a/decision_models/laya.py b/s1a/decision_models/laya.py index b4178a2..6f9c2df 100644 --- a/s1a/decision_models/laya.py +++ b/s1a/decision_models/laya.py @@ -51,6 +51,7 @@ class LayaModel(DecisionModel): """Laya's ``Agent`` (or anything with ``system_one(state, questions)`` and a ``cfg``) behind the interface.""" name = "laya" + bills_input_tokens = False deterministic = True def __init__(self, agent: Any, *, model: str) -> None: diff --git a/s1a/rails.py b/s1a/rails.py index 2935f18..d5369c6 100644 --- a/s1a/rails.py +++ b/s1a/rails.py @@ -120,8 +120,7 @@ async def evaluate( tp = sum(a and label for a, label in zip(acted, labels)) fp = sum(a and not label for a, label in zip(acted, labels)) fn = sum(label and not a for a, label in zip(acted, labels)) - # Local models have no Jev API charges, as on the tool and browser fronts. - jev_input_tokens = sum(verdict.input_tokens for verdict in verdicts) if decision_model.name == "jev" else 0 + jev_input_tokens = sum(verdict.input_tokens for verdict in verdicts) if decision_model.bills_input_tokens else 0 summary = { "rail": spec.name, "records": len(records), diff --git a/s1a/tool/loop.py b/s1a/tool/loop.py index 2d71ab1..8186402 100644 --- a/s1a/tool/loop.py +++ b/s1a/tool/loop.py @@ -273,8 +273,11 @@ async def run_episode( for event in state.rethinks: print(f" rethink {event}", file=sys.stderr) policy = model.name if isinstance(model, ToolDecisionModel) else "llm" - # Laya and Cua run in process: their tokens are free and unpriced - jev_input_tokens = sum(tick["input_tokens"] for tick in state.ticks if tick["source"] == "jev") + jev_input_tokens = ( + sum(tick["input_tokens"] for tick in state.ticks if tick["source"] != "llm") + if isinstance(model, ToolDecisionModel) and model.bills_input_tokens + else 0 + ) chat_input_tokens = sum(call["input_tokens"] for call in state.chat) chat_output_tokens = sum(call["output_tokens"] for call in state.chat) chat_cache_tokens = sum(call["cache_tokens"] for call in state.chat) diff --git a/s1a/tool/models.py b/s1a/tool/models.py index 1e76e33..937c029 100644 --- a/s1a/tool/models.py +++ b/s1a/tool/models.py @@ -80,6 +80,7 @@ def __init__( self._fallback = fallback self._act_name = ACT_TOOL self.name = decision_model.name # lands in every tick's ``source`` and in ``Episode.policy`` + self.bills_input_tokens = decision_model.bills_input_tokens async def invoke(self, messages: Any, *, tools: Any = None, **kwargs: Any) -> AssistantMessage: act_name = tool_name(tools, ACT_TOOL) diff --git a/tests/test_browser_policy.py b/tests/test_browser_policy.py index 6217256..288fb7f 100644 --- a/tests/test_browser_policy.py +++ b/tests/test_browser_policy.py @@ -433,8 +433,8 @@ async def test_the_tick_records_the_decisions_latency_tokens_and_the_answering_m report = slot_model.report() self.assertEqual((report["decisions"], report["median_decision_ms"], report["jev_input_tokens"]), (1, 7, 315)) - async def test_jev_input_tokens_are_zero_for_a_non_jev_decision_model(self) -> None: - """Laya and Cua run in process for free; only Jev-over-HTTP tokens are priced (s1a/tool/loop.py does the same).""" + async def test_report_prices_input_tokens_by_backend_flag(self) -> None: + """The scripted backend's token usage is priced only when it opts into the Jev input rate.""" from s1a.decision_models import ScriptedModel decision_model = ScriptedModel(latency_ms=3, usage=Usage(11, 0), model="laya-rl-agent") @@ -444,6 +444,8 @@ async def test_jev_input_tokens_are_zero_for_a_non_jev_decision_model(self) -> N self.assertEqual(slot_model.ticks[0]["input_tokens"], 11, "the tick itself still records what the model spent") self.assertEqual(slot_model.report()["jev_input_tokens"], 0) + decision_model.bills_input_tokens = True + self.assertEqual(slot_model.report()["jev_input_tokens"], 11) async def test_a_laya_shaped_model_fills_the_slot_the_same_way(self) -> None: """The policy asks any decision_model: a scripted one at the interface, no wire at all, decides a tick.""" diff --git a/tests/test_decision_models_base.py b/tests/test_decision_models_base.py index e627a58..e0dab1c 100644 --- a/tests/test_decision_models_base.py +++ b/tests/test_decision_models_base.py @@ -3,23 +3,48 @@ from __future__ import annotations -from unittest import IsolatedAsyncioTestCase +from unittest import IsolatedAsyncioTestCase, TestCase from openjiuwen.core.common.exception.codes import StatusCode from openjiuwen.core.common.exception.errors import BaseError, build_error from decision_model_contract import CHECK, OBSERVATION, PICK, DecisionModelContract from s1a.decision_models import ( + CuaS1Model, DecisionModel, + JevModel, + LayaModel, Choice, Image, Json, Noul, Observation, + RandomModel, + RuleModel, ScriptedModel, ) +class TestBillingContract(TestCase): + def test_every_backend_declares_whether_input_tokens_use_jev_pricing(self) -> None: + expected = { + JevModel: True, + LayaModel: False, + CuaS1Model: False, + RandomModel: False, + RuleModel: False, + ScriptedModel: False, + } + backends = set(DecisionModel.__subclasses__()) + self.assertLessEqual(set(expected), backends) + for backend in backends: + with self.subTest(backend=backend.__name__): + declared = vars(backend).get("bills_input_tokens") + self.assertIsInstance(declared, bool) + if backend in expected: + self.assertIs(declared, expected[backend]) + + class TestScriptedContract(DecisionModelContract, IsolatedAsyncioTestCase): def make(self) -> DecisionModel: return ScriptedModel() diff --git a/tests/test_rails.py b/tests/test_rails.py index a5f10dc..10363c2 100644 --- a/tests/test_rails.py +++ b/tests/test_rails.py @@ -246,6 +246,15 @@ async def test_laya_usage_is_not_charged_as_jev_in_the_returned_or_saved_summary self.assertEqual((result["records"], result["accuracy"]), (1, 1.0)) self.assertTrue(job_dir.name.endswith("__laya")) + async def test_a_priced_backend_counts_tokens_without_relying_on_its_name(self) -> None: + decision_model = _noul([0.9]) + decision_model.bills_input_tokens = True + with tempfile.TemporaryDirectory() as tmp: + labelled = Path(tmp) / "set.jsonl" + labelled.write_text(json.dumps({"state": {"text": INJECTED}, "label": True}) + "\n", encoding="utf-8") + summary = await rails.evaluate(GUARD, labelled, decision_model=decision_model, results_dir=Path(tmp)) + self.assertEqual((summary["jev_input_tokens"], summary["cost_usd"]), (300, 0.000013)) + def test_a_record_without_a_boolean_label_is_rejected(self) -> None: with tempfile.TemporaryDirectory() as tmp: labelled = Path(tmp) / "bad.jsonl" diff --git a/tests/test_tool_loop.py b/tests/test_tool_loop.py index ab63868..3623537 100644 --- a/tests/test_tool_loop.py +++ b/tests/test_tool_loop.py @@ -22,7 +22,9 @@ JevModel, RandomModel, RuleModel, + ScriptedModel, ScriptedTransport, + Usage, ) from s1a.spec import Budget, ToolAgentSpec from s1a.tool import loop as agent @@ -254,6 +256,15 @@ def _refusing() -> JevModel: class TestEpisodeThroughTheAgent(IsolatedAsyncioTestCase): """Episodes through ``create_deep_agent`` and the Runner, offline: a rule in the slot, no chat model.""" + async def test_a_priced_backend_counts_its_tokens_without_relying_on_its_name(self) -> None: + decision_model = ScriptedModel(choose="inc", usage=Usage(input_tokens=300)) + decision_model.bills_input_tokens = True + episode = await _play( + CountingEnv(), max_acts=1, timeout_s=60.0, model_name="random", decision_model=decision_model + ) + self.assertEqual((episode.decisions[0]["source"], episode.decisions[0]["input_tokens"]), ("scripted", 300)) + self.assertEqual((episode.jev_input_tokens, episode.cost_usd), (300, 0.000013)) + async def test_a_refused_decision_is_the_episodes_error_with_no_decisions(self) -> None: episode = await _play(CountingEnv(), max_acts=10, timeout_s=60.0, model_name="jev", decision_model=_refusing()) self.assertTrue(episode.error.startswith("decision failed: "), episode.error)