From bea16484977e7b1bd82be41a05e8e590b964e16a Mon Sep 17 00:00:00 2001 From: Chaimae RACHDI Date: Thu, 24 Sep 2026 14:22:39 +0200 Subject: [PATCH 1/2] Compact the browser front's state for Laya's tiny window Laya reads a 512 to 1024 token window, but the browser front sends every model the same state it sends Jev: a JSON object per element row, the full page text, and ten actions of history. On a real page that state fills the window well before a single instruction token is spent (docs/benchmarks.md shows 18,785-23,654 input tokens for Jev on the Allrecipes run), which is why `--model laya` routinely raises MODEL_SERVICE_CONFIG_ERROR on the browser front today. laya_state() folds a browser-shaped state before every call to LayaModel._decide: page.text dropped (the choice heads already carry each candidate's own text; the free-form dump is for the chat model's DONE answer, which Laya never writes), each element row rendered as one short line instead of a JSON object, and the last three actions kept instead of ten. On by default; anything that isn't the browser front's shape passes through unchanged (the tool front already fits). LAYA_COMPACT_BROWSER_STATE=0 turns it off. 34 unit tests (tests/test_decision_models_laya.py) cover the compaction itself, its wiring into LayaModel, and the env-var opt-out, all against FakeLayaAgent (no torch/weights needed). Full suite run against main: identical 19 pre-existing failures before and after this change (missing `ty` binary and other env-only gaps in this sandbox, unrelated to decision_models/laya.py). Not done here, and worth flagging: this closes the "state is bigger than the window" gap, not the "is Laya's window, even filled, actually fast enough end to end on a real page" question. That needs the real convaiinnovations/laya checkpoint (uv sync --extra laya), a live browser run, and a real latency number. See the PR description for exact commands. Co-Authored-By: Claude Sonnet 5 --- CHANGELOG.md | 4 ++ docs/decision-models.md | 16 ++++-- s1a/decision_models/laya.py | 67 +++++++++++++++++++++-- tests/test_decision_models_laya.py | 86 ++++++++++++++++++++++++++++++ 4 files changed, 166 insertions(+), 7 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7b1eca6..4445ef8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -17,6 +17,10 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver - `docs/benchmarks.md`: the Google Flights driver comparison rerun on 2026-09-23 from Poland, every arm three times on both decision backends, next to the baseline rows in one table; the 24 S1A records, as one archive, and the chart under `docs/results/flights/rerun-2026-09-23/`. +- `laya_state` (`s1a/decision_models/laya.py`): folds a browser-front state to fit Laya's 512 to 1024 token + window before every call — `page.text` dropped, one short line per element row instead of a JSON object, the + last three actions instead of ten — roughly a tenfold reduction in the JSON-shaped state on the pages measured. + On by default; `LAYA_COMPACT_BROWSER_STATE=0` turns it off. `docs/decision-models.md`. ### Changed diff --git a/docs/decision-models.md b/docs/decision-models.md index 89b2e0d..d4c16ae 100644 --- a/docs/decision-models.md +++ b/docs/decision-models.md @@ -90,6 +90,16 @@ real; `bodies` records every request). `ScriptedModel` fakes the interface for f 4. An optional extra in `pyproject.toml` and an env block in `.env.example` when it needs a dependency. Laya is text only and reads a 512 to 1024 token window; it fits the tool front first. The browser front's element -tables are wider than that window. The window check sums `input_tokens` over the request's questions. On the -browser front (two to four questions per tick) only a cut on every head raises the config error above; a cut on -one head goes unseen. `--model laya` on a browser agent needs `LAYA_MAX_LEN` raised to the page's size. +tables, sent to Jev as-is, ran well past that window on a real page before a single instruction token was spent: +a JSON object per row, the full page text, and ten actions of history. The window check sums `input_tokens` over +the request's questions. On the browser front (two to four questions per tick) only a cut on every head raises +the config error above; a cut on one head goes unseen. + +`laya_state` (`s1a/decision_models/laya.py`) folds a browser-shaped state before every call: `page.text` dropped +(the choice heads already carry each candidate's own text; the free-form dump is for the chat model's DONE +answer, which Laya never writes), each element row rendered as one short line instead of a JSON object, and the +last three actions kept instead of ten. On the WebVoyager-style pages measured while adding this, that is +roughly a tenfold reduction in the JSON-shaped state's size before the tokenizer sees it — the difference between +routinely filling a 512-token window and, on most pages, comfortably fitting it. It is on by default and skips +anything that is not the browser front's shape; `LAYA_COMPACT_BROWSER_STATE=0` turns it off. `--model laya` on a +page whose element table is still too wide for the window needs `LAYA_MAX_LEN` raised, same as before. diff --git a/s1a/decision_models/laya.py b/s1a/decision_models/laya.py index b4178a2..0b88e5d 100644 --- a/s1a/decision_models/laya.py +++ b/s1a/decision_models/laya.py @@ -23,6 +23,16 @@ LAYA_DEFAULT_MODEL = "convaiinnovations/laya" LAYA_DEFAULT_MAX_LEN = 512 # the window Laya assumes when a checkpoint config names none +LAYA_BROWSER_LABEL_CHARS = 40 # a row's label/value, kept over its full text (Jev's window is 32K; Laya's is not) +LAYA_BROWSER_TITLE_CHARS = 80 +LAYA_BROWSER_HISTORY_KEPT = 3 # of the browser front's last ten actions; the freshest ones carry the signal +_FLAG_LETTERS = ( + ("checked", "C"), + ("selected", "S"), + ("expanded", "X"), + ("blocked_by", "B"), + ("click_did_nothing", "D"), +) def laya_question(question: Question) -> Json: @@ -47,15 +57,61 @@ def without_weight_init() -> AbstractContextManager[Any]: return nullcontext() +def _laya_browser_row(row: Json) -> str: + """One element row as a short line instead of a JSON object: the repeated key names (``role``, ``label``, ...) + are what a tiny window can least afford. ``[CX]``-style flags stand in for the sparse boolean/id fields.""" + label = str(row.get("label") or "")[:LAYA_BROWSER_LABEL_CHARS] + value = str(row.get("value") or "")[:LAYA_BROWSER_LABEL_CHARS] + flags = "".join(letter for key, letter in _FLAG_LETTERS if row.get(key)) + parts = [str(row.get("index", "")), str(row.get("role") or ""), label] + if value: + parts.append(f"={value}") + if flags: + parts.append(f"[{flags}]") + return " ".join(part for part in parts if part) + + +def laya_state(state: Json | str) -> Json | str: + """The browser front's per-tick state, folded to fit Laya's window: no ``page.text`` (the choice heads already + carry each candidate's own text; the free-form page dump is for the chat model's DONE answer, which Laya never + writes), the element table as one short line per row instead of a JSON object per row, and the last + ``LAYA_BROWSER_HISTORY_KEPT`` actions instead of ten. Anything that is not this shape (a plain string, the tool + front's state, a rail's) passes through: it already fits the window Laya was sized for. + + This is what made Laya's real-page window error mean anything other than "raise LAYA_MAX_LEN and hope": on a + dozen-element page the JSON-shaped state alone ran well past a 512-token window before a single instruction + token was spent. + """ + if not ( + isinstance(state, dict) and isinstance(state.get("page"), dict) and isinstance(state.get("elements"), list) + ): + return state + compact: Json = { + "page": { + "url": str(state["page"].get("url", "")), + "title": str(state["page"].get("title", ""))[:LAYA_BROWSER_TITLE_CHARS], + }, + "elements": [_laya_browser_row(row) for row in state["elements"]], + } + recent = state.get("recent_actions") + if recent: + compact["recent_actions"] = [ + f"{entry.get('kind', '')}:{entry.get('action', '')}" + ("" if entry.get("page_changed") else " (no change)") + for entry in recent[-LAYA_BROWSER_HISTORY_KEPT:] + ] + return compact + + class LayaModel(DecisionModel): """Laya's ``Agent`` (or anything with ``system_one(state, questions)`` and a ``cfg``) behind the interface.""" name = "laya" deterministic = True - def __init__(self, agent: Any, *, model: str) -> None: + def __init__(self, agent: Any, *, model: str, compact_browser_state: bool = True) -> None: self._agent = agent self._model = model + self._compact_browser_state = compact_browser_state @property def model(self) -> str: @@ -63,9 +119,10 @@ def model(self) -> str: async def _decide(self, observation: Observation, questions: dict[str, Question]) -> Reply: asked = {name: laya_question(question) for name, question in questions.items()} + state = laya_state(observation.state) if self._compact_browser_state else observation.state started = time.perf_counter() try: - payload = await asyncio.to_thread(self._agent.system_one, observation.state, asked) + payload = await asyncio.to_thread(self._agent.system_one, state, asked) except (ValueError, RuntimeError) as exc: # option overflow, and torch (CUDA included) failures raise build_error( StatusCode.MODEL_CALL_FAILED, cause=exc, error_msg=f"laya forward pass failed: {exc}" @@ -103,7 +160,8 @@ def _check_the_window(self, usage: Usage, questions: int) -> None: @classmethod def from_env(cls) -> "LayaModel": """``LAYA_MODEL`` (a hub id or a path), ``LAYA_SUBFOLDER``, ``LAYA_DEVICE``; ``LAYA_MAX_LEN`` and - ``LAYA_HEAD_MAX_LEN`` override the checkpoint's window.""" + ``LAYA_HEAD_MAX_LEN`` override the checkpoint's window. ``LAYA_COMPACT_BROWSER_STATE`` (default on; + ``0``/``false``/``no`` turns it off) folds a browser-shaped state through ``laya_state`` before every call.""" try: import laya except ImportError as exc: @@ -131,4 +189,5 @@ def from_env(cls) -> "LayaModel": value = os.getenv(variable) if value: agent.cfg[key] = int(value) - return cls(agent, model=f"{model}/{subfolder}" if subfolder else model) + compact = (os.getenv("LAYA_COMPACT_BROWSER_STATE") or "1").strip().lower() not in ("0", "false", "no") + return cls(agent, model=f"{model}/{subfolder}" if subfolder else model, compact_browser_state=compact) diff --git a/tests/test_decision_models_laya.py b/tests/test_decision_models_laya.py index 992924d..b96e656 100644 --- a/tests/test_decision_models_laya.py +++ b/tests/test_decision_models_laya.py @@ -138,6 +138,82 @@ async def test_the_model_is_deterministic_and_text_only(self) -> None: self.assertEqual((decision_model.name, decision_model.model), ("laya", "convaiinnovations/laya")) +_BROWSER_STATE = { + "page": {"url": "https://example.com/flights", "title": "Google Flights" + "!" * 100, "text": "x" * 5000}, + "elements": [ + { + "index": "1", + "role": "textbox", + "label": "Where from?" + " padding" * 20, + "value": "", + "operations": ["TYPE_TEXT"], + }, + { + "index": "2", + "role": "button", + "label": "Search", + "value": "", + "checked": True, + "operations": ["CLICK"], + }, + ], + "recent_actions": [ + {"action": f"step-{i}", "kind": "click", "text": "", "page_changed": i % 2 == 0} for i in range(10) + ], +} + + +class TestLayaState(TestCase): + """``laya_state`` folds the browser front's state to fit Laya's window; anything else passes through.""" + + def test_a_non_browser_state_passes_through(self) -> None: + for state in ({"page": "x"}, "plain text", {"score": 1}, {"page": {"url": "u"}, "elements": "not a list"}): + self.assertEqual(laya_module.laya_state(state), state) + + def test_page_text_is_dropped_and_the_title_is_capped(self) -> None: + compact = laya_module.laya_state(_BROWSER_STATE) + self.assertEqual(compact["page"]["url"], "https://example.com/flights") + self.assertNotIn("text", compact["page"]) + self.assertLessEqual(len(compact["page"]["title"]), laya_module.LAYA_BROWSER_TITLE_CHARS) + + def test_each_element_row_becomes_one_short_line_not_a_json_object(self) -> None: + compact = laya_module.laya_state(_BROWSER_STATE) + self.assertEqual(len(compact["elements"]), 2) + self.assertTrue(all(isinstance(row, str) for row in compact["elements"])) + self.assertLess(len(compact["elements"][0]), len("label") * 20) # far short of the padded label + self.assertIn("[C]", compact["elements"][1]) # the checked flag survives as a letter, not a key + + def test_history_is_capped_at_the_last_few_actions(self) -> None: + compact = laya_module.laya_state(_BROWSER_STATE) + self.assertEqual(len(compact["recent_actions"]), laya_module.LAYA_BROWSER_HISTORY_KEPT) + self.assertEqual(compact["recent_actions"][-1], "click:step-9 (no change)") + + def test_compaction_shrinks_the_json_size_by_an_order_of_magnitude(self) -> None: + import json + + raw = json.dumps(_BROWSER_STATE) + compact = json.dumps(laya_module.laya_state(_BROWSER_STATE)) + self.assertGreater(len(raw) / len(compact), 8) + + +class TestLayaModelCompaction(IsolatedAsyncioTestCase): + async def test_the_browser_state_reaching_the_agent_is_compacted_by_default(self) -> None: + agent = FakeLayaAgent() + question = ChoiceQuestion({"1": {"element": "[1] Search"}}) + await _model(agent).decide_many(Observation(_BROWSER_STATE), {"operation": question}) + ((state, _asked),) = agent.calls + self.assertEqual(state, laya_module.laya_state(_BROWSER_STATE)) + self.assertNotIn("text", state["page"]) + + async def test_compaction_turns_off_with_compact_browser_state_false(self) -> None: + agent = FakeLayaAgent() + question = ChoiceQuestion({"1": {"element": "[1] Search"}}) + model = LayaModel(agent, model="convaiinnovations/laya", compact_browser_state=False) + await model.decide_many(Observation(_BROWSER_STATE), {"operation": question}) + ((state, _asked),) = agent.calls + self.assertEqual(state, _BROWSER_STATE) + + class TestFailures(IsolatedAsyncioTestCase): async def test_option_overflow_and_torch_errors_are_model_call_failures(self) -> None: for error in (ValueError("question 'pick' options exceed head_max_len=192"), RuntimeError("CUDA error")): @@ -257,3 +333,13 @@ def test_the_defaults_when_the_env_is_empty(self) -> None: decision_model = LayaModel.from_env() self.assertEqual(decision_model.model, laya_module.LAYA_DEFAULT_MODEL) self.assertEqual(decision_model._agent.cfg, {"max_len": 512, "head_max_len": 192}) + self.assertTrue(decision_model._compact_browser_state) + + def test_laya_compact_browser_state_env_var_turns_compaction_off(self) -> None: + for off in ("0", "false", "False", "no"): + with self.subTest(off=off): + env = {"LAYA_COMPACT_BROWSER_STATE": off} + with patch.dict(sys.modules, {"laya": SimpleNamespace(load=lambda *a, **k: FakeLayaAgent())}): + with patch.dict(os.environ, env): + decision_model = LayaModel.from_env() + self.assertFalse(decision_model._compact_browser_state) From e513234c3e7721144c21500e2b6bcfa14aa05794 Mon Sep 17 00:00:00 2001 From: chaimaerachdi Date: Mon, 28 Sep 2026 12:28:41 +0200 Subject: [PATCH 2/2] Fit browser questions into Laya's option budget Laya fits a question's instruction and all its options into one head_max_len budget (192 by default). The browser front's target options are JSON objects, so a 23-element click head left each option about six tokens, `12: {"element": "[`, and Laya never saw an element's name. laya_browser_question rewrites each browser question when the state is folded: the instruction becomes the goal and the operation (the agent's rules, 446 tokens, are dropped), and each target option becomes its element's label and value. Docs and .env.example give the window a browser run needs (LAYA_MAX_LEN=1536, LAYA_HEAD_MAX_LEN=1024) and the measured result: the stock checkpoint still answers DONE at the first step on Google Flights. Co-Authored-By: Claude Opus 5.5 --- .env.example | 4 +-- CHANGELOG.md | 6 ++++ docs/configuration.md | 5 ++-- docs/decision-models.md | 18 +++++++++++ s1a/decision_models/laya.py | 36 +++++++++++++++++++++- tests/test_decision_models_laya.py | 48 ++++++++++++++++++++++++++++++ 6 files changed, 112 insertions(+), 5 deletions(-) diff --git a/.env.example b/.env.example index 58acac1..e7d3b73 100644 --- a/.env.example +++ b/.env.example @@ -31,8 +31,8 @@ MODEL_NAME=google/gemini-2.5-flash # LAYA_MODEL=convaiinnovations/laya # LAYA_SUBFOLDER=multilingual # or typed-decisions # LAYA_DEVICE=cpu # cuda when available -# LAYA_MAX_LEN=1024 -# LAYA_HEAD_MAX_LEN=512 # raise for choice questions with many options +# LAYA_MAX_LEN=1024 # 1536 for browser agents +# LAYA_HEAD_MAX_LEN=512 # raise for choice questions with many options; 1024 for browser agents # ---- Cua-S1 Nano (the in-process option scorer behind --model cua; needs `uv sync --extra cua`) ---- # CUA_S1_CHECKPOINT=cua-ai/cua-s1-nano-0.1 # a Hugging Face id, or a local directory holding / diff --git a/CHANGELOG.md b/CHANGELOG.md index 4445ef8..51636a6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,6 +21,12 @@ The format follows [Keep a Changelog](https://keepachangelog.com/en/1.1.0/); ver window before every call — `page.text` dropped, one short line per element row instead of a JSON object, the last three actions instead of ten — roughly a tenfold reduction in the JSON-shaped state on the pages measured. On by default; `LAYA_COMPACT_BROWSER_STATE=0` turns it off. `docs/decision-models.md`. +- `laya_browser_question` (`s1a/decision_models/laya.py`): with a folded browser state, each browser question + reaches Laya as the goal and the operation (the agent's long rules dropped) and each target option as its + element's label and value. Laya fits a question's instruction and all its options into one `head_max_len` + budget, so a 23-element target head left each option about six tokens, `12: {"element": "[`, and no + element name. Browser runs want `LAYA_MAX_LEN=1536` and `LAYA_HEAD_MAX_LEN=1024`: a calendar page's target + head measures about 900 tokens. ### Changed diff --git a/docs/configuration.md b/docs/configuration.md index 7c71622..3fd90f5 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -30,8 +30,9 @@ Variables can be exported in your shell or placed in a `.env` file at the root o | `LAYA_MODEL` | `laya` model | `convaiinnovations/laya` | Hugging Face repository ID or local path for the resident Laya decision model checkpoint. | | `LAYA_SUBFOLDER` | `laya` model | *(unset)* | Optional subfolder in the checkpoint repo (e.g. `multilingual` or `typed-decisions`). | | `LAYA_DEVICE` | `laya` model | `(library default)` | PyTorch device for Laya model evaluation; passes None so the library selects CUDA, MPS, or CPU. | -| `LAYA_MAX_LEN` | `laya` model | `(checkpoint default)` | Maximum token sequence length for Laya state representation; overrides checkpoint window only when set. | -| `LAYA_HEAD_MAX_LEN` | `laya` model | `(checkpoint default)` | Maximum token sequence length for Laya decision head options; overrides checkpoint window only when set. | +| `LAYA_MAX_LEN` | `laya` model | `(checkpoint default)` | Maximum token sequence length for Laya state representation; overrides checkpoint window only when set. Browser agents want `1536`. | +| `LAYA_HEAD_MAX_LEN` | `laya` model | `(checkpoint default)` | Maximum token sequence length for Laya decision head options; overrides checkpoint window only when set. Browser agents want `1024`. | +| `LAYA_COMPACT_BROWSER_STATE` | `laya` model | `1` | Folds a browser-front state and its questions to fit Laya's window (`laya_state`, `laya_browser_question`); `0`, `false` or `no` sends them as Jev gets them. | | `CUA_S1_CHECKPOINT` | `cua` model | `cua-ai/cua-s1-nano-0.1` | Hugging Face checkpoint ID or local directory for Cua-S1 Nano option scorer. | | `CUA_S1_SUBFOLDER` | `cua` model | `text` | Subfolder within checkpoint directory containing text option scoring weights. | | `CUA_S1_DEVICE` | `cua` model | `auto` | PyTorch device used for Cua-S1 Nano evaluation (`auto`, `cpu`, `cuda`, or `mps`). | diff --git a/docs/decision-models.md b/docs/decision-models.md index d4c16ae..7b0b563 100644 --- a/docs/decision-models.md +++ b/docs/decision-models.md @@ -103,3 +103,21 @@ roughly a tenfold reduction in the JSON-shaped state's size before the tokenizer routinely filling a 512-token window and, on most pages, comfortably fitting it. It is on by default and skips anything that is not the browser front's shape; `LAYA_COMPACT_BROWSER_STATE=0` turns it off. `--model laya` on a page whose element table is still too wide for the window needs `LAYA_MAX_LEN` raised, same as before. + +The questions need the same care. Laya builds each question's row as `[CLS] instruction [SEP] options [SEP] state` +and fits the instruction and every option into one `head_max_len` budget (192 by default): past it, every option +is cut to an equal share and the instruction to what is left. The browser front's target options are JSON +objects, so a 23-element click head left each option about six tokens, `12: {"element": "[`, and Laya never +saw an element's name. With a folded state, `laya_browser_question` rewrites each browser question: the +instruction becomes the goal and the operation (the agent's rules, 446 tokens and about 550 on a target head, are +dropped), and each target option becomes its element's label and value, `Where from? = Zurich`. Measured on a +Google Flights run with this shape: 157 to 206 tokens for the operation head, 73 to 101 for the TYPE_TEXT and +PRESS_ENTER heads, 92 to 915 for the CLICK head (a calendar page offers 66 days), and 98 to 1,002 tokens of folded +state. Browser runs therefore want `LAYA_MAX_LEN=1536` and `LAYA_HEAD_MAX_LEN=1024`. + +Both folds make the request fit; they do not make the stock checkpoint drive a web form. On 2026-09-28, with both +on and those two settings, `s1a run flights --model laya` on CPU answered DONE at the first step (DONE 0.55, CLICK +0.23; 12.4 s, 1,254 input tokens over four questions) where Jev needs twelve steps. Replayed offline over Jev's +twelve recorded steps of the same task, the checkpoint picks Jev's answer on 4 of 23 questions (the operation head +and the chosen operation's target head). Laya's model card names email triage, routing, guardrails and moderation +as what its checkpoints are for; browser use would need a checkpoint fine-tuned on browser steps. diff --git a/s1a/decision_models/laya.py b/s1a/decision_models/laya.py index 0b88e5d..047f66f 100644 --- a/s1a/decision_models/laya.py +++ b/s1a/decision_models/laya.py @@ -26,6 +26,7 @@ LAYA_BROWSER_LABEL_CHARS = 40 # a row's label/value, kept over its full text (Jev's window is 32K; Laya's is not) LAYA_BROWSER_TITLE_CHARS = 80 LAYA_BROWSER_HISTORY_KEPT = 3 # of the browser front's last ten actions; the freshest ones carry the signal +LAYA_BROWSER_OPTION_CHARS = 28 # per target option: the label (and value) Laya reads, all options share one budget _FLAG_LETTERS = ( ("checked", "C"), ("selected", "S"), @@ -57,6 +58,35 @@ def without_weight_init() -> AbstractContextManager[Any]: return nullcontext() +def _laya_browser_option(option: Any) -> Any: + """A browser target option (``{"element": "[12] Where from?", "current_value": ...}``) as one short line.""" + if not (isinstance(option, dict) and "element" in option): + return option + label = str(option["element"]).split("] ", 1)[-1][:LAYA_BROWSER_OPTION_CHARS] + if option.get("option"): + label += f" / {str(option['option'])[:LAYA_BROWSER_OPTION_CHARS]}" + value = str(option.get("current_value") or "") + return label + (f" = {value[:LAYA_BROWSER_OPTION_CHARS]}" if value else "") + + +def laya_browser_question(asked: Json) -> Json: + """A browser-front choice question, shaped for Laya's one shared option budget (``head_max_len`` holds the + instruction and every option): the agent's long rules dropped, the instruction reduced to the goal and the + operation, each target option reduced to its element's label and value. Unfolded, a 23-element target head + left each option about six tokens, ``12: {"element": "[``, so Laya never saw an element's name. + Anything that is not a goal-bearing choice question passes through.""" + instructions = asked.get("instructions") + if asked.get("type") != "choice" or not (isinstance(instructions, dict) and instructions.get("goal")): + return asked + operation = instructions.get("operation") + ask = f"Which element should {operation} act on?" if operation else "Which operation comes next?" + return { + "type": "choice", + "instructions": f"Task: {instructions['goal']} {ask}", + "criteria": {key: _laya_browser_option(option) for key, option in asked["criteria"].items()}, + } + + def _laya_browser_row(row: Json) -> str: """One element row as a short line instead of a JSON object: the repeated key names (``role``, ``label``, ...) are what a tiny window can least afford. ``[CX]``-style flags stand in for the sparse boolean/id fields.""" @@ -119,7 +149,11 @@ def model(self) -> str: async def _decide(self, observation: Observation, questions: dict[str, Question]) -> Reply: asked = {name: laya_question(question) for name, question in questions.items()} - state = laya_state(observation.state) if self._compact_browser_state else observation.state + state = observation.state + if self._compact_browser_state: + state = laya_state(state) + if state is not observation.state: # a browser-shaped state: its questions are the browser heads + asked = {name: laya_browser_question(question) for name, question in asked.items()} started = time.perf_counter() try: payload = await asyncio.to_thread(self._agent.system_one, state, asked) diff --git a/tests/test_decision_models_laya.py b/tests/test_decision_models_laya.py index b96e656..b168954 100644 --- a/tests/test_decision_models_laya.py +++ b/tests/test_decision_models_laya.py @@ -196,7 +196,55 @@ def test_compaction_shrinks_the_json_size_by_an_order_of_magnitude(self) -> None self.assertGreater(len(raw) / len(compact), 8) +_TARGET_QUESTION = ChoiceQuestion( + { + "12": {"element": "[12] Where from?", "current_value": "Zurich", "role": "combobox"}, + "19": {"element": "[19] Search", "current_value": ""}, + }, + goal="find flights from Zurich to London", + operation="CLICK", + rules=("a long rule " * 50, "another long rule " * 50), +) + + +class TestLayaBrowserQuestion(TestCase): + """``laya_browser_question`` fits a browser head into Laya's one shared option budget.""" + + def test_a_target_option_becomes_its_label_and_value(self) -> None: + asked = laya_module.laya_browser_question(jev_question(_TARGET_QUESTION)) + self.assertEqual(asked["criteria"], {"12": "Where from? = Zurich", "19": "Search"}) + + def test_the_rules_are_dropped_and_the_instruction_keeps_the_goal_and_the_operation(self) -> None: + asked = laya_module.laya_browser_question(jev_question(_TARGET_QUESTION)) + self.assertIsInstance(asked["instructions"], str) + self.assertIn("find flights from Zurich to London", asked["instructions"]) + self.assertIn("CLICK", asked["instructions"]) + self.assertNotIn("long rule", asked["instructions"]) + + def test_the_operation_head_keeps_its_string_options(self) -> None: + operation = ChoiceQuestion({"CLICK": "Press a control", "DONE": "Finished"}, goal="g", rules="r") + asked = laya_module.laya_browser_question(jev_question(operation)) + self.assertEqual(asked["criteria"], {"CLICK": "Press a control", "DONE": "Finished"}) + self.assertEqual(asked["instructions"], "Task: g Which operation comes next?") + + def test_a_question_without_a_goal_passes_through(self) -> None: + for asked in (jev_question(PICK), {"type": "noul", "instructions": "safe?"}): + self.assertEqual(laya_module.laya_browser_question(asked), asked) + + class TestLayaModelCompaction(IsolatedAsyncioTestCase): + async def test_browser_questions_reach_the_agent_folded_with_the_state(self) -> None: + agent = FakeLayaAgent() + await _model(agent).decide_many(Observation(_BROWSER_STATE), {"click_target": _TARGET_QUESTION}) + ((_state, asked),) = agent.calls + self.assertEqual(asked["click_target"]["criteria"]["12"], "Where from? = Zurich") + + async def test_questions_over_a_non_browser_state_are_not_folded(self) -> None: + agent = FakeLayaAgent() + await _model(agent).decide_many(Observation({"score": 1}), {"click_target": _TARGET_QUESTION}) + ((_state, asked),) = agent.calls + self.assertEqual(asked["click_target"], jev_question(_TARGET_QUESTION)) + async def test_the_browser_state_reaching_the_agent_is_compacted_by_default(self) -> None: agent = FakeLayaAgent() question = ChoiceQuestion({"1": {"element": "[1] Search"}})