Skip to content

Commit e14bd52

Browse files
committed
test(context): use well-formed tool pairs in pending-token fixtures
The drop of id-less tool results during pairing repair (previous commit) correctly removes malformed history, but three pending-token tests fed bare `tool` messages (no tool_call_id, no opening assistant tool call) as token ballast through the restore/repair path, so they now under-counted. Production tool results always carry the originating tool_call_id, so model the fixtures realistically: an assistant message that opens a tool call plus a paired tool result, both after the last `_usage`. They survive pairing repair and keep the pending estimate intact — exercising the post-`_usage` slice accounting without depending on malformed history.
1 parent 6a3ecfe commit e14bd52

1 file changed

Lines changed: 33 additions & 11 deletions

File tree

‎tests/core/test_context_pending_tokens.py‎

Lines changed: 33 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -11,7 +11,7 @@
1111
from pathlib import Path
1212

1313
import pytest
14-
from pythinker_core.message import Message, Role
14+
from pythinker_core.message import Message, Role, ToolCall
1515

1616
from pythinker_code.soul.compaction import estimate_text_tokens, should_auto_compact
1717
from pythinker_code.soul.context import Context
@@ -22,6 +22,26 @@ def _msg(role: Role, text: str) -> Message:
2222
return Message(role=role, content=[TextPart(text=text)])
2323

2424

25+
def _assistant_call(text: str, call_id: str) -> Message:
26+
"""A well-formed assistant message that opens a tool call (id ``call_id``)."""
27+
return Message(
28+
role="assistant",
29+
content=[TextPart(text=text)],
30+
tool_calls=[
31+
ToolCall(id=call_id, function=ToolCall.FunctionBody(name="ReadFile", arguments="{}"))
32+
],
33+
)
34+
35+
36+
def _tool_result(text: str, call_id: str) -> Message:
37+
"""A well-formed tool result paired to ``call_id`` (survives pairing repair)."""
38+
return Message(role="tool", content=[TextPart(text=text)], tool_call_id=call_id)
39+
40+
41+
def _dict(msg: Message) -> dict:
42+
return json.loads(msg.model_dump_json(exclude_none=True))
43+
44+
2545
def _write_lines(path: Path, lines: list[dict]) -> None:
2646
path.write_text(
2747
"".join(json.dumps(line) + "\n" for line in lines),
@@ -169,8 +189,8 @@ async def test_revert_to_rebuilds_pending_from_messages_after_usage(tmp_path: Pa
169189
[
170190
_message_dict("user", "question"),
171191
{"role": "_usage", "token_count": 5000},
172-
_message_dict("assistant", "let me check"),
173-
_message_dict("tool", tool_text),
192+
_dict(_assistant_call("let me check", "tc0")),
193+
_dict(_tool_result(tool_text, "tc0")),
174194
{"role": "_checkpoint", "id": 0},
175195
_message_dict("user", "follow up"),
176196
{"role": "_checkpoint", "id": 1},
@@ -184,8 +204,8 @@ async def test_revert_to_rebuilds_pending_from_messages_after_usage(tmp_path: Pa
184204
# After revert to checkpoint 1: assistant, tool, and "follow up" are all after _usage
185205
expected_pending = estimate_text_tokens(
186206
[
187-
_msg("assistant", "let me check"),
188-
_msg("tool", tool_text),
207+
_assistant_call("let me check", "tc0"),
208+
_tool_result(tool_text, "tc0"),
189209
_msg("user", "follow up"),
190210
]
191211
)
@@ -248,8 +268,8 @@ async def test_restore_rebuilds_pending_for_messages_after_usage(tmp_path: Path)
248268
[
249269
_message_dict("user", "hello"),
250270
{"role": "_usage", "token_count": 10000},
251-
_message_dict("assistant", "let me read that file"),
252-
_message_dict("tool", tool_text),
271+
_dict(_assistant_call("let me read that file", "tc0")),
272+
_dict(_tool_result(tool_text, "tc0")),
253273
],
254274
)
255275

@@ -258,8 +278,8 @@ async def test_restore_rebuilds_pending_for_messages_after_usage(tmp_path: Path)
258278

259279
expected_pending = estimate_text_tokens(
260280
[
261-
_msg("assistant", "let me read that file"),
262-
_msg("tool", tool_text),
281+
_assistant_call("let me read that file", "tc0"),
282+
_tool_result(tool_text, "tc0"),
263283
]
264284
)
265285
assert ctx.token_count == 10000
@@ -383,19 +403,21 @@ async def test_pending_rebuilt_on_restore(tmp_path: Path) -> None:
383403
path = tmp_path / "ctx.jsonl"
384404
path.touch()
385405

386-
tool_msg = _msg("tool", "b" * 800)
406+
assistant_msg = _assistant_call("let me check", "tc0")
407+
tool_msg = _tool_result("b" * 800, "tc0")
387408

388409
ctx1 = Context(file_backend=path)
389410
await ctx1.append_message(_msg("user", "a" * 400))
390411
await ctx1.update_token_count(1000)
412+
await ctx1.append_message(assistant_msg)
391413
await ctx1.append_message(tool_msg)
392414
assert ctx1.token_count_with_pending > 1000
393415

394416
# Load from same file — pending is rebuilt for messages after _usage
395417
ctx2 = Context(file_backend=path)
396418
await ctx2.restore()
397419
assert ctx2.token_count == 1000
398-
expected_pending = estimate_text_tokens([tool_msg])
420+
expected_pending = estimate_text_tokens([assistant_msg, tool_msg])
399421
assert ctx2.token_count_with_pending == 1000 + expected_pending
400422

401423

0 commit comments

Comments
 (0)