From 62867b8868d371f885957c2a6ab8cf219925e8b1 Mon Sep 17 00:00:00 2001 From: elkaix Date: Tue, 15 Sep 2026 17:40:53 -0400 Subject: [PATCH 01/11] fix: flush print-mode journals before process exit Print-mode cleanup now drains prompts, stops tasks, and flushes wire journals before dispose so a termination signal cannot drop the closing records of a turn. --- .changeset/print-flush-journals-on-exit.md | 5 + .../pythinker-code/src/cli/v2/run-v2-print.ts | 120 +++++++++- .../test/cli/v2-run-print.test.ts | 208 +++++++++++++++++- .../agent-core-v2/src/agent/prompt/prompt.ts | 1 + .../src/agent/prompt/promptService.ts | 6 +- .../src/agent/task/taskService.ts | 15 +- .../agentLifecycle/agentLifecycleService.ts | 44 +++- .../test/agent/prompt/promptService.test.ts | 26 ++- .../test/agent/task/taskService.test.ts | 22 ++ .../test/agent/undo/undo.test.ts | 1 + .../test/app/gateway/gateway.test.ts | 2 +- .../agentLifecycle/agentLifecycle.test.ts | 10 + .../agentTitlePromptSourceService.test.ts | 5 +- packages/agent-gateway/test/snapshot.test.ts | 2 +- 14 files changed, 441 insertions(+), 26 deletions(-) create mode 100644 .changeset/print-flush-journals-on-exit.md diff --git a/.changeset/print-flush-journals-on-exit.md b/.changeset/print-flush-journals-on-exit.md new file mode 100644 index 000000000..200a0093d --- /dev/null +++ b/.changeset/print-flush-journals-on-exit.md @@ -0,0 +1,5 @@ +--- +"@pymodel/pythinker-code": patch +--- + +Keep print-mode session journals complete when the process exits or receives a termination signal. diff --git a/apps/pythinker-code/src/cli/v2/run-v2-print.ts b/apps/pythinker-code/src/cli/v2/run-v2-print.ts index e8cbd276d..36e4b9ad8 100644 --- a/apps/pythinker-code/src/cli/v2/run-v2-print.ts +++ b/apps/pythinker-code/src/cli/v2/run-v2-print.ts @@ -22,6 +22,7 @@ import { AgentCron, AgentGoal, IAgentLifecycleService, + IAgentLoopService, IAgentPermissionModeService, IAgentProfileService, IAgentPromptService, @@ -30,6 +31,7 @@ import { IBootstrapService, IConfigService, IEventBus, + IEventDispatcher, IHostFileSystem, ISessionIndex, IWorkspaceInstanceManager, @@ -125,6 +127,8 @@ import { const PROMPT_UI_MODE = 'print'; /** Re-check `goalActive` at least this often while waiting for goal turns. */ const GOAL_WAIT_POLL_MS = 250; +/** Re-check each agent's prompt queue while waiting for it to drain at exit. */ +const PROMPT_QUIESCE_POLL_MS = 10; /** * Slack on top of a scheduled cron fire time while waiting for the steered * turn: covers the 1s tick poll interval plus fire → inject → turn-launch @@ -197,6 +201,9 @@ export async function runV2Print( } let restorePermission = async (): Promise => {}; + let quiesceAgents = async (): Promise => {}; + let releaseQuiescence: (() => void) | undefined; + let flushWires = async (): Promise => {}; let removeTerminationCleanup: (() => void) | undefined; let cleanupPromise: Promise | undefined; let telemetryService: ITelemetryService | undefined; @@ -206,14 +213,19 @@ export async function runV2Print( setCrashPhase('shutdown'); try { await restorePermission(); + await raceWithTimeout(quiesceAgents(), CLI_SHUTDOWN_TIMEOUT_MS).catch(() => {}); } finally { try { - if (telemetryService !== undefined) { - await raceWithTimeout(telemetryService.shutdown(), CLI_SHUTDOWN_TIMEOUT_MS); - } - } finally { - await shutdownTelemetry({ timeoutMs: CLI_SHUTDOWN_TIMEOUT_MS }).catch(() => {}); + await Promise.all([ + raceWithTimeout(flushWires(), CLI_SHUTDOWN_TIMEOUT_MS).catch(() => {}), + telemetryService !== undefined + ? raceWithTimeout(telemetryService.shutdown(), CLI_SHUTDOWN_TIMEOUT_MS) + : Promise.resolve(), + shutdownTelemetry({ timeoutMs: CLI_SHUTDOWN_TIMEOUT_MS }).catch(() => {}), + ]); app.dispose(); + } finally { + releaseQuiescence?.(); } } })()); @@ -262,6 +274,10 @@ export async function runV2Print( const resolved = await resolveNativeSession(app, opts, workDir, defaultModel, stderr); restorePermission = resolved.restorePermission; + quiesceAgents = async () => { + releaseQuiescence = await quiesceSessionAgents(resolved.session, resolved.agent); + }; + flushWires = () => flushSessionWires(resolved.session, resolved.agent); telemetryService.setContext({ sessionId: resolved.session.id, model: resolved.telemetryModel }); setTelemetryContext({ sessionId: resolved.session.id }); @@ -935,6 +951,100 @@ function countPendingBackgroundTasks(session: ISessionScopeHandle): number { return count; } +function collectSessionAgentHandles( + session: ISessionScopeHandle, + mainAgent: IAgentScopeHandle, +): IAgentScopeHandle[] { + const agentManager = session.accessor.get(IAgentLifecycleService); + const handles = new Set([mainAgent]); + for (const agent of agentManager.list()) { + const handle = agentManager.handleOf(agent.agentId); + if (handle !== undefined) handles.add(handle); + } + return [...handles]; +} + +async function quiesceSessionAgents( + session: ISessionScopeHandle, + mainAgent: IAgentScopeHandle, +): Promise<(() => void) | undefined> { + const handles = collectSessionAgentHandles(session, mainAgent); + const promptServices = handles.flatMap((handle) => { + try { + return [handle.accessor.get(IAgentPromptService)]; + } catch { + return []; + } + }); + const loops = handles.flatMap((handle) => { + try { + return [handle.accessor.get(IAgentLoopService)]; + } catch { + return []; + } + }); + await Promise.allSettled( + handles.flatMap((handle) => { + try { + return [handle.accessor.get(IAgentTaskService).stopAllOnExit('Session closed')]; + } catch { + return []; + } + }), + ); + for (;;) { + await Promise.allSettled(promptServices.map((service) => service.drain())); + for (const loop of loops) { + for (const turnId of loop.status().pendingTurnIds) loop.cancel(turnId); + loop.cancel(); + } + await Promise.allSettled(loops.map((loop) => loop.settled())); + const guards: { dispose(): void }[] = []; + let frozen = true; + for (const loop of loops) { + let guard: { dispose(): void } | undefined; + try { + guard = loop.tryAcquireQuiescence(); + } catch { + continue; + } + if (guard === undefined) { + frozen = false; + break; + } + guards.push(guard); + } + const busy = promptServices.some((service) => { + try { + const snapshot = service.list(); + return snapshot.launching || snapshot.active !== undefined || snapshot.pending.length > 0; + } catch { + return false; + } + }); + if (frozen && !busy) { + return () => { + for (const guard of guards) guard.dispose(); + }; + } + for (const guard of guards) guard.dispose(); + await new Promise((resolve) => { + setTimeout(resolve, PROMPT_QUIESCE_POLL_MS); + }); + } +} + +async function flushSessionWires( + session: ISessionScopeHandle, + mainAgent: IAgentScopeHandle, +): Promise { + await Promise.allSettled( + collectSessionAgentHandles(session, mainAgent).map((handle) => + handle.accessor.get(IEventDispatcher).flush(), + ), + ); +} + async function drainBackgroundTasks( session: ISessionScopeHandle, ceilingS: number | undefined, diff --git a/apps/pythinker-code/test/cli/v2-run-print.test.ts b/apps/pythinker-code/test/cli/v2-run-print.test.ts index f43a8b088..802923120 100644 --- a/apps/pythinker-code/test/cli/v2-run-print.test.ts +++ b/apps/pythinker-code/test/cli/v2-run-print.test.ts @@ -8,6 +8,7 @@ import { AgentCron, AgentGoal, IAgentLifecycleService, + IAgentLoopService, IAgentPermissionModeService, IAgentProfileService, IAgentPromptService, @@ -17,6 +18,7 @@ import { IBootstrapService, IConfigService, IEventBus, + IEventDispatcher, IFileSystemStorageService, ISessionIndex, ISessionManager, @@ -179,9 +181,21 @@ function makeFakeHarness() { }), }; }), + drain: vi.fn(async () => {}), + list: vi.fn(() => ({ launching: false, active: undefined, pending: [] })), + }, + ], + [IAgentTaskService, { list: vi.fn(() => []), stopAllOnExit: vi.fn(async () => []) }], + [IEventDispatcher, { flush: vi.fn(async () => {}) }], + [ + IAgentLoopService, + { + status: vi.fn(() => ({ state: 'idle', pendingTurnIds: [] })), + cancel: vi.fn(() => false), + settled: vi.fn(async () => {}), + tryAcquireQuiescence: vi.fn(() => ({ dispose: vi.fn() })), }, ], - [IAgentTaskService, { list: vi.fn(() => []) }], [ IAgentScopeContext, makeAgentScopeContext({ agentId: 'main', agentScope: 'agents/main' }), @@ -614,4 +628,196 @@ describe('runV2Print', () => { expect(initOrder).toBeDefined(); expect(reconcileOrder).toBeGreaterThan(initOrder!); }); + + it('flushes the wire journal before disposing the app', async () => { + const stdout = writer(); + const stderr = writer(); + const { app, agentServices } = makeFakeHarness(); + + mocks.bootstrap.mockReturnValue({ app }); + mocks.ensureMainAgent.mockResolvedValue({ agentId: 'main', generation: 1 }); + + await runV2Print(opts() as never, '1.2.3-test', { stdout, stderr }); + + const dispatcher = agentServices.get(IEventDispatcher) as { + flush: ReturnType; + }; + expect(dispatcher.flush).toHaveBeenCalled(); + const flushOrder = dispatcher.flush.mock.invocationCallOrder[0]; + const disposeOrder = app.dispose.mock.invocationCallOrder[0]; + expect(flushOrder).toBeDefined(); + expect(disposeOrder).toBeGreaterThan(flushOrder!); + }); + + it('flushes the wire journal when the turn fails', async () => { + const stdout = writer(); + const stderr = writer(); + const { app, agentServices } = makeFakeHarness(); + + const promptService = agentServices.get(IAgentPromptService) as { + enqueue: ReturnType; + }; + promptService.enqueue.mockResolvedValueOnce({ + launched: Promise.resolve({ + id: 1, + result: Promise.resolve({ + type: 'failed', + error: { code: 'provider.overloaded', message: 'llm request failed' }, + }), + }), + }); + + mocks.bootstrap.mockReturnValue({ app }); + mocks.ensureMainAgent.mockResolvedValue({ agentId: 'main', generation: 1 }); + + await expect(runV2Print(opts() as never, '1.2.3-test', { stdout, stderr })).rejects.toThrow( + 'provider.overloaded: llm request failed', + ); + + const dispatcher = agentServices.get(IEventDispatcher) as { + flush: ReturnType; + }; + expect(dispatcher.flush).toHaveBeenCalled(); + const flushOrder = dispatcher.flush.mock.invocationCallOrder[0]; + const disposeOrder = app.dispose.mock.invocationCallOrder[0]; + expect(disposeOrder).toBeGreaterThan(flushOrder!); + }); + + it('does not let a wire flush failure mask the turn outcome', async () => { + const stdout = writer(); + const stderr = writer(); + const { app, agentServices } = makeFakeHarness(); + + const promptService = agentServices.get(IAgentPromptService) as { + enqueue: ReturnType; + }; + promptService.enqueue.mockResolvedValueOnce({ + launched: Promise.resolve({ + id: 1, + result: Promise.resolve({ + type: 'failed', + error: { code: 'provider.overloaded', message: 'llm request failed' }, + }), + }), + }); + const dispatcher = agentServices.get(IEventDispatcher) as { + flush: ReturnType; + }; + dispatcher.flush.mockRejectedValueOnce(new Error('disk full')); + + mocks.bootstrap.mockReturnValue({ app }); + mocks.ensureMainAgent.mockResolvedValue({ agentId: 'main', generation: 1 }); + + await expect(runV2Print(opts() as never, '1.2.3-test', { stdout, stderr })).rejects.toThrow( + 'provider.overloaded: llm request failed', + ); + expect(app.dispose).toHaveBeenCalled(); + }); + + it('cancels and settles the active turn before flushing on a termination signal', async () => { + const stdout = writer(); + const stderr = writer(); + const { app, agentServices } = makeFakeHarness(); + + const order: string[] = []; + const loop = agentServices.get(IAgentLoopService) as { + status: ReturnType; + cancel: ReturnType; + settled: ReturnType; + tryAcquireQuiescence: ReturnType; + }; + loop.status.mockReturnValue({ state: 'running', pendingTurnIds: [] }); + loop.cancel.mockImplementation(() => { + if (!order.includes('cancel')) order.push('cancel'); + return true; + }); + loop.settled = vi.fn(async () => { + if (!order.includes('settled')) order.push('settled'); + }); + const guardDispose = vi.fn(); + loop.tryAcquireQuiescence = vi.fn(() => ({ dispose: guardDispose })); + const taskService = agentServices.get(IAgentTaskService) as { + stopAllOnExit: ReturnType; + }; + taskService.stopAllOnExit = vi.fn(async () => { + if (!order.includes('stop')) order.push('stop'); + return []; + }); + const dispatcher = agentServices.get(IEventDispatcher) as { + flush: ReturnType; + }; + dispatcher.flush = vi.fn(async () => { + order.push('flush'); + }); + + const promptService = agentServices.get(IAgentPromptService) as { + enqueue: ReturnType; + drain: ReturnType; + list: ReturnType; + }; + let settleTurn!: (result: unknown) => void; + promptService.enqueue.mockResolvedValueOnce({ + launched: Promise.resolve({ + id: 1, + result: new Promise((resolve) => { + settleTurn = resolve; + }), + }), + }); + let promptPhase: 'launching' | 'active' | 'empty' = 'launching'; + promptService.list = vi.fn(() => { + if (promptPhase === 'launching') { + return { launching: true, active: undefined, pending: [] }; + } + if (promptPhase === 'active') { + return { launching: false, active: { id: 'p1' }, pending: [] }; + } + return { launching: false, active: undefined, pending: [] }; + }); + + const handlers = new Map Promise>(); + const fakeProcess = { + once: (signal: string, handler: () => Promise) => { + handlers.set(signal, handler); + }, + off: () => {}, + exit: vi.fn((code?: number) => { + order.push(`exit:${code}`); + }), + }; + + mocks.bootstrap.mockReturnValue({ app }); + mocks.ensureMainAgent.mockResolvedValue({ agentId: 'main', generation: 1 }); + + const run = runV2Print(opts() as never, '1.2.3-test', { + stdout, + stderr, + process: fakeProcess as never, + }); + const outcome = run.catch((error: unknown) => error); + for (let i = 0; i < 100 && !handlers.has('SIGINT'); i++) { + await new Promise((resolve) => setTimeout(resolve, 5)); + } + const onSigint = handlers.get('SIGINT')!; + settleTurn({ type: 'cancelled', steps: 0, reason: new Error('aborted') }); + const sigintRun = onSigint(); + for (let i = 0; i < 100 && !order.includes('settled'); i++) { + await new Promise((resolve) => setTimeout(resolve, 5)); + } + await new Promise((resolve) => setTimeout(resolve, 30)); + expect(promptService.drain).toHaveBeenCalled(); + expect(order).toEqual(['stop', 'cancel', 'settled']); + promptPhase = 'active'; + await new Promise((resolve) => setTimeout(resolve, 30)); + expect(order).toEqual(['stop', 'cancel', 'settled']); + promptPhase = 'empty'; + await sigintRun; + + expect(order).toEqual(['stop', 'cancel', 'settled', 'flush', 'exit:130']); + expect(loop.tryAcquireQuiescence).toHaveBeenCalled(); + const lastGuardRelease = guardDispose.mock.invocationCallOrder.at(-1); + const appDisposeOrder = app.dispose.mock.invocationCallOrder[0]; + expect(lastGuardRelease).toBeGreaterThan(appDisposeOrder!); + expect(await outcome).toBeInstanceOf(Error); + }); }); diff --git a/packages/agent-core-v2/src/agent/prompt/prompt.ts b/packages/agent-core-v2/src/agent/prompt/prompt.ts index 5a9c51ff3..d4ff315d9 100644 --- a/packages/agent-core-v2/src/agent/prompt/prompt.ts +++ b/packages/agent-core-v2/src/agent/prompt/prompt.ts @@ -49,6 +49,7 @@ export interface PromptHandle extends PromptSnapshot { export interface PromptQueueSnapshot { readonly active: PromptSnapshot | undefined; readonly pending: readonly PromptSnapshot[]; + readonly launching: boolean; } export interface PromptPayload { diff --git a/packages/agent-core-v2/src/agent/prompt/promptService.ts b/packages/agent-core-v2/src/agent/prompt/promptService.ts index 4716f68c7..41f916c8c 100644 --- a/packages/agent-core-v2/src/agent/prompt/promptService.ts +++ b/packages/agent-core-v2/src/agent/prompt/promptService.ts @@ -401,7 +401,11 @@ export class AgentPromptService implements IAgentPromptService { } list(): PromptQueueSnapshot { - return { active: this.active === undefined ? undefined : snapshot(this.active), pending: this.pending.map(snapshot) }; + return { + active: this.active === undefined ? undefined : snapshot(this.active), + pending: this.pending.map(snapshot), + launching: this.launching, + }; } async steer(promptIds: readonly string[]): Promise { diff --git a/packages/agent-core-v2/src/agent/task/taskService.ts b/packages/agent-core-v2/src/agent/task/taskService.ts index 385db169a..a9413edb9 100644 --- a/packages/agent-core-v2/src/agent/task/taskService.ts +++ b/packages/agent-core-v2/src/agent/task/taskService.ts @@ -808,15 +808,22 @@ export class AgentTaskService extends Disposable implements IAgentTaskService { async stopAllOnExit(reason: string): Promise { if (this.keepAliveOnExit()) return []; const active = this.list(true); - await Promise.all( + await Promise.allSettled( active .filter((task) => task.detached === true) .map(async (task) => { const entry = this.tasks.get(task.taskId); if (entry === undefined) return; - entry.stoppedOnExit = true; - entry.terminalNotificationSuppressed = true; - await this.persistLive(entry); + try { + entry.stoppedOnExit = true; + entry.terminalNotificationSuppressed = true; + await this.persistLive(entry); + } catch (error: unknown) { + this.log.error('terminal notification suppression failed', { + taskId: task.taskId, + error, + }); + } }), ); return this.stopAll(reason); diff --git a/packages/agent-core-v2/src/session/agentLifecycle/agentLifecycleService.ts b/packages/agent-core-v2/src/session/agentLifecycle/agentLifecycleService.ts index a9fe0d158..bca4d978a 100644 --- a/packages/agent-core-v2/src/session/agentLifecycle/agentLifecycleService.ts +++ b/packages/agent-core-v2/src/session/agentLifecycle/agentLifecycleService.ts @@ -68,6 +68,9 @@ import { let nextAgentId = 0; +const REMOVE_PROMPT_QUIESCE_TIMEOUT_MS = 3_000; +const REMOVE_PROMPT_QUIESCE_POLL_MS = 10; + export class AgentLifecycleService extends Disposable implements IAgentLifecycleService { declare readonly _serviceBrand: undefined; private readonly roster = new Map(); @@ -472,18 +475,41 @@ export class AgentLifecycleService extends Disposable implements IAgentLifecycle const compactionSettled = compaction?.promise.catch(() => undefined) ?? Promise.resolve(); const prompt = handle.accessor.get(IAgentPromptService); await phase(async () => { - await prompt.drain(reason); - for (const turnId of loop.status().pendingTurnIds) { - loop.cancel(turnId, reason); - } - loop.cancel(undefined, reason); if (compaction !== null && !compaction.abortController.signal.aborted) { compaction.abortController.abort(reason); } - await Promise.all([loop.settled(), compactionSettled]); - }); - await phase(() => { - quiescence = loop.tryAcquireQuiescence(); + const deadline = Date.now() + REMOVE_PROMPT_QUIESCE_TIMEOUT_MS; + for (;;) { + await prompt.drain(reason); + for (const turnId of loop.status().pendingTurnIds) { + loop.cancel(turnId, reason); + } + loop.cancel(undefined, reason); + await Promise.all([loop.settled(), compactionSettled]); + let idle = true; + try { + const snapshot = prompt.list(); + idle = + !snapshot.launching && snapshot.active === undefined && snapshot.pending.length === 0; + } catch { + idle = true; + } + if (idle) { + try { + const guard = loop.tryAcquireQuiescence(); + if (guard !== undefined) { + quiescence = guard; + break; + } + } catch { + break; + } + } + if (Date.now() >= deadline) break; + await new Promise((resolve) => { + setTimeout(resolve, REMOVE_PROMPT_QUIESCE_POLL_MS); + }); + } }); await phase(() => handle.accessor.get(IAgentTaskService).stopAllOnExit('Session closed')); await phase(() => handle.accessor.get(IEventDispatcher).flush()); diff --git a/packages/agent-core-v2/test/agent/prompt/promptService.test.ts b/packages/agent-core-v2/test/agent/prompt/promptService.test.ts index 587f891ff..72a72dfed 100644 --- a/packages/agent-core-v2/test/agent/prompt/promptService.test.ts +++ b/packages/agent-core-v2/test/agent/prompt/promptService.test.ts @@ -268,7 +268,7 @@ describe('AgentPromptService', () => { await expect(handle.completion).resolves.toMatchObject({ state: 'cancelled' }); await expect(handle.launched).resolves.toBeUndefined(); expect(enqueueSpy).not.toHaveBeenCalled(); - expect(prompt.list()).toEqual({ active: undefined, pending: [] }); + expect(prompt.list()).toEqual({ active: undefined, pending: [], launching: false }); expect(aborted.map((event) => event.promptId)).toEqual(['launching']); }); @@ -359,7 +359,27 @@ describe('AgentPromptService', () => { it('keeps injections outside the prompt queue', async () => { const { prompt } = harness(); await prompt.inject({ ...message('system'), origin: { kind: 'injection', variant: 'test' } }); - expect(prompt.list()).toEqual({ active: undefined, pending: [] }); + expect(prompt.list()).toEqual({ active: undefined, pending: [], launching: false }); + }); + + it('marks the launch window as busy in the queue snapshot', async () => { + const { prompt } = harness(); + let releaseHook!: () => void; + prompt.hooks.onBeforeSubmitPrompt.register('gate', async (_ctx, next) => { + await new Promise((resolve) => { + releaseHook = resolve; + }); + await next(); + }); + const enqueued = prompt.enqueue({ message: message('launching') }); + await vi.waitFor(() => { + expect(prompt.list().launching).toBe(true); + }); + expect(prompt.list().active).toBeUndefined(); + expect(prompt.list().pending).toEqual([]); + releaseHook(); + await enqueued; + expect(prompt.list().launching).toBe(false); }); it('settles blocked prompts', async () => { @@ -410,7 +430,7 @@ describe('AgentPromptService', () => { expect(handle.state).toBe('failed'); await expect(handle.launched).resolves.toBeUndefined(); await expect(handle.completion).resolves.toMatchObject({ state: 'failed', result: undefined }); - expect(prompt.list()).toEqual({ active: undefined, pending: [] }); + expect(prompt.list()).toEqual({ active: undefined, pending: [], launching: false }); }); it('replaces an unsupported prompt image with a text notice at the history funnel', async () => { diff --git a/packages/agent-core-v2/test/agent/task/taskService.test.ts b/packages/agent-core-v2/test/agent/task/taskService.test.ts index 17cc036fc..a26a7d0f2 100644 --- a/packages/agent-core-v2/test/agent/task/taskService.test.ts +++ b/packages/agent-core-v2/test/agent/task/taskService.test.ts @@ -545,6 +545,28 @@ describe('AgentTaskService', () => { } }); + it('stopAllOnExit still stops tasks when suppression persistence fails', async () => { + let writes = 0; + ix.stub(IAtomicDocumentStore, { + get: async () => undefined, + set: async () => { + writes += 1; + if (writes === 1) throw new Error('disk full'); + }, + delete: async () => {}, + list: async () => [], + }); + const svc = ix.get(IAgentTaskService); + const first = svc.registerTask(fakeProcessTask()); + const second = svc.registerTask(fakeProcessTask()); + + const stopped = await svc.stopAllOnExit('Session closed'); + + expect(stopped.map((info) => info.taskId).toSorted()).toEqual([first, second].toSorted()); + expect(svc.getTask(first)?.status).toBe('killed'); + expect(svc.getTask(second)?.status).toBe('killed'); + }); + it('stopAllOnExit does not persist a foreground-only task', async () => { const writes = stubTaskWrites(); const svc = ix.get(IAgentTaskService); diff --git a/packages/agent-core-v2/test/agent/undo/undo.test.ts b/packages/agent-core-v2/test/agent/undo/undo.test.ts index 9077c510f..a5416fc0d 100644 --- a/packages/agent-core-v2/test/agent/undo/undo.test.ts +++ b/packages/agent-core-v2/test/agent/undo/undo.test.ts @@ -538,6 +538,7 @@ describe('AgentConversationUndoService', () => { ctx.appendTurnExchange('u2', 'a2'); const list = vi.spyOn(ctx.get(IAgentPromptService), 'list').mockReturnValue({ active: undefined, + launching: false, pending: [ { id: 'queued', diff --git a/packages/agent-core-v2/test/app/gateway/gateway.test.ts b/packages/agent-core-v2/test/app/gateway/gateway.test.ts index efeb3ad22..a6c2ead25 100644 --- a/packages/agent-core-v2/test/app/gateway/gateway.test.ts +++ b/packages/agent-core-v2/test/app/gateway/gateway.test.ts @@ -59,7 +59,7 @@ describe('RestGateway', () => { submit: () => Promise.resolve(undefined), submitSteer: () => Promise.resolve(undefined), steer: () => Promise.resolve([]), - list: () => ({ active: undefined, pending: [] }), + list: () => ({ active: undefined, pending: [], launching: false }), abort: () => true, drain: () => Promise.resolve(), inject: () => Promise.resolve(undefined), diff --git a/packages/agent-core-v2/test/session/agentLifecycle/agentLifecycle.test.ts b/packages/agent-core-v2/test/session/agentLifecycle/agentLifecycle.test.ts index 2f2b338b8..755f5d410 100644 --- a/packages/agent-core-v2/test/session/agentLifecycle/agentLifecycle.test.ts +++ b/packages/agent-core-v2/test/session/agentLifecycle/agentLifecycle.test.ts @@ -404,6 +404,7 @@ describe('AgentLifecycleService', () => { ix.stub(IAgentPromptService, { _serviceBrand: undefined, drain: promptDrain, + list: () => ({ launching: false, active: undefined, pending: [] }), } as unknown as IAgentPromptService); ix.stub(ITelemetryService, { _serviceBrand: undefined, @@ -559,6 +560,15 @@ describe('AgentLifecycleService', () => { expect(svc.handleOf('main')).toBeUndefined(); }); + it('remove flushes the agent wire journal before disposal', async () => { + const svc = ix.get(IAgentLifecycleService); + await svc.create({ agentId: 'main' }); + const dispatcher = svc.handleOf('main')!.accessor.get(IEventDispatcher); + const flush = vi.spyOn(dispatcher, 'flush'); + await svc.remove(svc.get('main')!); + expect(flush).toHaveBeenCalled(); + }); + it('remove keeps the lifecycle context active through async scope teardown', async () => { const svc = ix.get(IAgentLifecycleService); const bus = ix.get(ISessionEventBus); diff --git a/packages/agent-core-v2/test/session/sessionTitle/agentTitlePromptSourceService.test.ts b/packages/agent-core-v2/test/session/sessionTitle/agentTitlePromptSourceService.test.ts index 919364eb5..46a23d58b 100644 --- a/packages/agent-core-v2/test/session/sessionTitle/agentTitlePromptSourceService.test.ts +++ b/packages/agent-core-v2/test/session/sessionTitle/agentTitlePromptSourceService.test.ts @@ -41,7 +41,7 @@ describe('AgentTitlePromptSource', () => { beforeEach(() => { liveMessages = []; - queue = { active: undefined, pending: [] }; + queue = { active: undefined, pending: [], launching: false }; disposables = new DisposableStore(); ix = createServices(disposables, { additionalServices: (reg) => { @@ -60,6 +60,7 @@ describe('AgentTitlePromptSource', () => { liveMessages = [userMessage('one', 'First entry')]; queue = { active: undefined, + launching: false, pending: [ { id: 'two', @@ -130,6 +131,7 @@ describe('AgentTitlePromptSource', () => { message: userMessage('one', 'Same entry'), }, pending: [], + launching: false, }; await expect(ix.get(IAgentTitlePromptSource).firstUserPrompts(3)).resolves.toEqual(['Same entry']); @@ -179,6 +181,7 @@ describe('AgentTitlePromptSource', () => { message: userMessage('two', 'in-progress question'), }, pending: [], + launching: false, }; await expect(ix.get(IAgentTitlePromptSource).digestExcerpt()).resolves.toEqual({ diff --git a/packages/agent-gateway/test/snapshot.test.ts b/packages/agent-gateway/test/snapshot.test.ts index 0515d4b37..7aa57d652 100644 --- a/packages/agent-gateway/test/snapshot.test.ts +++ b/packages/agent-gateway/test/snapshot.test.ts @@ -58,7 +58,7 @@ describe('server-v2 snapshot route enrichment', () => { [IAgentContextMemoryService, { get: () => [] }], [ IAgentPromptService, - { list: () => ({ active: { id: promptId }, pending: [] }) }, + { list: () => ({ active: { id: promptId }, pending: [], launching: false }) }, ], [IWireService, { flush: async () => {} }], [IAgentScopeContext, { scope: () => 'scope/sess_snapshot' }], From 3758dd91b22e2be7586c489ee1971065739925ad Mon Sep 17 00:00:00 2001 From: elkaix Date: Tue, 15 Sep 2026 18:14:25 -0400 Subject: [PATCH 02/11] fix: point compaction notes at the session event log Compaction records journal line ranges and appends a recovery footer so later turns can Read exact earlier outputs from wire.jsonl. Empty-history overflow shrink fails closed. --- .changeset/compaction-recovery-pointer.md | 5 + .../src/agent/contextMemory/contextEvents.ts | 3 + .../src/agent/contextMemory/contextMemory.ts | 2 + .../contextMemory/contextMemoryService.ts | 1 + .../fullCompaction/compaction-instruction.md | 2 + .../fullCompaction/compactionInstruction.ts | 15 ++ .../src/agent/fullCompaction/compactionOps.ts | 17 ++ .../fullCompaction/context-recovery-footer.md | 10 + .../agent/fullCompaction/contextRecovery.ts | 28 +++ .../fullCompaction/fullCompactionService.ts | 52 ++++- .../toolResultTruncation.ts | 2 + .../toolResultTruncationService.ts | 9 +- .../src/agent/tools/os/read/read.md | 1 + .../src/agent/tools/os/read/read.ts | 1 + .../src/agent/tools/os/read/readTool.ts | 51 ++++- packages/agent-core-v2/src/wire/record.ts | 5 + packages/agent-core-v2/src/wire/wire.ts | 3 + .../agent-core-v2/src/wire/wireService.ts | 33 +++ .../fullCompaction/fullCompaction.test.ts | 212 +++++++++++++++++- .../agent-core-v2/test/agent/loop/stubs.ts | 2 +- .../test/agent/task/taskService.test.ts | 3 + .../agent/toolExecutor/toolExecutor.test.ts | 1 + .../test/agent/toolResultTruncation/stubs.ts | 1 + packages/agent-core-v2/test/index.test.ts | 1 + .../os/backends/node-local/tools/read.test.ts | 91 ++++++++ .../test/state/builtinReplayableKeys.ts | 3 +- .../test/state/eventDispatcher.test.ts | 3 + packages/agent-core-v2/test/wire/stubs.ts | 3 + 28 files changed, 536 insertions(+), 24 deletions(-) create mode 100644 .changeset/compaction-recovery-pointer.md create mode 100644 packages/agent-core-v2/src/agent/fullCompaction/compactionInstruction.ts create mode 100644 packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md create mode 100644 packages/agent-core-v2/src/agent/fullCompaction/contextRecovery.ts diff --git a/.changeset/compaction-recovery-pointer.md b/.changeset/compaction-recovery-pointer.md new file mode 100644 index 000000000..e8dc82bc3 --- /dev/null +++ b/.changeset/compaction-recovery-pointer.md @@ -0,0 +1,5 @@ +--- +"@pymodel/pythinker-code": patch +--- + +Point compacted conversation notes at the on-disk event log so later turns can recover exact outputs. diff --git a/packages/agent-core-v2/src/agent/contextMemory/contextEvents.ts b/packages/agent-core-v2/src/agent/contextMemory/contextEvents.ts index 2021c8bd7..3300d4ce6 100644 --- a/packages/agent-core-v2/src/agent/contextMemory/contextEvents.ts +++ b/packages/agent-core-v2/src/agent/contextMemory/contextEvents.ts @@ -63,6 +63,9 @@ const contextCompactionBaseShape = { keptHeadUserMessageCount: z.number().optional(), droppedCount: z.number().optional(), legacyTail: z.boolean().optional(), + wireLines: z + .object({ start: z.number().int().nonnegative(), end: z.number().int().nonnegative() }) + .optional(), }; const contextApplyCompactionSchema = z.union([ diff --git a/packages/agent-core-v2/src/agent/contextMemory/contextMemory.ts b/packages/agent-core-v2/src/agent/contextMemory/contextMemory.ts index 1e7162b96..9a490ec9a 100644 --- a/packages/agent-core-v2/src/agent/contextMemory/contextMemory.ts +++ b/packages/agent-core-v2/src/agent/contextMemory/contextMemory.ts @@ -1,4 +1,5 @@ import { createDecorator } from "#/_base/di/instantiation"; +import type { WireLineRange } from '#/wire/record'; import type { UndoCut } from './contextOps'; import type { LoopRecordedEvent } from './loopEventFold'; @@ -15,6 +16,7 @@ export interface ContextCompactionInput { readonly keptUserMessageCount?: number; readonly keptHeadUserMessageCount?: number; readonly droppedCount?: number; + readonly wireLines?: WireLineRange; } export interface ContextCompactionResult { diff --git a/packages/agent-core-v2/src/agent/contextMemory/contextMemoryService.ts b/packages/agent-core-v2/src/agent/contextMemory/contextMemoryService.ts index a2ccddc31..f29cc89e8 100644 --- a/packages/agent-core-v2/src/agent/contextMemory/contextMemoryService.ts +++ b/packages/agent-core-v2/src/agent/contextMemory/contextMemoryService.ts @@ -131,6 +131,7 @@ export class AgentContextMemoryService extends Disposable implements IAgentConte keptUserMessageCount: result.keptUserMessageCount, keptHeadUserMessageCount: result.keptHeadUserMessageCount, droppedCount: result.droppedCount, + wireLines: input.wireLines, }), ); this.tokenCounting.rebase(this.scopeContext.agentContext, { diff --git a/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md b/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md index 4f0b4279c..fc30e61a3 100644 --- a/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md +++ b/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md @@ -52,6 +52,8 @@ continue: here is one less thing the next turn must rediscover. Include any required format for the final answer. +This conversation's event log stays on disk and a recovery pointer is appended below your note automatically, so you need not reproduce long outputs verbatim — keep exact identifiers, key values and error lines, and name anything the next turn should look up. + Your TODO list is re-attached automatically below this note from its live source, so do not transcribe it — copying it wastes space and can contradict the live version. What that list cannot hold is the reasoning between tasks — why one diff --git a/packages/agent-core-v2/src/agent/fullCompaction/compactionInstruction.ts b/packages/agent-core-v2/src/agent/fullCompaction/compactionInstruction.ts new file mode 100644 index 000000000..cd3a10f91 --- /dev/null +++ b/packages/agent-core-v2/src/agent/fullCompaction/compactionInstruction.ts @@ -0,0 +1,15 @@ +import { renderPrompt } from '#/_base/utils/render-prompt'; + +import compactionInstructionTemplate from './compaction-instruction.md?raw'; + +export interface CompactionInstructionInput { + readonly customInstruction?: string; +} + +export function renderCompactionInstruction(input: CompactionInstructionInput): string { + const customInstruction = input.customInstruction?.trim() ?? ''; + return renderPrompt(compactionInstructionTemplate, { + custom_instruction_block: + customInstruction.length > 0 ? `\nOptional user instruction:\n${customInstruction}\n` : '', + }).trimEnd(); +} diff --git a/packages/agent-core-v2/src/agent/fullCompaction/compactionOps.ts b/packages/agent-core-v2/src/agent/fullCompaction/compactionOps.ts index c58165945..a1b2bb25a 100644 --- a/packages/agent-core-v2/src/agent/fullCompaction/compactionOps.ts +++ b/packages/agent-core-v2/src/agent/fullCompaction/compactionOps.ts @@ -3,6 +3,12 @@ import { z } from 'zod'; import { AgentEvent2, type AgentDomainTrait } from '#/app/event/event2'; import { defineState } from '#/state/state'; +import { + ContextApplyCompaction, + ContextClear, + type ContextApplyCompactionPayload, +} from '#/agent/contextMemory/contextEvents'; +import type { WireLineRange } from '#/wire/record'; import type { CompactionBeginData, CompactionResult, CompactionSource } from './types'; @@ -123,3 +129,14 @@ export const fullCompactionKey = defineState( s.phase = 'idle'; } }); + +export const fullCompactionWireRangesKey = defineState( + 'fullCompaction.wireRanges', + () => [], +) + .replayable({ schema: z.custom() }) + .on(ContextApplyCompaction, (s, e) => { + const wireLines = (e as unknown as ContextApplyCompactionPayload).wireLines; + return wireLines === undefined ? undefined : [...s, wireLines]; + }) + .on(ContextClear, (s) => (s.length === 0 ? undefined : [])); diff --git a/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md b/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md new file mode 100644 index 000000000..5aba814ea --- /dev/null +++ b/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md @@ -0,0 +1,10 @@ +## Context Recovery +Everything before this note is still on disk in this agent's event log (read-only, append-only): + ${wire_path} +${window_lines} +If you need exact command output, file contents, error text, or the wording of an earlier request, look it up there instead of guessing. How to read it: +- Layout: one file per agent. agents/main/ is the main agent; each subagent has its own agents//wire.jsonl. A parent's log holds only the Agent tool call and the subagent's returned result — the subagent's own steps are in its own file. +- Format: one JSON record per line, append-only; `type` says what it is. The conversation is in `context.append_message` (user prompts) and `context.append_loop_event` (event.type: step.begin | content.part [text|think] | tool.call | tool.result | step.end). Every other type (llm.request, usage.record, token_counting.measured, metadata, profile.bind, …) is bookkeeping — skip it. +- Boundaries: `context.apply_compaction` marks a compaction (older lines stay in the file; grep for it to find exact boundaries). `context.undo` count=N retracts the previous N messages — treat retracted content as never having happened. `context.clear` resets the conversation. +- Externalized content: tool results over 50k chars are stored truncated, with an `output_path` to a tool-results/*.txt file holding the full text. Media parts are blob references, not inline. +- Reading: lines are long JSON (often 10k+ chars). Grep the file for a keyword to get line numbers, then Read exactly that line (line_offset=N, n_lines=1) — Read returns wire.jsonl lines whole up to ~150k chars. To pull one field with real newlines: sed -n 'Np' wire.jsonl | jq -r '.event.result.output'. Never Read large ranges — a handful of records can exceed the per-call byte cap. diff --git a/packages/agent-core-v2/src/agent/fullCompaction/contextRecovery.ts b/packages/agent-core-v2/src/agent/fullCompaction/contextRecovery.ts new file mode 100644 index 000000000..430f821c4 --- /dev/null +++ b/packages/agent-core-v2/src/agent/fullCompaction/contextRecovery.ts @@ -0,0 +1,28 @@ +import { renderPrompt } from '#/_base/utils/render-prompt'; +import type { WireLineRange } from '#/wire/record'; + +import contextRecoveryTemplate from './context-recovery-footer.md?raw'; + +export const CONTEXT_RECOVERY_HEADING = '## Context Recovery'; + +export interface ContextRecoveryPointer { + readonly journalPath: string; + readonly windows: readonly WireLineRange[]; +} + +export function renderContextRecoveryPointer(pointer: ContextRecoveryPointer): string { + const windows = pointer.windows; + const summarized = windows.length - 1; + const lines = windows.map((range, index) => { + const label = `window ${String(index + 1)}: lines ${String(range.start)}–${String(range.end)}`; + return index === summarized ? `${label} ← the conversation this note summarizes` : label; + }); + const nextStart = windows[summarized]!.end + 1; + lines.push( + `window ${String(windows.length + 1)} (the one you are in now) starts at line ${String(nextStart)} with the \`context.apply_compaction\` record that carries this note — it is already in your context; no need to read it.`, + ); + return renderPrompt(contextRecoveryTemplate, { + wire_path: pointer.journalPath, + window_lines: lines.map((line) => ` ${line}`).join('\n'), + }).trimEnd(); +} diff --git a/packages/agent-core-v2/src/agent/fullCompaction/fullCompactionService.ts b/packages/agent-core-v2/src/agent/fullCompaction/fullCompactionService.ts index 4bed2888b..4ec1d300d 100644 --- a/packages/agent-core-v2/src/agent/fullCompaction/fullCompactionService.ts +++ b/packages/agent-core-v2/src/agent/fullCompaction/fullCompactionService.ts @@ -3,7 +3,6 @@ import { Service } from "#/_base/di/service"; import { LifecycleScope } from '#/app/scopes'; import { ScopeActivation, registerScopedService } from '#/_base/di/scope'; import { defineState } from '#/state/state'; -import { renderPrompt } from "#/_base/utils/render-prompt"; import { estimateTokensForMessage } from "#/kosong/contract/tokens"; import { buildCompactionSummaryText, isRealUserInput } from '#/agent/contextMemory/compactionHandoff'; import { IAgentContextMemoryService } from '#/agent/contextMemory/contextMemory'; @@ -43,7 +42,11 @@ import { ITelemetryService } from '#/app/telemetry/telemetry'; import { ErrorCodes, Error2, isCodedError, isError2, toPythinkerErrorPayload, unwrapErrorCause } from "#/errors"; import { AgentErrorEvent } from '#/agent/mcp/mcpEvents'; import { IEventDispatcher } from '#/state/eventDispatcher'; -import compactionInstructionTemplate from './compaction-instruction.md?raw'; +import { onUnexpectedError } from '#/_base/errors/unexpectedError'; +import type { WireLineRange } from '#/wire/record'; +import { IWireService } from '#/wire/wire'; +import { renderCompactionInstruction } from './compactionInstruction'; +import { renderContextRecoveryPointer } from './contextRecovery'; import { IAgentFullCompactionService, type FullCompactionInput, @@ -58,6 +61,7 @@ import { CompactionCancelled, CompactionCompleted, fullCompactionKey, + fullCompactionWireRangesKey, FullCompactionBegin, FullCompactionCancel, FullCompactionComplete, @@ -152,10 +156,12 @@ export class AgentFullCompactionService extends Service implements IAgentFullCom @IEventBus private readonly eventBus: IEventBus, @IAgentLoopService private readonly loopService: IAgentLoopService, @IAgentStateService private readonly states: IAgentStateService, + @IWireService private readonly wire: IWireService, ) { super(); this.todo = manager.resolve(agent.agentContext, AgentTodo); this.states.contributeState(fullCompactionKey); + this.states.contributeState(fullCompactionWireRangesKey); this.states.contributeState(fullCompactionCompactionCountInTurnKey); this.states.contributeState(fullCompactionObservedMaxContextTokensByModelKey); this.states.contributeState(fullCompactionLastCompactedTokenCountKey); @@ -636,11 +642,7 @@ export class AgentFullCompactionService extends Service implements IAgentFullCom : undefined; const compactionMaxOutputSize = resolvedModel.maxOutputSize ?? defaultCompactionCap; - const customInstruction = data.instruction?.trim() ?? ''; - const instruction = renderPrompt(compactionInstructionTemplate, { - custom_instruction_block: - customInstruction.length > 0 ? `\nOptional user instruction:\n${customInstruction}\n` : '', - }).trimEnd(); + const instruction = renderCompactionInstruction({ customInstruction: data.instruction }); const delays = retryBackoffDelays(MAX_COMPACTION_RETRY_ATTEMPTS); let attempt: CompactionAttemptResult | undefined; @@ -691,6 +693,7 @@ export class AgentFullCompactionService extends Service implements IAgentFullCom overflowShrinkCount, (message) => this.tokenCounting.estimateMessage(message), ); + if (historyForModel.length === 0) throw error; droppedCount += before - historyForModel.length; retryCount = 0; continue; @@ -738,14 +741,23 @@ export class AgentFullCompactionService extends Service implements IAgentFullCom } const summary = await this.postProcessSummary(attempt.summary); + const wireLines = await this.captureWireLines(); + const recoveryFooter = this.renderRecoveryFooter(wireLines); + const summaryText = buildCompactionSummaryText(summary); const result = this.context.applyCompaction({ summary, - contextSummary: buildCompactionSummaryText(summary), + contextSummary: + recoveryFooter === undefined ? summaryText : `${summaryText}\n\n${recoveryFooter}`, compactedCount: originalHistory.length, tokensBefore, - summaryOutputTokens: attempt.usage?.output, + summaryOutputTokens: + attempt.usage === null || attempt.usage === undefined + ? undefined + : attempt.usage.output + + (recoveryFooter === undefined ? 0 : this.tokenCounting.estimateText(recoveryFooter)), requestOverheadTokens: this.requestTokens([]), droppedCount: droppedCount === 0 ? undefined : droppedCount, + wireLines, }); const properties: CompactionFinishedEvent = { @@ -789,6 +801,28 @@ export class AgentFullCompactionService extends Service implements IAgentFullCom } } + private async captureWireLines(): Promise { + try { + await this.wire.flush(); + } catch (error) { + onUnexpectedError(error); + return undefined; + } + const end = this.wire.lineCount(); + const previous = this.states.get(fullCompactionWireRangesKey).at(-1); + const start = Math.max(previous?.end ?? 0, this.wire.lastContextClearLine() ?? 0) + 1; + if (end < start) return undefined; + return { start, end }; + } + + private renderRecoveryFooter(wireLines: WireLineRange | undefined): string | undefined { + if (wireLines === undefined) return undefined; + const journalPath = this.wire.journalPath(); + if (journalPath === undefined) return undefined; + const windows = [...this.states.get(fullCompactionWireRangesKey), wireLines]; + return renderContextRecoveryPointer({ journalPath, windows }); + } + private async postProcessSummary(summary: string): Promise { const todos = this.todo.get(); if (todos.length === 0) { diff --git a/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncation.ts b/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncation.ts index 2bf2743a0..6aa87c155 100644 --- a/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncation.ts +++ b/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncation.ts @@ -17,6 +17,8 @@ export interface IAgentToolResultTruncationService { ): Promise; isSpillFilePath(path: string): boolean; + + isWireJournalPath(path: string): boolean; } export const IAgentToolResultTruncationService: ServiceIdentifier< diff --git a/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncationService.ts b/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncationService.ts index 6841df602..efdf8c897 100644 --- a/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncationService.ts +++ b/packages/agent-core-v2/src/agent/toolResultTruncation/toolResultTruncationService.ts @@ -8,9 +8,10 @@ import { type ExecutableToolResult, } from '#/tool/toolContract'; import { IBootstrapService } from '#/app/bootstrap/bootstrap'; +import { AGENT_WIRE_RECORD_KEY } from '#/wire/record'; import type { ContentPart } from '#/kosong/contract/message'; import { IFileSystemStorageService } from '#/persistence/interface/storage'; -import { join, normalize } from 'pathe'; +import { basename, join, normalize } from 'pathe'; import { IAgentToolResultTruncationService, type ToolResultTruncationInput, @@ -117,6 +118,12 @@ export class ToolResultTruncationService implements IAgentToolResultTruncationSe return normalized === dir || normalized.startsWith(`${dir}/`); } + isWireJournalPath(path: string): boolean { + const sessionsDir = normalize(join(this.bootstrap.homeDir, this.bootstrap.scope('sessions'))); + const normalized = normalize(path); + return normalized.startsWith(`${sessionsDir}/`) && basename(normalized) === AGENT_WIRE_RECORD_KEY; + } + private async saveToolResult( toolName: string, toolCallId: string, diff --git a/packages/agent-core-v2/src/agent/tools/os/read/read.md b/packages/agent-core-v2/src/agent/tools/os/read/read.md index 8597a6647..b331f90a6 100644 --- a/packages/agent-core-v2/src/agent/tools/os/read/read.md +++ b/packages/agent-core-v2/src/agent/tools/os/read/read.md @@ -7,6 +7,7 @@ When you need several files, prefer to read them in parallel: emit multiple `Rea - Relative paths resolve against the working directory; a path outside the working directory must be absolute. - Returns up to ${MAX_LINES} lines or ${MAX_BYTES_KB} KB per call, whichever comes first; lines longer than ${MAX_LINE_LENGTH} chars are truncated mid-line. - Page larger files with `line_offset` (1-based start line) and `n_lines`. Omit `n_lines` to read up to the ${MAX_LINES}-line cap. +- Pythinker Code agent event logs (`wire.jsonl` under the sessions directory) are returned with whole lines (up to ~150k chars per record); read them one record at a time with `n_lines=1` after locating the line with Grep. - Sensitive files (`.env` files, credential stores, SSH private keys, and similar secrets) are refused to protect secrets; do not attempt to read them. Templates and public keys are exempt: `.env.example` / `.env.sample` / `.env.template` and public SSH keys such as `id_rsa.pub` read normally. - UTF-8 text files are read directly. UTF-16 LE/BE text files (with or without a BOM) are detected automatically and transcoded to UTF-8 for display; the status block notes the detected encoding, and Edit/Write on such a file still expect UTF-8 — convert its encoding first (e.g. with `iconv`). Other encodings (e.g. GBK), binary files, and files containing NUL bytes are refused. - Negative line_offset reads from the end of the file (for example, -100 reads the last 100 lines); the absolute value cannot exceed ${MAX_LINES}. diff --git a/packages/agent-core-v2/src/agent/tools/os/read/read.ts b/packages/agent-core-v2/src/agent/tools/os/read/read.ts index 881b20182..66411c96d 100644 --- a/packages/agent-core-v2/src/agent/tools/os/read/read.ts +++ b/packages/agent-core-v2/src/agent/tools/os/read/read.ts @@ -6,6 +6,7 @@ import { type AgentTool } from '#/tool/toolContract'; export const MAX_LINES: number = 1000; export const MAX_LINE_LENGTH: number = 2000; export const MAX_BYTES: number = 100 * 1024; +export const EVENT_LOG_MAX_LINE_LENGTH: number = 150_000; export const TRANSCODE_MAX_BYTES: number = 10 * 1024 * 1024; diff --git a/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts b/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts index 95d8aa2e8..d9c5e1686 100644 --- a/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts +++ b/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts @@ -22,6 +22,7 @@ import { makeCarriageReturnsVisible, splitLinesKeepingTerminator, type LineEndin import { decodeUtfText, detectTextEncoding, type UtfTextEncoding } from '#/_base/text/encoding'; import { renderPrompt } from '#/_base/utils/render-prompt'; import { + EVENT_LOG_MAX_LINE_LENGTH, IReadTool, MAX_BYTES, MAX_LINE_LENGTH, @@ -59,6 +60,11 @@ interface FinishReadResultInput { readonly totalLines: number; readonly requestedLines: number; readonly detectedEncoding?: UtfTextEncoding; + readonly eventLog: boolean; +} + +function lineLengthLimit(eventLog: boolean): number { + return eventLog ? EVENT_LOG_MAX_LINE_LENGTH : MAX_LINE_LENGTH; } function truncateLine(line: string, maxLength: number): string { @@ -94,12 +100,16 @@ function lineEndingStyleFromFlags(flags: LineEndingFlags): LineEndingStyle { return 'lf'; } -function renderLine(entry: ReadLineEntry, lineEndingStyle: LineEndingStyle): RenderedLine { +function renderLine( + entry: ReadLineEntry, + lineEndingStyle: LineEndingStyle, + maxLineLength: number, +): RenderedLine { const modelContent = lineEndingStyle === 'crlf' && entry.rawContent.endsWith('\r') ? entry.rawContent.slice(0, -1) : entry.rawContent; - const truncated = truncateLine(modelContent, MAX_LINE_LENGTH); + const truncated = truncateLine(modelContent, maxLineLength); const renderedContent = lineEndingStyle === 'mixed' ? makeCarriageReturnsVisible(truncated) : truncated; return { @@ -115,6 +125,7 @@ function renderedLineBytes(renderedLine: string, isFirst: boolean): number { function renderEntries( entries: readonly ReadLineEntry[], lineEndingStyle: LineEndingStyle, + maxLineLength: number, ): { renderedLines: string[]; truncatedLineNumbers: number[]; @@ -126,7 +137,7 @@ function renderEntries( let maxBytesReached = false; for (const entry of entries) { - const rendered = renderLine(entry, lineEndingStyle); + const rendered = renderLine(entry, lineEndingStyle, maxLineLength); const lineBytes = renderedLineBytes(rendered.line, renderedLines.length === 0); if (renderedLines.length > 0 && bytes + lineBytes > MAX_BYTES) { maxBytesReached = true; @@ -248,8 +259,9 @@ export class ReadTool implements IReadTool { } const denied = await sensitiveTargetError(lease.runtime.fs!, args.path, path); if (denied !== undefined) return { isError: true, output: denied }; - const result = await this.execution(lease.runtime.fs!, args, path); - return this.resultTruncation.isSpillFilePath(path) + const eventLog = this.resultTruncation.isWireJournalPath(path); + const result = await this.execution(lease.runtime.fs!, args, path, eventLog); + return eventLog || this.resultTruncation.isSpillFilePath(path) ? { ...result, spillExempt: true as const } : result; } finally { @@ -259,7 +271,12 @@ export class ReadTool implements IReadTool { }; } - private async execution(fs: IHostFileSystem, args: ReadInput, safePath: string): Promise { + private async execution( + fs: IHostFileSystem, + args: ReadInput, + safePath: string, + eventLog: boolean, + ): Promise { try { let stat: Awaited>; try { @@ -319,6 +336,7 @@ export class ReadTool implements IReadTool { lineOffset, effectiveLimit, requestedLines, + eventLog, detectedEncoding, ); } @@ -328,6 +346,7 @@ export class ReadTool implements IReadTool { lineOffset, effectiveLimit, requestedLines, + eventLog, detectedEncoding, ); } catch (error) { @@ -347,6 +366,7 @@ export class ReadTool implements IReadTool { lineOffset: number, effectiveLimit: number, requestedLines: number, + eventLog: boolean, detectedEncoding?: UtfTextEncoding, ): Promise { const selectedEntries: ReadLineEntry[] = []; @@ -385,7 +405,7 @@ export class ReadTool implements IReadTool { } const lineEndingStyle = lineEndingStyleFromFlags(flags); - const rendered = renderEntries(selectedEntries, lineEndingStyle); + const rendered = renderEntries(selectedEntries, lineEndingStyle, lineLengthLimit(eventLog)); return this.finishReadResult({ renderedLines: rendered.renderedLines, @@ -397,6 +417,7 @@ export class ReadTool implements IReadTool { totalLines: currentLineNo, requestedLines, detectedEncoding, + eventLog, }); } @@ -406,6 +427,7 @@ export class ReadTool implements IReadTool { lineOffset: number, effectiveLimit: number, requestedLines: number, + eventLog: boolean, detectedEncoding?: UtfTextEncoding, ): Promise { const tailCount = Math.abs(lineOffset); @@ -434,6 +456,7 @@ export class ReadTool implements IReadTool { effectiveLimit, totalLines: currentLineNo, requestedLines, + eventLog, detectedEncoding, }); } @@ -444,11 +467,13 @@ export class ReadTool implements IReadTool { effectiveLimit: number; totalLines: number; requestedLines: number; + eventLog: boolean; detectedEncoding?: UtfTextEncoding; }): ExecutableToolResult { const lineEndingStyle = lineEndingStyleFromFlags(input.lineEndingFlags); + const maxLineLength = lineLengthLimit(input.eventLog); let renderedCandidates = input.entries.slice(0, input.effectiveLimit).map((entry) => { - return { entry, rendered: renderLine(entry, lineEndingStyle) }; + return { entry, rendered: renderLine(entry, lineEndingStyle, maxLineLength) }; }); let totalBytes = 0; @@ -465,7 +490,7 @@ export class ReadTool implements IReadTool { const candidate = renderedCandidates[i]; if (candidate === undefined) continue; const lineBytes = renderedLineBytes(candidate.rendered.line, kept.length === 0); - if (bytes + lineBytes > MAX_BYTES) break; + if (kept.length > 0 && bytes + lineBytes > MAX_BYTES) break; kept.unshift(candidate); bytes += lineBytes; } @@ -491,6 +516,7 @@ export class ReadTool implements IReadTool { totalLines: input.totalLines, requestedLines: input.requestedLines, detectedEncoding: input.detectedEncoding, + eventLog: input.eventLog, }); } @@ -521,7 +547,12 @@ export class ReadTool implements IReadTool { } if (input.truncatedLineNumbers.length > 0) { parts.push( - `Lines [${input.truncatedLineNumbers.join(', ')}] were truncated to ${String(MAX_LINE_LENGTH)} characters; use Bash (e.g. cut or sed) to read the elided content of those lines.`, + `Lines [${input.truncatedLineNumbers.join(', ')}] were truncated to ${String(lineLengthLimit(input.eventLog))} characters; use Bash (e.g. cut or sed) to read the elided content of those lines.`, + ); + } + if (input.eventLog) { + parts.push( + `Agent event log: records are returned whole up to ${String(EVENT_LOG_MAX_LINE_LENGTH)} characters per line; read one record at a time (n_lines=1). For a longer record, extract fields with Bash: sed -n 'Np' | jq. A primer on this format appears in your compaction note once a compaction has run.`, ); } if (input.lineEndingStyle === 'mixed') { diff --git a/packages/agent-core-v2/src/wire/record.ts b/packages/agent-core-v2/src/wire/record.ts index 758f68bb1..f5fbd16d7 100644 --- a/packages/agent-core-v2/src/wire/record.ts +++ b/packages/agent-core-v2/src/wire/record.ts @@ -9,6 +9,11 @@ export type RecordDehydrator = ( transform: PartsTransformer, ) => WireRecord | Promise; +export interface WireLineRange { + readonly start: number; + readonly end: number; +} + export interface WireRecord { readonly type: string; readonly time?: number; diff --git a/packages/agent-core-v2/src/wire/wire.ts b/packages/agent-core-v2/src/wire/wire.ts index 13d052786..215ad8b75 100644 --- a/packages/agent-core-v2/src/wire/wire.ts +++ b/packages/agent-core-v2/src/wire/wire.ts @@ -9,6 +9,9 @@ export interface IWireService { appendRecord(record: WireRecord, dehydrate?: RecordDehydrator): void; readJournal(): AsyncIterable; flush(): Promise; + lineCount(): number; + lastContextClearLine(): number | undefined; + journalPath(): string | undefined; } export const IWireService: ServiceIdentifier = diff --git a/packages/agent-core-v2/src/wire/wireService.ts b/packages/agent-core-v2/src/wire/wireService.ts index 5ae82baf4..b01528557 100644 --- a/packages/agent-core-v2/src/wire/wireService.ts +++ b/packages/agent-core-v2/src/wire/wireService.ts @@ -38,6 +38,8 @@ export class WireService extends Service implements IWireService { declare readonly _serviceBrand: undefined; private readonly wireScope: string; + private lines = 0; + private lastClearLine: number | undefined; private readonly agentId: string; private persistQueue: Promise | undefined; private pendingRepair: @@ -111,16 +113,20 @@ export class WireService extends Service implements IWireService { let rewrittenRecords: WireRecord[] | undefined; let newerWireVersion = false; let recordIndex = 0; + let lineCount = 0; let hasRecords = false; let legacyPlanRevisionMigrated = false; for await (const candidate of source) { + lineCount++; + this.lines = lineCount; const sourceRecord: unknown = candidate; if (!isWireRecord(sourceRecord)) { this.reportSkippedRecord(undefined, recordIndex, true); recordIndex++; continue; } + if (sourceRecord.type === 'context.clear') this.lastClearLine = lineCount; if (!hasRecords) { hasRecords = true; if (sourceRecord.type !== 'metadata') { @@ -179,9 +185,23 @@ export class WireService extends Service implements IWireService { await this.repairJournal(truncation, rewrittenRecords); } else if (rewrittenRecords !== undefined) { await this.log.rewrite(this.wireScope, AGENT_WIRE_RECORD_KEY, rewrittenRecords); + this.lines = rewrittenRecords.length; + this.lastClearLine = lastContextClearLineOf(rewrittenRecords); } } + lineCount(): number { + return this.lines; + } + + lastContextClearLine(): number | undefined { + return this.lastClearLine; + } + + journalPath(): string | undefined { + return this.storage.pathFor(this.wireScope, AGENT_WIRE_RECORD_KEY); + } + private async repairJournal( truncation: AppendLogTruncation, rewrittenRecords: WireRecord[] | undefined, @@ -210,6 +230,10 @@ export class WireService extends Service implements IWireService { truncation, ); this.pendingRepair = outcome === 'failed' ? { records, truncation } : undefined; + if (outcome !== 'failed') { + this.lines = records.length; + this.lastClearLine = lastContextClearLineOf(records); + } } private async repairPendingJournal(): Promise { @@ -320,7 +344,16 @@ export class WireService extends Service implements IWireService { this.log.append(this.wireScope, AGENT_WIRE_RECORD_KEY, record, { onError: onUnexpectedError, }); + this.lines += 1; + if (record.type === 'context.clear') this.lastClearLine = this.lines; + } +} + +function lastContextClearLineOf(records: readonly WireRecord[]): number | undefined { + for (let index = records.length - 1; index >= 0; index -= 1) { + if (records[index]!.type === 'context.clear') return index + 1; } + return undefined; } function extractLegacyPlanRevisionKey(path: string, agentId: string): string | undefined { diff --git a/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts b/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts index 24608b4c8..7280c1fb4 100644 --- a/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts +++ b/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts @@ -24,7 +24,12 @@ import { MASTER_ENV } from '#/app/flag/flagService'; import { estimateTokensForMessages } from '#/kosong/contract/tokens'; import { recordingTelemetry, type TelemetryRecord } from '../../app/telemetry/stubs'; import type { TestAgentContext, TestAgentOptions, TestAgentServiceOverride } from '../../harness'; -import { agentService, appServices, createCommandRunner, execEnvServices, hostEnvironmentServices, sessionServices, testAgent as createTestAgent } from '../../harness'; +import { agentService, appService, appServices, createCommandRunner, execEnvServices, hostEnvironmentServices, sessionServices, testAgent as createTestAgent } from '../../harness'; +import { IFileSystemStorageService } from '#/persistence/interface/storage'; +import { InMemoryStorageService } from '#/persistence/backends/memory/inMemoryStorageService'; +import { ISessionTokenCountingService } from '#/session/tokenCounting/sessionTokenCounting'; +import { renderCompactionInstruction } from '#/agent/fullCompaction/compactionInstruction'; +import { IWireService } from '#/wire/wire'; import { IAgentToolSelectAnnouncementsService } from '#/agent/toolSelect/toolSelectAnnouncements'; import { IAgentFullCompactionService, @@ -304,7 +309,7 @@ describe('FullCompaction', () => { compacted_count: 6, retry_count: 0, thinking_effort: 'off', - input_tokens: 1181, + input_tokens: 1247, output_tokens: 8, input_cache_read: 0, input_cache_creation: 0, @@ -2179,6 +2184,35 @@ describe('FullCompaction', () => { await ctx.expectResumeMatches(); }); + it('fails the compaction instead of compacting an empty history when overflow shrink drops everything', async () => { + let calls = 0; + const generate: GenerateFn = async () => { + calls += 1; + if (calls === 1) { + throw new APIContextOverflowError(400, 'Context length exceeded', 'req-shrink-empty'); + } + return textResult('Groundless summary.'); + }; + const ctx = testAgent({ generate }); + ctx.configure({ + provider: CATALOGUED_PROVIDER, + modelCapabilities: CATALOGUED_MODEL_CAPABILITIES, + }); + ctx.appendExchange(1, 'small user one', 'small assistant one', 20); + ctx.context.append({ + role: 'user', + content: [{ type: 'text', text: 'X'.repeat(400_000) }], + toolCalls: [], + }); + const failed = ctx.once('error'); + + await ctx.rpc.beginCompaction({}); + await failed; + + expect(calls).toBe(1); + expect(ctx.context.get()).toHaveLength(3); + }); + it('compacts and retries when the provider reports context overflow', async () => { let callCount = 0; const inputs: string[][] = []; @@ -3034,6 +3068,180 @@ describe('FullCompaction', () => { }); }); +describe('FullCompaction context recovery pointer', () => { + const JOURNAL_HOME = '/home/user/.pythinker'; + + interface ApplyCompactionArgs { + readonly summary?: string; + readonly contextSummary?: string; + readonly wireLines?: { readonly start: number; readonly end: number }; + } + + function locatedStorage(base: string): IFileSystemStorageService { + const memory = new InMemoryStorageService(); + return new Proxy(memory, { + get(target, property, receiver) { + if (property === 'pathFor') { + return (scope: string, key: string) => `${base}/${scope}/${key}`; + } + const value = Reflect.get(target, property, receiver) as unknown; + return typeof value === 'function' + ? (value as (...args: unknown[]) => unknown).bind(target) + : value; + }, + }) as unknown as IFileSystemStorageService; + } + + function recoveryAgent( + ...inputs: readonly (TestAgentServiceOverride | TestAgentOptions)[] + ): TestAgentContext { + const ctx = testAgent(...inputs); + ctx.configure({ + provider: CATALOGUED_PROVIDER, + modelCapabilities: CATALOGUED_MODEL_CAPABILITIES, + tools: SNAPSHOT_VISIBLE_TOOLS, + }); + return ctx; + } + + async function compactOnce(ctx: TestAgentContext, summary: string): Promise { + const completed = ctx.once('compaction.completed'); + ctx.mockNextResponse({ type: 'text', text: summary }); + await ctx.rpc.beginCompaction({}); + await completed; + } + + function noteText(ctx: TestAgentContext): string { + const part = ctx.context.get().at(-1)?.content[0]; + return part?.type === 'text' ? part.text : ''; + } + + function applyCompactionRecords(ctx: TestAgentContext): ApplyCompactionArgs[] { + return ctx.newEvents().flatMap((event) => { + if (event === null || typeof event !== 'object') return []; + const candidate = event as { type?: unknown; event?: unknown; args?: unknown }; + if (candidate.type !== '[wire]' || candidate.event !== 'context.apply_compaction') return []; + return [candidate.args as ApplyCompactionArgs]; + }); + } + + it('appends the journal location and window line ranges to the model-facing note', async () => { + const ctx = recoveryAgent(appService(IFileSystemStorageService, locatedStorage(JOURNAL_HOME))); + ctx.appendExchange(1, 'old user one', 'old assistant one', 20); + ctx.appendExchange(2, 'recent user two', 'recent assistant two', 40); + + await compactOnce(ctx, 'Compacted summary.'); + + const [record] = applyCompactionRecords(ctx); + expect(record?.wireLines).toEqual({ start: 1, end: expect.any(Number) }); + const end = record!.wireLines!.end; + expect(end).toBeGreaterThan(1); + const note = noteText(ctx); + expect(note).toContain('Compacted summary.'); + expect(note).toContain('## Context Recovery'); + expect(note).toContain(`${JOURNAL_HOME}/`); + expect(note).toContain('/wire.jsonl'); + expect(note).toContain(`window 1: lines 1–${String(end)} ← the conversation this note summarizes`); + expect(note).toContain(`window 2 (the one you are in now) starts at line ${String(end + 1)}`); + expect(note).toContain('context.append_loop_event'); + expect(record?.summary).not.toContain('Context Recovery'); + expect(record?.contextSummary).toContain('Context Recovery'); + await ctx.expectResumeMatches(); + }); + + it('lists every earlier window after repeated compactions', async () => { + const ctx = recoveryAgent(appService(IFileSystemStorageService, locatedStorage(JOURNAL_HOME))); + ctx.appendExchange(1, 'old user one', 'old assistant one', 20); + await compactOnce(ctx, 'First summary.'); + ctx.appendExchange(2, 'recent user two', 'recent assistant two', 40); + await compactOnce(ctx, 'Second summary.'); + + const [first, second] = applyCompactionRecords(ctx); + const firstLines = first!.wireLines!; + const secondLines = second!.wireLines!; + expect(secondLines.start).toBe(firstLines.end + 1); + expect(secondLines.end).toBeGreaterThan(secondLines.start); + const note = noteText(ctx); + expect(note).toContain(`window 1: lines 1–${String(firstLines.end)}\n`); + expect(note).not.toContain(`window 1: lines 1–${String(firstLines.end)} ←`); + expect(note).toContain( + `window 2: lines ${String(secondLines.start)}–${String(secondLines.end)} ← the conversation this note summarizes`, + ); + expect(note).toContain(`window 3 (the one you are in now) starts at line ${String(secondLines.end + 1)}`); + await ctx.expectResumeMatches(); + }); + + it('records window line ranges but omits the pointer when the journal has no on-disk path', async () => { + const ctx = recoveryAgent(); + ctx.appendExchange(1, 'old user one', 'old assistant one', 20); + ctx.appendExchange(2, 'recent user two', 'recent assistant two', 40); + + await compactOnce(ctx, 'Compacted summary.'); + + const [record] = applyCompactionRecords(ctx); + expect(record?.wireLines).toEqual({ start: 1, end: expect.any(Number) }); + expect(noteText(ctx)).not.toContain('Context Recovery'); + expect(record?.contextSummary).not.toContain('Context Recovery'); + }); + + it('starts the window after the latest context.clear record', async () => { + const ctx = recoveryAgent(appService(IFileSystemStorageService, locatedStorage(JOURNAL_HOME))); + ctx.appendExchange(1, 'discarded user one', 'discarded assistant one', 20); + ctx.context.clear(); + ctx.appendExchange(2, 'post-clear user two', 'post-clear assistant two', 40); + + await compactOnce(ctx, 'Post-clear summary.'); + + const wire = ctx.get(IWireService); + await wire.flush(); + let line = 0; + let clearLine = 0; + for await (const record of wire.readJournal()) { + line += 1; + if (record.type === 'context.clear') clearLine = line; + } + expect(clearLine).toBeGreaterThan(1); + const [record] = applyCompactionRecords(ctx); + expect(record?.wireLines?.start).toBe(clearLine + 1); + expect(record!.wireLines!.end).toBeGreaterThan(clearLine); + const note = noteText(ctx); + expect(note).toContain(`window 1: lines ${String(clearLine + 1)}–`); + expect(note).not.toContain('window 1: lines 1–'); + }); + + it('counts the appended recovery footer into the compacted token floor', async () => { + const withFooter = recoveryAgent( + appService(IFileSystemStorageService, locatedStorage(JOURNAL_HOME)), + ); + const bare = recoveryAgent(); + for (const ctx of [withFooter, bare]) { + ctx.appendExchange(1, 'old user one', 'old assistant one', 20); + ctx.appendExchange(2, 'recent user two', 'recent assistant two', 40); + await compactOnce(ctx, 'Compacted summary.'); + } + + const [footerRecord] = applyCompactionRecords(withFooter); + const contextSummary = footerRecord!.contextSummary!; + const footer = contextSummary.slice(contextSummary.indexOf('## Context Recovery')); + expect(footer.length).toBeGreaterThan(0); + const withFooterTokens = withFooter.tokenCounting.get().size; + const bareTokens = bare.tokenCounting.get().size; + expect(withFooterTokens - bareTokens).toBe( + withFooter.get(ISessionTokenCountingService).estimateText(footer), + ); + }); + + it('tells the summarizer a recovery pointer follows the note', () => { + const withPointer = renderCompactionInstruction({}); + const withCustom = renderCompactionInstruction({ customInstruction: ' keep the API facts ' }); + + expect(withPointer).toContain('a recovery pointer is appended below your note automatically'); + expect(withPointer).toContain('format for the final answer.\n\nThis conversation'); + expect(withPointer).not.toContain('${'); + expect(withCustom).toContain('Optional user instruction:\nkeep the API facts'); + }); +}); + afterEach(() => { vi.useRealTimers(); vi.unstubAllEnvs(); diff --git a/packages/agent-core-v2/test/agent/loop/stubs.ts b/packages/agent-core-v2/test/agent/loop/stubs.ts index b0ce4fb63..d12891c85 100644 --- a/packages/agent-core-v2/test/agent/loop/stubs.ts +++ b/packages/agent-core-v2/test/agent/loop/stubs.ts @@ -91,5 +91,5 @@ export async function runWillBeginStepHooks( signal: new AbortController().signal, }); } -export function stubWire(): IWireService { return { _serviceBrand: undefined, seal: async () => {}, appendRecord: () => {}, readJournal: async function* () {}, flush: async () => {} }; } +export function stubWire(): IWireService { return { _serviceBrand: undefined, seal: async () => {}, appendRecord: () => {}, readJournal: async function* () {}, flush: async () => {}, lineCount: () => 0, lastContextClearLine: () => undefined, journalPath: () => undefined }; } export function stubToolExecutor(): IAgentToolExecutorService { return { _serviceBrand: undefined, execute: async function* () {}, onBeforeExecuteTool: Event.None as Event, onWillExecuteTool: Event.None as Event, hooks: { onDidExecuteTool: new OrderedHookSlot() }, recordDupType: () => {}, registerToolCallGuard: () => ({ dispose() {} }), registerUnavailableToolDescriber: () => ({ dispose() {} }), registerMissingToolDescriber: () => ({ dispose() {} }) }; } diff --git a/packages/agent-core-v2/test/agent/task/taskService.test.ts b/packages/agent-core-v2/test/agent/task/taskService.test.ts index a26a7d0f2..091a47417 100644 --- a/packages/agent-core-v2/test/agent/task/taskService.test.ts +++ b/packages/agent-core-v2/test/agent/task/taskService.test.ts @@ -83,6 +83,9 @@ function stubWireService(): IWireService { appendRecord: () => {}, readJournal: async function* () {}, flush: async () => {}, + lineCount: () => 0, + lastContextClearLine: () => undefined, + journalPath: () => undefined, }; } diff --git a/packages/agent-core-v2/test/agent/toolExecutor/toolExecutor.test.ts b/packages/agent-core-v2/test/agent/toolExecutor/toolExecutor.test.ts index d65ef62aa..615589727 100644 --- a/packages/agent-core-v2/test/agent/toolExecutor/toolExecutor.test.ts +++ b/packages/agent-core-v2/test/agent/toolExecutor/toolExecutor.test.ts @@ -70,6 +70,7 @@ beforeEach(() => { _serviceBrand: undefined, truncateForModel: (input) => truncateForModel(input), isSpillFilePath: () => false, + isWireJournalPath: () => false, }); reg.defineInstance(IEventBus, { publish: (event: ProtocolEvent) => { diff --git a/packages/agent-core-v2/test/agent/toolResultTruncation/stubs.ts b/packages/agent-core-v2/test/agent/toolResultTruncation/stubs.ts index 60b972033..feed36260 100644 --- a/packages/agent-core-v2/test/agent/toolResultTruncation/stubs.ts +++ b/packages/agent-core-v2/test/agent/toolResultTruncation/stubs.ts @@ -9,6 +9,7 @@ export function stubToolResultTruncationService(): ToolResultTruncationServiceSt _serviceBrand: undefined, truncateForModel: async ({ result }) => result, isSpillFilePath: () => false, + isWireJournalPath: () => false, }; } diff --git a/packages/agent-core-v2/test/index.test.ts b/packages/agent-core-v2/test/index.test.ts index 6750afb20..f86dd8774 100644 --- a/packages/agent-core-v2/test/index.test.ts +++ b/packages/agent-core-v2/test/index.test.ts @@ -285,6 +285,7 @@ describe('conversation-time checkpoint registration', () => { const CHECKPOINT_EXEMPT_STATES: ReadonlySet = new Set([ 'goalForkNotice', 'turn', + 'fullCompaction.wireRanges', ]); const CONTEXT_OWNER_STATE = 'contextMemory'; const CONTEXT_EVENTS: readonly Event2Class[] = [ diff --git a/packages/agent-core-v2/test/os/backends/node-local/tools/read.test.ts b/packages/agent-core-v2/test/os/backends/node-local/tools/read.test.ts index 39056a2b9..3d814e9c5 100644 --- a/packages/agent-core-v2/test/os/backends/node-local/tools/read.test.ts +++ b/packages/agent-core-v2/test/os/backends/node-local/tools/read.test.ts @@ -6,6 +6,7 @@ import type { ISessionSkillCatalog } from '#/features/skill/session/skillCatalog import { stubWorkspaceContext } from '../../../../session/workspaceContext/stub-workspace-context'; import type { IHostFileSystem } from '#/os/interface/hostFileSystem'; import { + EVENT_LOG_MAX_LINE_LENGTH, MAX_BYTES, MAX_LINE_LENGTH, MAX_LINES, @@ -13,6 +14,7 @@ import { ReadInputSchema, TRANSCODE_MAX_BYTES, } from '#/agent/tools/os/read/read'; +import type { IAgentToolResultTruncationService } from '#/agent/toolResultTruncation/toolResultTruncation'; import { ReadTool } from '#/agent/tools/os/read/readTool'; import { stubToolResultTruncationService } from '../../../../agent/toolResultTruncation/stubs'; import type { IAgentRuntimeService } from '#/agent/runtimeBinding/agentRuntime'; @@ -234,6 +236,95 @@ describe('ReadTool', () => { }); }); + it('returns agent event log lines untruncated and marks the read spill-exempt', async () => { + const long = 'x'.repeat(MAX_LINE_LENGTH + 10); + const truncation: IAgentToolResultTruncationService = { + ...stubToolResultTruncationService(), + isWireJournalPath: (path) => path.endsWith('/wire.jsonl'), + }; + const tool = createReadTool( + createSpiedFs([long, 'short'].join('\n')).fs, + createTestEnv(), + PERMISSIVE_WORKSPACE, + { catalog: { getSkillRoots: () => [] } } as unknown as ISessionSkillCatalog, + truncation, + ); + + const result = await execute(tool, { + path: '/home/user/.pythinker/sessions/ws/session/agents/main/wire.jsonl', + }); + const output = toolContentString(result); + + expect(output).toContain(long); + expect(output).not.toContain('...'); + expect(result.note).not.toContain('were truncated'); + expect(result.note).toContain('Agent event log'); + expect(result.spillExempt).toBe(true); + }); + + it('caps a single event log record and reports the cap in the note', async () => { + const huge = 'y'.repeat(EVENT_LOG_MAX_LINE_LENGTH + 100); + const truncation: IAgentToolResultTruncationService = { + ...stubToolResultTruncationService(), + isWireJournalPath: (path) => path.endsWith('/wire.jsonl'), + }; + const tool = createReadTool( + createSpiedFs(`${huge}\nshort`).fs, + createTestEnv(), + PERMISSIVE_WORKSPACE, + { catalog: { getSkillRoots: () => [] } } as unknown as ISessionSkillCatalog, + truncation, + ); + + const result = await execute(tool, { + path: '/home/user/.pythinker/sessions/ws/session/agents/main/wire.jsonl', + line_offset: 1, + n_lines: 1, + }); + const output = toolContentString(result); + + expect(output.length).toBeLessThanOrEqual(EVENT_LOG_MAX_LINE_LENGTH + 20); + expect(output).toContain('...'); + expect(result.note).toContain(`truncated to ${String(EVENT_LOG_MAX_LINE_LENGTH)} characters`); + expect(result.spillExempt).toBe(true); + }); + + it('returns the last oversized event log record when reading from the tail', async () => { + const huge = 'z'.repeat(MAX_BYTES + 5_000); + const truncation: IAgentToolResultTruncationService = { + ...stubToolResultTruncationService(), + isWireJournalPath: (path) => path.endsWith('/wire.jsonl'), + }; + const tool = createReadTool( + createSpiedFs(`first\n${huge}`).fs, + createTestEnv(), + PERMISSIVE_WORKSPACE, + { catalog: { getSkillRoots: () => [] } } as unknown as ISessionSkillCatalog, + truncation, + ); + + const result = await execute(tool, { + path: '/home/user/.pythinker/sessions/ws/session/agents/main/wire.jsonl', + line_offset: -1, + }); + const output = toolContentString(result); + + expect(output).toContain('z'.repeat(1_000)); + expect(result.note).not.toContain('No lines read'); + }); + + it('keeps truncating long lines when the truncation service does not recognize the path', async () => { + const long = 'x'.repeat(MAX_LINE_LENGTH + 10); + const tool = toolWithContent([long, 'short'].join('\n')); + + const result = await execute(tool, { + path: '/home/user/.pythinker/sessions/ws/session/agents/main/wire.jsonl', + }); + + expect(result.note).toContain('Lines [1] were truncated to 2000 characters'); + expect(result.spillExempt).toBeUndefined(); + }); + it('denies a benign alias that resolves to a sensitive file', async () => { const { fs, readBytes } = createSpiedFs('SECRET=1\n'); (fs as { realpath?: (path: string) => Promise }).realpath = vi.fn(async (path: string) => diff --git a/packages/agent-core-v2/test/state/builtinReplayableKeys.ts b/packages/agent-core-v2/test/state/builtinReplayableKeys.ts index 14fff8689..f31db58ef 100644 --- a/packages/agent-core-v2/test/state/builtinReplayableKeys.ts +++ b/packages/agent-core-v2/test/state/builtinReplayableKeys.ts @@ -1,7 +1,7 @@ import type { ReplayableStateKey } from '#/state/state'; import { contextMemoryKey } from '#/agent/contextMemory/contextOps'; -import { fullCompactionKey } from '#/agent/fullCompaction/compactionOps'; +import { fullCompactionKey, fullCompactionWireRangesKey } from '#/agent/fullCompaction/compactionOps'; import { interruptionReminderKey } from '#/agent/interruptionReminder/interruptionReminderOps'; import { llmRequestTraceKey } from '#/agent/llmRequester/llmRequestOps'; import { turnKey } from '#/agent/loop/turnOps'; @@ -28,6 +28,7 @@ export const BUILTIN_REPLAYABLE_STATE_KEYS: readonly ReplayableStateKey[] = subagentBindingProvenanceKey, contextMemoryKey, fullCompactionKey, + fullCompactionWireRangesKey, interruptionReminderKey, llmRequestTraceKey, turnKey, diff --git a/packages/agent-core-v2/test/state/eventDispatcher.test.ts b/packages/agent-core-v2/test/state/eventDispatcher.test.ts index fb4044988..b00817d13 100644 --- a/packages/agent-core-v2/test/state/eventDispatcher.test.ts +++ b/packages/agent-core-v2/test/state/eventDispatcher.test.ts @@ -43,6 +43,9 @@ function stubWireJournal(journal: WireRecord[]): IWireService { for (const record of journal) yield record; }, flush: async () => {}, + lineCount: () => journal.length, + lastContextClearLine: () => undefined, + journalPath: () => undefined, }; } diff --git a/packages/agent-core-v2/test/wire/stubs.ts b/packages/agent-core-v2/test/wire/stubs.ts index c2c69cb1e..6c1932f94 100644 --- a/packages/agent-core-v2/test/wire/stubs.ts +++ b/packages/agent-core-v2/test/wire/stubs.ts @@ -215,6 +215,9 @@ export function stubAgentWire( appendRecord: () => {}, readJournal: async function* () {}, flush, + lineCount: () => 0, + lastContextClearLine: () => undefined, + journalPath: () => undefined, }; } From 0b87927d7fe66f967834dc33ed09e6023e020710 Mon Sep 17 00:00:00 2001 From: elkaix Date: Tue, 15 Sep 2026 18:51:02 -0400 Subject: [PATCH 03/11] feat: resume compacted conversations from the latest user message After compaction, append a short continue-work instruction as the last user message so the next turn keeps going instead of waiting on the summary. --- .changeset/compaction-resume-anchor.md | 5 ++ apps/vis/server/src/lib/context-projector.ts | 30 +++++++- .../server/test/lib/context-projector.test.ts | 44 +++++++---- apps/vis/server/test/routes/context.test.ts | 6 +- .../compaction-summary-prefix.md | 2 +- .../agent/contextMemory/compactionHandoff.ts | 29 ++++++- .../agent/contextMemory/contextTranscript.ts | 2 +- .../fullCompaction/compaction-instruction.md | 34 ++++----- .../test/agent/contextMemory/context.test.ts | 3 +- .../contextMemory/contextTranscript.test.ts | 11 +-- .../agent/contextMemory/splice-replay.test.ts | 20 ++++- .../fullCompaction/fullCompaction.test.ts | 76 ++++++++++++++----- .../agent/tokenCounting/tokenCounting.test.ts | 2 +- .../test/agent/undo/undo.test.ts | 3 +- .../agent-core-v2/test/harness/snapshots.ts | 2 +- 15 files changed, 197 insertions(+), 72 deletions(-) create mode 100644 .changeset/compaction-resume-anchor.md diff --git a/.changeset/compaction-resume-anchor.md b/.changeset/compaction-resume-anchor.md new file mode 100644 index 000000000..01220a82e --- /dev/null +++ b/.changeset/compaction-resume-anchor.md @@ -0,0 +1,5 @@ +--- +"@pymodel/pythinker-code": patch +--- + +After context compaction, keep a short continue-work instruction as the latest user message. diff --git a/apps/vis/server/src/lib/context-projector.ts b/apps/vis/server/src/lib/context-projector.ts index 1b9fa8d8b..3d233a269 100644 --- a/apps/vis/server/src/lib/context-projector.ts +++ b/apps/vis/server/src/lib/context-projector.ts @@ -1,6 +1,8 @@ import { COMPACT_USER_MESSAGE_MAX_TOKENS, + COMPACTION_CONTINUATION_VARIANT, COMPACTION_ELISION_VARIANT, + buildCompactionContinuationText, buildCompactionElisionText, collectCompactableUserMessages, isRealUserInput, @@ -346,7 +348,7 @@ export function projectContext( const original = realUserEntries[suffixStart + i]!; return original.message === message ? original : { ...original, message }; }); - messages = [...keptEntries, modelSummaryBubble]; + messages = [...keptEntries, modelSummaryBubble, continuationBubble(entry, rec)]; } else { // Head/tail record: mirror `selectCompactionUserMessages` and the // elision marker `ContextMemory.applyCompaction` inserts between the @@ -389,7 +391,13 @@ export function projectContext( } as ContextMessage, toolStepUuids: [], }; - messages = [...headEntries, markerBubble, ...tailEntries, modelSummaryBubble]; + messages = [ + ...headEntries, + markerBubble, + ...tailEntries, + modelSummaryBubble, + continuationBubble(entry, rec), + ]; } } else { // Full history: keep ALL preceding messages, just append the summary @@ -692,6 +700,24 @@ function estimateContentTokens(content: readonly ContentPart[]): number { return total; } +function continuationBubble( + entry: { readonly lineNo: number }, + rec: { readonly time?: number }, +): ProjectedMessage { + return { + lineNo: entry.lineNo + 0.25, + time: rec.time, + source: 'append_message', + message: { + role: 'user', + content: [{ type: 'text', text: buildCompactionContinuationText() }], + toolCalls: [], + origin: { kind: 'injection', variant: COMPACTION_CONTINUATION_VARIANT }, + } as ContextMessage, + toolStepUuids: [], + }; +} + /** True for messages that correspond to a real `_history` entry — * i.e. `append_message` and `compaction_summary` (the summary IS in `_history`). * The synthetic UI-only markers (`undo` / `clear`) are NOT in `_history`, so diff --git a/apps/vis/server/test/lib/context-projector.test.ts b/apps/vis/server/test/lib/context-projector.test.ts index b4b7c1f1e..032e1d02f 100644 --- a/apps/vis/server/test/lib/context-projector.test.ts +++ b/apps/vis/server/test/lib/context-projector.test.ts @@ -1,6 +1,7 @@ // apps/vis/server/test/lib/context-projector.test.ts import { describe, it, expect, afterEach } from 'vitest'; import { estimateTokensForMessages } from '@pymodel/agent-core-v2/kosong/contract/tokens'; +import { buildCompactionContinuationText } from '@pymodel/agent-core-v2/agent/contextMemory/compactionHandoff'; import { buildSessionFixture } from '../fixtures/build'; import { projectContext } from '../../src/lib/context-projector'; import { readAgentWire } from '../../src/lib/wire-reader'; @@ -327,16 +328,23 @@ describe('context-projector', () => { keptUserMessageCount: 2 }, raw: {} }, ]; const proj = projectContext(entries as any); - // [m0, m1, summary] — real user prompts are kept verbatim, the assistant - // tail is dropped. - expect(proj.messages).toHaveLength(3); + // [m0, m1, summary, anchor] — real user prompts are kept verbatim, the + // assistant tail is dropped, and the continuation anchor follows the summary. + expect(proj.messages).toHaveLength(4); expect(proj.messages.map((m) => m.source)).toEqual([ - 'append_message', 'append_message', 'compaction_summary', + 'append_message', 'append_message', 'compaction_summary', 'append_message', ]); expect(proj.messages[0]!.message.content[0]).toMatchObject({ text: 'm0' }); expect(proj.messages[1]!.message.content[0]).toMatchObject({ text: 'm1' }); expect(proj.messages[2]!.compaction).toEqual({ compactedCount: 3, tokensBefore: 100, tokensAfter: 10 }); expect(proj.messages[2]!.message.content[0]).toMatchObject({ text: 'sum' }); + expect(proj.messages[3]!.message.origin).toEqual({ + kind: 'injection', + variant: 'compaction_continuation', + }); + expect(proj.messages[3]!.message.content[0]).toMatchObject({ + text: buildCompactionContinuationText(), + }); }); it('apply_compaction mirrors the legacy verbatim tail for records without keptUserMessageCount (model)', () => { @@ -386,9 +394,10 @@ describe('context-projector', () => { ]; const proj = projectContext(entries as any); - // [FIRST, head slice of middle, marker, tail slice of middle, LAST, summary] - // — mirrors the engine's selectCompactionUserMessages + elision marker. - expect(proj.messages).toHaveLength(6); + // [FIRST, head slice of middle, marker, tail slice of middle, LAST, summary, anchor] + // — mirrors the engine's selectCompactionUserMessages + elision marker, with + // the continuation anchor after the summary. + expect(proj.messages).toHaveLength(7); const texts = proj.messages.map((m) => m.message.content.map((p: any) => (p.type === 'text' ? p.text : '')).join(''), ); @@ -404,9 +413,14 @@ describe('context-projector', () => { expect(middle.endsWith(texts[3]!)).toBe(true); expect(texts[4]).toBe(last); expect(proj.messages[5]!.source).toBe('compaction_summary'); + expect(proj.messages[6]!.message.origin).toEqual({ + kind: 'injection', + variant: 'compaction_continuation', + }); + expect(texts[6]).toBe(buildCompactionContinuationText()); // Synthesized entries (the head slice of the same message that anchors the // tail, and the marker) get fractional lineNos so keys stay unique. - expect(new Set(proj.messages.map((m) => m.lineNo)).size).toBe(6); + expect(new Set(proj.messages.map((m) => m.lineNo)).size).toBe(7); }); it('apply_compaction drops shell/local-command/background messages in model mode only', () => { @@ -430,10 +444,10 @@ describe('context-projector', () => { const model = projectContext(entries as any); expect(model.messages.map((m) => m.source)).toEqual([ - 'append_message', 'compaction_summary', 'append_message', + 'append_message', 'compaction_summary', 'append_message', 'append_message', ]); expect(model.messages.map((m) => m.message.content[0])).toMatchObject([ - { text: 'real user' }, { text: 'sum' }, { text: 'new' }, + { text: 'real user' }, { text: 'sum' }, { text: buildCompactionContinuationText() }, { text: 'new' }, ]); const full = projectContext(entries as any, 'full'); @@ -478,12 +492,14 @@ describe('context-projector', () => { keptUserMessageCount: 3 }, raw: {} }, ]; const proj = projectContext(entries as any); - // Correct: [u1, u3, u4, summary]. The marker is gone, all real prompts kept. + // Correct: [u1, u3, u4, summary, anchor]. The marker is gone, all real + // prompts kept, and the continuation anchor follows the summary. expect(proj.messages.map((m) => m.source)).toEqual([ - 'append_message', 'append_message', 'append_message', 'compaction_summary', + 'append_message', 'append_message', 'append_message', 'compaction_summary', 'append_message', ]); expect(proj.messages.map((m) => m.message.content[0])).toMatchObject([ { text: 'u1' }, { text: 'u3' }, { text: 'u4' }, { text: 'sum' }, + { text: buildCompactionContinuationText() }, ]); }); @@ -854,10 +870,10 @@ describe('context-projector', () => { keptUserMessageCount: 2 }, raw: {} }, ]; // No 2nd arg → 'model' default: the real user prompts are kept verbatim and - // the summary is appended after them. + // the summary is appended after them, followed by the continuation anchor. const proj = projectContext(entries as any); expect(proj.messages.map((m) => m.source)).toEqual([ - 'append_message', 'append_message', 'compaction_summary', + 'append_message', 'append_message', 'compaction_summary', 'append_message', ]); expect(proj.messages[0]!.message.content[0]).toMatchObject({ text: 'm0' }); expect(proj.messages[1]!.message.content[0]).toMatchObject({ text: 'm1' }); diff --git a/apps/vis/server/test/routes/context.test.ts b/apps/vis/server/test/routes/context.test.ts index b7bcd8614..a101194ed 100644 --- a/apps/vis/server/test/routes/context.test.ts +++ b/apps/vis/server/test/routes/context.test.ts @@ -1,4 +1,5 @@ import { describe, it, expect, afterEach } from 'vitest'; +import { buildCompactionContinuationText } from '@pymodel/agent-core-v2/agent/contextMemory/compactionHandoff'; import { buildSessionFixture } from '../fixtures/build'; import { contextRoute } from '../../src/routes/context'; @@ -77,10 +78,11 @@ describe('context route', () => { messages: { source: string; message: { content: { type: string; text?: string }[] } }[]; }; expect(modelBody.messages.map((m) => m.source)).toEqual([ - 'append_message', 'compaction_summary', 'append_message', + 'append_message', 'compaction_summary', 'append_message', 'append_message', ]); expect(modelBody.messages[0]!.message.content[0]).toMatchObject({ text: 'before compaction' }); - expect(modelBody.messages[2]!.message.content[0]).toMatchObject({ text: 'after compaction' }); + expect(modelBody.messages[2]!.message.content[0]).toMatchObject({ text: buildCompactionContinuationText() }); + expect(modelBody.messages[3]!.message.content[0]).toMatchObject({ text: 'after compaction' }); // Full history: every pre-compaction message (user prompt + assistant reply) // is KEPT, then the summary marker, then the post-compaction tail. diff --git a/packages/agent-core-v2/src/agent/contextMemory/compaction-summary-prefix.md b/packages/agent-core-v2/src/agent/contextMemory/compaction-summary-prefix.md index f814a9f84..3b8345bf3 100644 --- a/packages/agent-core-v2/src/agent/contextMemory/compaction-summary-prefix.md +++ b/packages/agent-core-v2/src/agent/contextMemory/compaction-summary-prefix.md @@ -1 +1 @@ -The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. +The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. The summary records which earlier requests were already addressed. diff --git a/packages/agent-core-v2/src/agent/contextMemory/compactionHandoff.ts b/packages/agent-core-v2/src/agent/contextMemory/compactionHandoff.ts index 11a6a9f73..4ae71ebbf 100644 --- a/packages/agent-core-v2/src/agent/contextMemory/compactionHandoff.ts +++ b/packages/agent-core-v2/src/agent/contextMemory/compactionHandoff.ts @@ -8,6 +8,7 @@ export const COMPACTION_SUMMARY_PREFIX = summaryPrefixTemplate.trimEnd(); export const COMPACT_USER_MESSAGE_MAX_TOKENS = 20_000; export const COMPACT_USER_MESSAGE_HEAD_TOKENS = 2_000; export const COMPACTION_ELISION_VARIANT = 'compaction_elision'; +export const COMPACTION_CONTINUATION_VARIANT = 'compaction_continuation'; type MessageLike = ContextMessage; @@ -94,11 +95,12 @@ export function buildContextCompactionShape( ? [...selection.head, ...selection.tail] : [...selection.head, elisionMessage, ...selection.tail]; const contextSummary = input.contextSummary ?? input.summary; + const continuationMessage = createCompactionContinuationMessage(); const tokensAfter = input.tokensAfter ?? (input.requestOverheadTokens ?? 0) + (input.summaryOutputTokens ?? estimate.text(contextSummary)) + - estimate.messages(keptMessages); + estimate.messages([...keptMessages, continuationMessage]); const keptUserMessageCount = input.keptUserMessageCount ?? selection.head.length + selection.tail.length; const keptHeadUserMessageCount = @@ -113,7 +115,11 @@ export function buildContextCompactionShape( keptUserMessageCount, keptHeadUserMessageCount, droppedCount: input.droppedCount, - messages: [...keptMessages, createCompactionSummaryMessage(contextSummary)], + messages: [ + ...keptMessages, + createCompactionSummaryMessage(contextSummary), + continuationMessage, + ], }; } @@ -146,6 +152,21 @@ export function buildCompactionElisionText(omittedTokens: number): string { ); } +export function createCompactionContinuationMessage(): ContextMessage { + return { + role: 'user', + content: [{ type: 'text', text: buildCompactionContinuationText() }], + toolCalls: [], + origin: { kind: 'injection', variant: COMPACTION_CONTINUATION_VARIANT }, + }; +} + +export function buildCompactionContinuationText(): string { + return wrapSystemReminder( + 'Context compaction is complete — continue the work that was in progress when it began.', + ); +} + export function collectCompactableUserMessages(messages: readonly T[]): T[] { return messages.filter( (message) => isRealUserInput(message) && !isCompactionSummaryMessage(message), @@ -313,9 +334,9 @@ function truncateTextToTokensFromEnd(text: string, maxTokens: number): string { let start = text.length; for (let i = text.length - 1; i >= 0; i--) { let isAscii = false; - const code = text.charCodeAt(i); + const code = text.codePointAt(i); if (code >= 0xdc00 && code <= 0xdfff && i > 0) { - const high = text.charCodeAt(i - 1); + const high = text.codePointAt(i - 1); if (high >= 0xd800 && high <= 0xdbff) { i--; } diff --git a/packages/agent-core-v2/src/agent/contextMemory/contextTranscript.ts b/packages/agent-core-v2/src/agent/contextMemory/contextTranscript.ts index aeb36bd1a..6eb4bdb1c 100644 --- a/packages/agent-core-v2/src/agent/contextMemory/contextTranscript.ts +++ b/packages/agent-core-v2/src/agent/contextMemory/contextTranscript.ts @@ -207,7 +207,7 @@ function recoverFoldedLength( const keptHeadUserMessageCount = readNumber(record, 'keptHeadUserMessageCount'); const compactedCount = readNumber(record, 'compactedCount'); if (keptUserMessageCount !== undefined) { - return keptUserMessageCount + (keptHeadUserMessageCount === undefined ? 1 : 2); + return keptUserMessageCount + (keptHeadUserMessageCount === undefined ? 2 : 3); } if (compactedCount !== undefined && compactedCount < foldedLength) { return 1 + (foldedLength - compactedCount); diff --git a/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md b/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md index fc30e61a3..90742b820 100644 --- a/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md +++ b/packages/agent-core-v2/src/agent/fullCompaction/compaction-instruction.md @@ -1,24 +1,20 @@ -You are about to run out of context. Write a first-person handoff note to -yourself so you can seamlessly continue this task after the earlier -conversation is cleared. +You are about to run out of context. Create a handoff summary for the +model that will resume this task after the earlier conversation is cleared. --- This message is a direct task, not part of the above conversation --- -Write the note as your own continuing train of thought — first person, present -tense, the way you would reason through the next move. Do not write a -third-party report about someone else's work, and do not impose rigid section -headings; let the shape follow the task. Write the note in the same language the -conversation has been using — do not switch to English just because these -instructions happen to be in English. +Do not impose rigid section headings; let the shape follow the task. Write it +in the same language the conversation has been using — do not switch to English +just because these instructions happen to be in English. -Make the note self-sufficient: the next turn will see only your most recent user -messages and this note — every assistant message, tool call, and tool result -above will be gone. In your own words, preserve what you genuinely need to -continue: +Make the summary self-sufficient: the next turn will see only the preserved +messages and this summary — every other assistant message, tool call, and tool +result above will be gone. In your own words, preserve what you genuinely need +to continue: - What the latest request is actually asking for: your reading of its intent and any ambiguity you have already resolved — not a re-transcription, since what - fits is kept verbatim in your most recent messages. But those kept messages are + fits is kept verbatim in the preserved messages. But those kept messages are size-capped, so a long request is truncated there: if the latest request is large (a big paste or file), preserve the parts at risk of being dropped — above all the actual ask. If several requests are in play, say which one governs @@ -52,9 +48,9 @@ continue: here is one less thing the next turn must rediscover. Include any required format for the final answer. -This conversation's event log stays on disk and a recovery pointer is appended below your note automatically, so you need not reproduce long outputs verbatim — keep exact identifiers, key values and error lines, and name anything the next turn should look up. +This conversation's event log stays on disk and a recovery pointer is appended below this summary automatically, so you need not reproduce long outputs verbatim — keep exact identifiers, key values and error lines, and name anything the next turn should look up. -Your TODO list is re-attached automatically below this note from its live +Your TODO list is re-attached automatically below this summary from its live source, so do not transcribe it — copying it wastes space and can contradict the live version. What that list cannot hold is the reasoning between tasks — why one was reordered or dropped, or a decision on one that constrains another — so @@ -65,9 +61,9 @@ was never verified (tests "passing", a fix "working", a file "created"), say so plainly and treat it as unverified rather than fact — re-check before relying on it. -Be concise, and keep the note proportional to the task: a long multi-step task -warrants detail, but a trivial or nearly finished exchange needs only a sentence -or two — do not pad it out. Include the critical data, identifiers, and +Be concise, and keep the summary proportional to the task: a long multi-step +task warrants detail, but a trivial or nearly finished exchange needs only a +sentence or two — do not pad it out. Include the critical data, identifiers, and references needed to continue, and omit anything that does not change the next move. diff --git a/packages/agent-core-v2/test/agent/contextMemory/context.test.ts b/packages/agent-core-v2/test/agent/contextMemory/context.test.ts index 9e5860081..3fe653472 100644 --- a/packages/agent-core-v2/test/agent/contextMemory/context.test.ts +++ b/packages/agent-core-v2/test/agent/contextMemory/context.test.ts @@ -787,8 +787,9 @@ describe('Agent context', () => { ); expect(shape.tokensAfter).toBe(0); - expect(shape.messages.map((m) => m.role)).toEqual(['user', 'user']); + expect(shape.messages.map((m) => m.role)).toEqual(['user', 'user', 'user']); expect(shape.messages[1]?.origin?.kind).toBe('compaction_summary'); + expect(shape.messages[2]?.origin).toEqual({ kind: 'injection', variant: 'compaction_continuation' }); }); it('prefers the measured summary output tokens over the text estimate', () => { diff --git a/packages/agent-core-v2/test/agent/contextMemory/contextTranscript.test.ts b/packages/agent-core-v2/test/agent/contextMemory/contextTranscript.test.ts index e9501d05a..21845e648 100644 --- a/packages/agent-core-v2/test/agent/contextMemory/contextTranscript.test.ts +++ b/packages/agent-core-v2/test/agent/contextMemory/contextTranscript.test.ts @@ -109,7 +109,7 @@ describe('reduceContextTranscript', () => { compaction('SUM', 3, 1), appendMessage(userMessage('u4')), ]); - expect(result.foldedLength).toBe(3); + expect(result.foldedLength).toBe(4); }); it('accounts for the elision marker when the record kept a head segment', () => { @@ -119,7 +119,7 @@ describe('reduceContextTranscript', () => { ...assistantStep('s1', 'a1'), compaction('SUM', 3, 2, 1), ]); - expect(result.foldedLength).toBe(4); + expect(result.foldedLength).toBe(5); }); it('carries the originating wire record time per entry', () => { @@ -160,7 +160,7 @@ describe('reduceContextTranscript', () => { ]); expect(texts(result)).toEqual(['message A', 'reply A', 'summary text']); expect(result.entries.map((m) => m.role)).toEqual(['user', 'assistant', 'user']); - expect(result.foldedLength).toBe(2); + expect(result.foldedLength).toBe(3); }); it('undo without compaction keeps the earlier exchange intact', () => { @@ -389,9 +389,10 @@ describe('live fold parity', () => { ]; const live = foldLive(records); const transcript = reduceContextTranscript(records); - expect(live).toHaveLength(5); + expect(live).toHaveLength(6); expect(transcript.foldedLength).toBe(live.length); expect(live[2]!.origin).toEqual({ kind: 'compaction_summary' }); + expect(live[3]!.origin).toEqual({ kind: 'injection', variant: 'compaction_continuation' }); }); it('settles a frame left open by a failed attempt when compaction lands mid-fold', () => { @@ -404,7 +405,7 @@ describe('live fold parity', () => { ]; const live = foldLive(records); const transcript = reduceContextTranscript(records); - expect(live.map((m) => m.role)).toEqual(['user', 'user', 'assistant']); + expect(live.map((m) => m.role)).toEqual(['user', 'user', 'user', 'assistant']); expect(texts(transcript)).toEqual(['u1', 'a1', 'SUM', 'a3']); expect(transcript.foldedLength).toBe(live.length); }); diff --git a/packages/agent-core-v2/test/agent/contextMemory/splice-replay.test.ts b/packages/agent-core-v2/test/agent/contextMemory/splice-replay.test.ts index 0d9b03672..3af862bf8 100644 --- a/packages/agent-core-v2/test/agent/contextMemory/splice-replay.test.ts +++ b/packages/agent-core-v2/test/agent/contextMemory/splice-replay.test.ts @@ -15,6 +15,7 @@ import { ContextUndo, } from '#/agent/contextMemory/contextEvents'; import { contextMemoryKey } from '#/agent/contextMemory/contextOps'; +import { buildCompactionContinuationText } from '#/agent/contextMemory/compactionHandoff'; import type { ContextMessage } from '#/agent/contextMemory/types'; import { ISessionTokenCountingService } from '#/session/tokenCounting/sessionTokenCounting'; import { IEventBus } from '#/app/event/eventBus'; @@ -376,11 +377,19 @@ describe('AgentContextMemoryService (wire-backed)', () => { ); const model = replay.agentState.get(contextMemoryKey); - expect(model.map((message) => message.role)).toEqual(['user', 'user', 'user']); - expect(model.map(textOf)).toEqual(['old user', 'recent user', 'model-facing summary']); + expect(model.map((message) => message.role)).toEqual(['user', 'user', 'user', 'user']); + expect(model.map(textOf)).toEqual([ + 'old user', + 'recent user', + 'model-facing summary', + buildCompactionContinuationText(), + ]); expect(model[2]).toMatchObject({ origin: { kind: 'compaction_summary' }, }); + expect(model[3]).toMatchObject({ + origin: { kind: 'injection', variant: 'compaction_continuation' }, + }); }); it('replays pre-contextSummary kept-user records without adding a new prefix', async () => { @@ -406,7 +415,12 @@ describe('AgentContextMemoryService (wire-backed)', () => { ); const model = replay.agentState.get(contextMemoryKey); - expect(model.map(textOf)).toEqual(['old user', 'recent user', 'OLD SUMMARY']); + expect(model.map(textOf)).toEqual([ + 'old user', + 'recent user', + 'OLD SUMMARY', + buildCompactionContinuationText(), + ]); expect(model[2]).toMatchObject({ role: 'user', origin: { kind: 'compaction_summary' }, diff --git a/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts b/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts index 7280c1fb4..7f8355cd2 100644 --- a/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts +++ b/packages/agent-core-v2/test/agent/fullCompaction/fullCompaction.test.ts @@ -17,7 +17,10 @@ import { afterEach, describe, expect, it, vi } from 'vitest'; import { DefaultCompactionStrategy, } from '#/agent/fullCompaction/strategy'; -import { COMPACTION_SUMMARY_PREFIX } from '#/agent/contextMemory/compactionHandoff'; +import { + COMPACTION_SUMMARY_PREFIX, + buildCompactionContinuationText, +} from '#/agent/contextMemory/compactionHandoff'; import { makeHookRunner } from '../../features/externalHooks/runner-stub'; import type { IExternalHooksRunnerService } from '#/features/externalHooks/app/externalHooksRunner'; import { MASTER_ENV } from '#/app/flag/flagService'; @@ -293,11 +296,16 @@ describe('FullCompaction', () => { role: 'user', text: expect.stringContaining('Compacted summary.'), }, + { role: 'user', text: buildCompactionContinuationText() }, ]); - expect(ctx.context.get().at(-1)?.content[0]).toMatchObject({ + expect(ctx.context.get().at(-2)?.content[0]).toMatchObject({ type: 'text', text: expect.stringContaining('The conversation so far has been compacted'), }); + expect(ctx.context.get().at(-1)).toMatchObject({ + role: 'user', + origin: { kind: 'injection', variant: 'compaction_continuation' }, + }); expect(records).toContainEqual({ event: 'compaction_finished', properties: expect.objectContaining({ @@ -309,7 +317,7 @@ describe('FullCompaction', () => { compacted_count: 6, retry_count: 0, thinking_effort: 'off', - input_tokens: 1247, + input_tokens: 1192, output_tokens: 8, input_cache_read: 0, input_cache_creation: 0, @@ -534,6 +542,7 @@ describe('FullCompaction', () => { role: 'user', text: expect.stringContaining('Recovered compacted summary.'), }, + { role: 'user', text: buildCompactionContinuationText() }, ]); await ctx.expectResumeMatches(); }); @@ -836,6 +845,7 @@ describe('FullCompaction', () => { { role: 'user', text: 'old user one' }, { role: 'user', text: 'recent user two' }, { role: 'user', text: `${COMPACTION_SUMMARY_PREFIX}\nRecovered compacted summary.` }, + { role: 'user', text: buildCompactionContinuationText() }, ]); expect( ctx.allEvents.filter((event) => event.event === 'compaction.completed'), @@ -888,6 +898,7 @@ describe('FullCompaction', () => { { role: 'user', text: 'old user one' }, { role: 'user', text: 'recent user two' }, { role: 'user', text: `${COMPACTION_SUMMARY_PREFIX}\nRecovered compacted summary.` }, + { role: 'user', text: buildCompactionContinuationText() }, ]); vi.useRealTimers(); await ctx.expectResumeMatches(); @@ -1421,6 +1432,7 @@ describe('FullCompaction', () => { 'user', 'user', 'user', + 'user', ]); await ctx.dispatch({ type: 'context.append_loop_event', @@ -1435,6 +1447,7 @@ describe('FullCompaction', () => { 'user', 'user', 'user', + 'user', ]); await ctx.expectResumeMatches(); }); @@ -1493,9 +1506,15 @@ describe('FullCompaction', () => { }, { "role": "user", - "text": "The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. + "text": "The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. The summary records which earlier requests were already addressed. Compacted prefix.", }, + { + "role": "user", + "text": " + Context compaction is complete — continue the work that was in progress when it began. + ", + }, ] `); await ctx.expectResumeMatches(); @@ -1723,14 +1742,15 @@ describe('FullCompaction', () => { call 2: messages: user: text "old user one\\n\\nold user two\\n\\nrecent user three\\n\\nAnswer after compacting" - user: text "The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary.\\nAuto compacted summary." + user: text "The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. The summary records which earlier requests were already addressed.\\nAuto compacted summary." + user: text "\\nContext compaction is complete — continue the work that was in progress when it began.\\n" `); expect(records).toContainEqual({ event: 'compaction_finished', properties: expect.objectContaining({ source: 'auto', tokens_before: 6_455, - tokens_after: 6_439, + tokens_after: 6_472, compacted_count: 7, retry_count: 0, }), @@ -1832,8 +1852,10 @@ describe('FullCompaction', () => { 'user', 'user', 'user', + 'user', ]); - expect(ctx.context.get().at(-1)?.origin).toEqual({ kind: 'compaction_summary' }); + expect(ctx.context.get().at(-2)?.origin).toEqual({ kind: 'compaction_summary' }); + expect(ctx.context.get().at(-1)?.origin).toEqual({ kind: 'injection', variant: 'compaction_continuation' }); await ctx.dispatch({ type: 'context.append_loop_event', @@ -1857,6 +1879,7 @@ describe('FullCompaction', () => { 'user', 'user', 'user', + 'user', ]); }); @@ -1900,8 +1923,10 @@ describe('FullCompaction', () => { 'user', 'user', 'user', + 'user', ]); - expect(ctx.context.get().at(-1)?.origin).toEqual({ kind: 'compaction_summary' }); + expect(ctx.context.get().at(-2)?.origin).toEqual({ kind: 'compaction_summary' }); + expect(ctx.context.get().at(-1)?.origin).toEqual({ kind: 'injection', variant: 'compaction_continuation' }); await ctx.dispatch({ type: 'context.append_loop_event', @@ -1916,6 +1941,7 @@ describe('FullCompaction', () => { 'user', 'user', 'user', + 'user', ]); }); @@ -1941,6 +1967,7 @@ describe('FullCompaction', () => { role: 'user', text: `${COMPACTION_SUMMARY_PREFIX}\nSingle message summary.`, }, + { role: 'user', text: buildCompactionContinuationText() }, ]); await ctx.expectResumeMatches(); }); @@ -1976,6 +2003,7 @@ describe('FullCompaction', () => { role: 'user', text: expect.stringContaining('Compacted after single-message compact.'), }, + { role: 'user', text: buildCompactionContinuationText() }, ]); await ctx.expectResumeMatches(); }); @@ -2104,7 +2132,7 @@ describe('FullCompaction', () => { expect(ctx.llmCalls).toHaveLength(2); const [compactionCall, answerCall] = ctx.llmCalls; - expect(messageText(compactionCall?.history.at(-1))).toContain('first-person handoff note'); + expect(messageText(compactionCall?.history.at(-1))).toContain('You are about to run out of context.'); expect( answerCall?.history.map(messageText).some((text) => text.includes('Reserved compacted summary.')), ).toBe(true); @@ -2286,8 +2314,11 @@ describe('FullCompaction', () => { "user: old user one Retry after provider overflow", - "user: The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. + "user: The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. The summary records which earlier requests were already addressed. Overflow compacted summary.", + "user: + Context compaction is complete — continue the work that was in progress when it began. + ", ], ] `); @@ -3011,8 +3042,11 @@ describe('FullCompaction', () => { "user: old user one xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx", - "user: The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. + "user: The conversation so far has been compacted to free up context. What follows is your own working summary of this task — use it to continue your train of thought rather than starting over. Treat it as notes, not proof: where it says a step was done, tests passed, or a fix worked, verify that yourself before relying on it. Any user messages earlier in this context are preserved verbatim from the compacted conversation; where a system-reminder note among them marks an omitted middle section, the user messages it replaced are covered by this summary. The summary records which earlier requests were already addressed. Placeholder compacted summary.", + "user: + Context compaction is complete — continue the work that was in progress when it began. + ", ], ] `); @@ -3045,7 +3079,7 @@ describe('FullCompaction', () => { await completed; const history = ctx.compactHistory(); - expect(history).toHaveLength(3); + expect(history).toHaveLength(4); expect(history[0]).toMatchObject({ role: 'user', text: 'old user one', @@ -3060,10 +3094,18 @@ describe('FullCompaction', () => { 'Compacted summary.\n\n## TODO List\n [in_progress] Fix the auth bug\n [pending] Add tests', ), }); - expect(ctx.context.get().at(-1)?.content[0]).toMatchObject({ + expect(history[3]).toMatchObject({ + role: 'user', + text: buildCompactionContinuationText(), + }); + expect(ctx.context.get().at(-2)?.content[0]).toMatchObject({ type: 'text', text: expect.stringContaining('The conversation so far has been compacted'), }); + expect(ctx.context.get().at(-1)).toMatchObject({ + role: 'user', + origin: { kind: 'injection', variant: 'compaction_continuation' }, + }); await ctx.expectResumeMatches(); }); }); @@ -3112,7 +3154,7 @@ describe('FullCompaction context recovery pointer', () => { } function noteText(ctx: TestAgentContext): string { - const part = ctx.context.get().at(-1)?.content[0]; + const part = ctx.context.get().at(-2)?.content[0]; return part?.type === 'text' ? part.text : ''; } @@ -3231,11 +3273,11 @@ describe('FullCompaction context recovery pointer', () => { ); }); - it('tells the summarizer a recovery pointer follows the note', () => { + it('tells the summarizer a recovery pointer follows the summary', () => { const withPointer = renderCompactionInstruction({}); const withCustom = renderCompactionInstruction({ customInstruction: ' keep the API facts ' }); - expect(withPointer).toContain('a recovery pointer is appended below your note automatically'); + expect(withPointer).toContain('a recovery pointer is appended below this summary automatically'); expect(withPointer).toContain('format for the final answer.\n\nThis conversation'); expect(withPointer).not.toContain('${'); expect(withCustom).toContain('Optional user instruction:\nkeep the API facts'); @@ -3480,7 +3522,7 @@ function inputHistorySnapshot(history: readonly Message[]): string[] { } function normalizeInputText(text: string): string { - return text.includes('first-person handoff note') ? '' : text; + return text.includes('You are about to run out of context.') ? '' : text; } describe('prompt deferral during full compaction', () => { diff --git a/packages/agent-core-v2/test/agent/tokenCounting/tokenCounting.test.ts b/packages/agent-core-v2/test/agent/tokenCounting/tokenCounting.test.ts index b6baeedef..241518c00 100644 --- a/packages/agent-core-v2/test/agent/tokenCounting/tokenCounting.test.ts +++ b/packages/agent-core-v2/test/agent/tokenCounting/tokenCounting.test.ts @@ -131,7 +131,7 @@ describe('Agent token counting', () => { }); const history = context.get(); - const kept = estimateTokensForMessages(history.filter((m) => m.origin?.kind === 'user')); + const kept = estimateTokensForMessages(history.filter((m) => m.origin?.kind !== 'compaction_summary')); const expected = 500 + kept; expect(tokenCountingState(ctx).anchors).toEqual([ { length: history.length, tokens: expected, measured: false }, diff --git a/packages/agent-core-v2/test/agent/undo/undo.test.ts b/packages/agent-core-v2/test/agent/undo/undo.test.ts index a5416fc0d..6e36b0e24 100644 --- a/packages/agent-core-v2/test/agent/undo/undo.test.ts +++ b/packages/agent-core-v2/test/agent/undo/undo.test.ts @@ -190,8 +190,9 @@ describe('AgentConversationUndoService', () => { await undo.undo(1); const history = ctx.context.get(); - expect(history.map((m) => m.role)).toEqual(['user', 'user']); + expect(history.map((m) => m.role)).toEqual(['user', 'user', 'user']); expect(history[1]?.origin?.kind).toBe('compaction_summary'); + expect(history[2]?.origin).toEqual({ kind: 'injection', variant: 'compaction_continuation' }); }); it('refuses loudly when a legacy compaction leaves anchors without checkpoints', async () => { diff --git a/packages/agent-core-v2/test/harness/snapshots.ts b/packages/agent-core-v2/test/harness/snapshots.ts index 6d5e12d75..315ae109d 100644 --- a/packages/agent-core-v2/test/harness/snapshots.ts +++ b/packages/agent-core-v2/test/harness/snapshots.ts @@ -237,7 +237,7 @@ function formatText(text: string): string { if (isDateReminder(text)) { return ''; } - if (text.includes('first-person handoff note')) { + if (text.includes('You are about to run out of context.')) { return ''; } return JSON.stringify(text); From f40321cf215ac672fff495fa1173a6740711525e Mon Sep 17 00:00:00 2001 From: elkaix Date: Tue, 15 Sep 2026 19:02:07 -0400 Subject: [PATCH 04/11] feat: page large file reads with a character budget Read returns complete lines within max_chars, with Next Read arguments and column offsets for long lines. UTF-16 stays readable after a lossy decode, and tail reads detect mid-scan file changes. --- .changeset/read-character-budgets.md | 5 + docs/configuration/config-files.md | 17 + docs/reference/tools.md | 6 +- .../agent-core-v2/docs/config-manifest.toml | 13 +- .../fullCompaction/context-recovery-footer.md | 4 +- .../src/agent/tools/os/read/configSection.ts | 14 + .../src/agent/tools/os/read/read.md | 15 +- .../src/agent/tools/os/read/read.ts | 18 +- .../src/agent/tools/os/read/readTool.ts | 560 ++++++------- .../test/agent/loop/loop.test.ts | 4 +- .../test/app/config/config.test.ts | 15 + .../os/backends/node-local/tools/read.test.ts | 753 +++++++++++++----- packages/agent-core-v2/test/tool/tool.test.ts | 10 +- 13 files changed, 939 insertions(+), 495 deletions(-) create mode 100644 .changeset/read-character-budgets.md create mode 100644 packages/agent-core-v2/src/agent/tools/os/read/configSection.ts diff --git a/.changeset/read-character-budgets.md b/.changeset/read-character-budgets.md new file mode 100644 index 000000000..e310380d8 --- /dev/null +++ b/.changeset/read-character-budgets.md @@ -0,0 +1,5 @@ +--- +"@pymodel/pythinker-code": minor +--- + +Read large files in pages with a character budget instead of a 1000-line or 100 KB cap. Set `[read] default_max_chars` and `[read] max_chars` in config.toml, or pass `max_chars` and `column_offset` on Read. diff --git a/docs/configuration/config-files.md b/docs/configuration/config-files.md index 7af29eb00..356847a06 100644 --- a/docs/configuration/config-files.md +++ b/docs/configuration/config-files.md @@ -454,6 +454,23 @@ disabled = ["EnterPlanMode", "ExitPlanMode", "mcp__github__*"] Like the `tools` / `disallowedTools` fields of an agent file, this section shapes the tools shown to the model and is enforced again before execution. [Permission rules](#permission) remain a separate control for operations that require approval. ::: +## `read` + +`read` controls the character limits for the [`Read` tool](../reference/tools.md). The limit includes file content, line numbers, and the status block; it does not impose a separate line-count or UTF-8 byte limit. + +| Field | Type | Default | Description | +| --- | --- | --- | --- | +| `default_max_chars` | `integer` | `100000` | Character budget when the tool call omits `max_chars` | +| `max_chars` | `integer` | `500000` | Maximum character budget a tool call may request | + +```toml +[read] +default_max_chars = 100000 +max_chars = 500000 +``` + +Both values must be positive integers. A call's `max_chars` overrides the default, but is capped at the configured maximum; the result reports the effective budget. If the configured default exceeds the maximum, the maximum also limits default reads. Raise `default_max_chars` when you want larger documents to be returned in one call without the agent requesting a larger budget. + ## `image` `image` controls how images are compressed before being sent to the model, across every ingestion point (pasted images, `ReadMediaFile` reads, images in MCP tool results, and so on). diff --git a/docs/reference/tools.md b/docs/reference/tools.md index 499c644b1..0bd5e2684 100644 --- a/docs/reference/tools.md +++ b/docs/reference/tools.md @@ -17,7 +17,11 @@ File tools handle reading, writing, and searching the local filesystem — the f | `Glob` | Auto-allow | Find files by glob pattern | | `ReadMediaFile` | Auto-allow | Read an image or video file | -**`Read`** accepts a file path (`path`) plus optional `line_offset` (starting line number; negative values count from the end) and `n_lines` (maximum number of lines to read). Returns at most 1000 lines or 100 KB per call; content beyond that limit is accompanied by a truncation notice. If the file is an image or video, the tool suggests using `ReadMediaFile` instead. +**`Read`** accepts a file path (`path`) plus optional `line_offset` (starting line number; negative values count from the end), `column_offset` (zero-based position within the first line of a forward read), `n_lines` (requested number of source lines), and `max_chars` (maximum characters in the result, including line numbers and status). Omitting `n_lines` reads toward the end of the file. The default is 100,000 characters, and calls can request up to 500,000; both values can be changed in the [`read` configuration](../configuration/config-files.md#read). Characters and column offsets use JavaScript string length in the displayed text, excluding the line-number prefix for column offsets: common letters count as one, while many emoji count as two. + +`Read` prefers complete lines and its results are not shortened again by the general tool-output limit. A line that cannot fit on its own page is returned in fragments; the status reports the column range and `Next Read` arguments to retrieve the rest without raising the budget. Join fragments of the same line without adding a newline. A partial line remains in the requested `n_lines` range until its ending is returned. Invalid column positions return an error rather than skipping content. + +Tail reads return the newest complete lines in the requested range first. If no complete line fits, the result includes forward `Next Read` arguments for the unread range; `column_offset` cannot be combined with a negative `line_offset`. Continuation positions refer to the current file contents, so start a new read if the file changes. If a tail read reports that the file changed during reading, retry against the updated file. UTF-16 LE/BE files up to 10 MiB are checked with strict decoding first. If decoding fails, `Read` returns readable text with malformed sequences replaced by U+FFFD, and every page warns that decoding was lossy and the text may differ from the original. The warning counts toward the character budget; a literal U+FFFD in a valid file does not trigger it. Use `ReadMediaFile` for images or videos. **`Write`** accepts `path`, `content`, and an optional `mode` (`overwrite` or `append`; defaults to overwrite). Missing parent directories are created automatically; `append` mode appends content to the end of the file without automatically adding a newline. Writing to an existing file — in either `overwrite` or `append` mode — requires a prior `Read` of that file in the session; the write is rejected if the file changed on disk since the last read, while creating a new file is exempt. diff --git a/packages/agent-core-v2/docs/config-manifest.toml b/packages/agent-core-v2/docs/config-manifest.toml index 5b12d8a77..44ee3f153 100644 --- a/packages/agent-core-v2/docs/config-manifest.toml +++ b/packages/agent-core-v2/docs/config-manifest.toml @@ -8,7 +8,7 @@ # commented "# field: type" lines describe the remaining schema fields. # Values resolve as: default -> config.toml -> env overlay -> memory. -# Index (30 sections · 2 overlay(s)) +# Index (31 sections · 2 overlay(s)) # advisor src/session/advisor/configSection.ts # background src/agent/task/configSection.ts # builtinProductSkills src/features/skill/catalog/configSection.ts @@ -32,6 +32,7 @@ # models src/app/kosongConfig/configSection.ts # permission src/agent/permissionRules/configSection.ts # providers src/app/kosongConfig/configSection.ts +# read src/agent/tools/os/read/configSection.ts # secondaryModel src/session/subagent/configSection.ts # services src/app/auth/configSection.ts # subagent src/session/subagent/configSection.ts @@ -385,6 +386,16 @@ merge_all_available_skills = true # env: record # source: record +# ########################################################################## +# read +# owner: src/agent/tools/os/read/configSection.ts +# scope: core +# ########################################################################## + +[read] +# default_max_chars: integer +# max_chars: integer + # ########################################################################## # secondaryModel (config.toml: secondary_model) # owner: src/session/subagent/configSection.ts diff --git a/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md b/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md index 5aba814ea..f3ba114f7 100644 --- a/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md +++ b/packages/agent-core-v2/src/agent/fullCompaction/context-recovery-footer.md @@ -6,5 +6,5 @@ If you need exact command output, file contents, error text, or the wording of a - Layout: one file per agent. agents/main/ is the main agent; each subagent has its own agents//wire.jsonl. A parent's log holds only the Agent tool call and the subagent's returned result — the subagent's own steps are in its own file. - Format: one JSON record per line, append-only; `type` says what it is. The conversation is in `context.append_message` (user prompts) and `context.append_loop_event` (event.type: step.begin | content.part [text|think] | tool.call | tool.result | step.end). Every other type (llm.request, usage.record, token_counting.measured, metadata, profile.bind, …) is bookkeeping — skip it. - Boundaries: `context.apply_compaction` marks a compaction (older lines stay in the file; grep for it to find exact boundaries). `context.undo` count=N retracts the previous N messages — treat retracted content as never having happened. `context.clear` resets the conversation. -- Externalized content: tool results over 50k chars are stored truncated, with an `output_path` to a tool-results/*.txt file holding the full text. Media parts are blob references, not inline. -- Reading: lines are long JSON (often 10k+ chars). Grep the file for a keyword to get line numbers, then Read exactly that line (line_offset=N, n_lines=1) — Read returns wire.jsonl lines whole up to ~150k chars. To pull one field with real newlines: sed -n 'Np' wire.jsonl | jq -r '.event.result.output'. Never Read large ranges — a handful of records can exceed the per-call byte cap. +- Externalized content: tool results over 50k chars may contain an `output_path` pointing to saved output; check the result's preservation notice before assuming the file contains everything. Read results are bounded by their own character budget and are not spilled again. Media parts are blob references, not inline. +- Reading: lines are long JSON (often 10k+ chars). Grep the file for a keyword to get line numbers, then Read exactly that line (line_offset=N, n_lines=1). Long records can span several Read results: follow Next Read with its column_offset until the line is complete, joining fragments without inserting newlines. To pull one field with real newlines when Bash is available: sed -n 'Np' wire.jsonl | jq -r '.event.result.output'. Prefer individual records over large ranges. diff --git a/packages/agent-core-v2/src/agent/tools/os/read/configSection.ts b/packages/agent-core-v2/src/agent/tools/os/read/configSection.ts new file mode 100644 index 000000000..d8431d26e --- /dev/null +++ b/packages/agent-core-v2/src/agent/tools/os/read/configSection.ts @@ -0,0 +1,14 @@ +import { z } from 'zod'; + +import { registerConfigSection } from '#/app/config/configSectionContributions'; + +export const READ_SECTION = 'read'; + +export const ReadConfigSchema = z.object({ + defaultMaxChars: z.number().int().positive().optional(), + maxChars: z.number().int().positive().optional(), +}); + +export type ReadConfig = z.infer; + +registerConfigSection(READ_SECTION, ReadConfigSchema, { defaultValue: {} }); diff --git a/packages/agent-core-v2/src/agent/tools/os/read/read.md b/packages/agent-core-v2/src/agent/tools/os/read/read.md index b331f90a6..247b98475 100644 --- a/packages/agent-core-v2/src/agent/tools/os/read/read.md +++ b/packages/agent-core-v2/src/agent/tools/os/read/read.md @@ -5,14 +5,17 @@ If the user provides a concrete file path to a text file, call Read directly. Do When you need several files, prefer to read them in parallel: emit multiple `Read` calls in a single response instead of reading one file per turn. - Relative paths resolve against the working directory; a path outside the working directory must be absolute. -- Returns up to ${MAX_LINES} lines or ${MAX_BYTES_KB} KB per call, whichever comes first; lines longer than ${MAX_LINE_LENGTH} chars are truncated mid-line. -- Page larger files with `line_offset` (1-based start line) and `n_lines`. Omit `n_lines` to read up to the ${MAX_LINES}-line cap. -- Pythinker Code agent event logs (`wire.jsonl` under the sessions directory) are returned with whole lines (up to ~150k chars per record); read them one record at a time with `n_lines=1` after locating the line with Grep. +- Returns text within `max_chars`, including line numbers and the status block, preferring complete lines. The configured default is ${DEFAULT_MAX_CHARS} characters; calls can request up to ${MAX_CHARS}. Characters use JavaScript string length, not UTF-8 bytes or tokens. Read results are not spilled or shortened again by the general tool-output limit. +- Omit `n_lines` to read toward the end of the file. There is no fixed line-count cap. When the task requires the full text of a large file, request a larger `max_chars`, up to ${MAX_CHARS}, in the first call. +- Page larger files with `line_offset` (1-based start line) and `n_lines`. If the result is incomplete, copy the `Next Read` arguments in the status block to continue without gaps or overlaps. Do not answer from a partial page when the task requires the remaining content. +- If a single line cannot fit on its own page, Read returns a fragment and reports its column range. Continue on the same line with the supplied `column_offset`; do not insert a newline between fragments of one source line. A partial line still counts toward the remaining `n_lines` until its ending is returned. +- `column_offset` is a zero-based position in the first line's displayed text, excluding its line-number prefix. It is supported only for forward reads. Offsets past the line or inside a Unicode surrogate pair return an error. Continuation refers to the current file contents; start a new read if the file changed. +- Pythinker Code agent event logs (`wire.jsonl` under the sessions directory) follow the same character budget; locate a record with Grep, read it with `n_lines=1`, and follow `Next Read` to retrieve every fragment of a long record. - Sensitive files (`.env` files, credential stores, SSH private keys, and similar secrets) are refused to protect secrets; do not attempt to read them. Templates and public keys are exempt: `.env.example` / `.env.sample` / `.env.template` and public SSH keys such as `id_rsa.pub` read normally. -- UTF-8 text files are read directly. UTF-16 LE/BE text files (with or without a BOM) are detected automatically and transcoded to UTF-8 for display; the status block notes the detected encoding, and Edit/Write on such a file still expect UTF-8 — convert its encoding first (e.g. with `iconv`). Other encodings (e.g. GBK), binary files, and files containing NUL bytes are refused. -- Negative line_offset reads from the end of the file (for example, -100 reads the last 100 lines); the absolute value cannot exceed ${MAX_LINES}. +- UTF-8 text files are read directly. UTF-16 LE/BE text files (with or without a BOM) are detected automatically and checked with strict decoding first. If malformed sequences are found, Read returns readable text with U+FFFD replacements and a lossy-decoding warning on every page; do not treat this view as exact original text. The status block notes the detected encoding, and Edit/Write on such a file still expect UTF-8 — convert its encoding first (e.g. with `iconv`). Other encodings (e.g. GBK), binary files, and files containing NUL bytes are refused. +- Negative `line_offset` reads from the end of the file (for example, -100 reads the last 100 lines). If the requested tail range exceeds the character budget, the newest complete lines in that range are returned first; `Next Read` covers the omitted earlier range. If no complete line fits, Read reports this and supplies forward `Next Read` arguments for the entire unread range. Omit `column_offset` when using a negative `line_offset`. - Output format: `\t` per line. -- A `...` status block is appended after the file content; it summarizes how much was read (line and byte counts, truncation, line-ending notes) and is not part of the file itself. +- A `...` status block is appended after the file content. It reports the actual returned range, total lines, effective character budget, whether the requested range is complete, and whether EOF was reached. The block is not part of the file itself. - Pure CRLF files are displayed with LF line endings; `Edit` matches this output and preserves CRLF when writing back. - Mixed or lone carriage-return line endings are shown as `\r` and require exact `Edit.old_string` escapes. - After a successful `Edit`/`Write`, do not re-read solely to prove the write landed. When the task depends on an exact file, API, or output shape, inspect the final external contract before finishing. diff --git a/packages/agent-core-v2/src/agent/tools/os/read/read.ts b/packages/agent-core-v2/src/agent/tools/os/read/read.ts index 66411c96d..05262569a 100644 --- a/packages/agent-core-v2/src/agent/tools/os/read/read.ts +++ b/packages/agent-core-v2/src/agent/tools/os/read/read.ts @@ -3,15 +3,13 @@ import { z } from 'zod'; import { createDecorator } from '#/_base/di/instantiation'; import { type AgentTool } from '#/tool/toolContract'; -export const MAX_LINES: number = 1000; -export const MAX_LINE_LENGTH: number = 2000; -export const MAX_BYTES: number = 100 * 1024; -export const EVENT_LOG_MAX_LINE_LENGTH: number = 150_000; +export const DEFAULT_MAX_CHARS = 100_000; +export const DEFAULT_MAX_CHARS_LIMIT = 500_000; export const TRANSCODE_MAX_BYTES: number = 10 * 1024 * 1024; const PositiveLineOffsetSchema = z.number().int().min(1); -const TailLineOffsetSchema = z.number().int().min(-MAX_LINES).max(-1); +const TailLineOffsetSchema = z.number().int().negative(); export const ReadInputSchema = z.object({ path: z @@ -23,16 +21,22 @@ export const ReadInputSchema = z.object({ .union([PositiveLineOffsetSchema, TailLineOffsetSchema]) .optional() .describe( - `The line number to start reading from. Omit to start at line 1. Negative values read from the end of the file; the absolute value cannot exceed ${String(MAX_LINES)}.`, + 'The line number to start reading from. Omit to start at line 1. Negative values read from the end of the file (for example, -100 reads the last 100 lines).', ), + column_offset: z.number().int().nonnegative().optional().describe( + 'Zero-based character offset within the first line of a forward read, excluding its line-number prefix. Uses JavaScript string length in the displayed text. Copy continuation arguments from the previous result to resume a long line.', + ), n_lines: z .number() .int() .positive() .optional() .describe( - `The number of lines to read; the tool also applies its internal cap. Omit to read up to the internal cap of ${String(MAX_LINES)} lines.`, + 'The number of lines to read. Omit to read toward the end of the file. Results are bounded by max_chars, with continuation arguments when the requested range is incomplete.', ), + max_chars: z.number().int().positive().optional().describe( + 'Maximum characters in the returned text, including line numbers and status. Omit for the configured default; requests above the configured maximum are capped.', + ), }); export const ReadOutputSchema = z.object({ diff --git a/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts b/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts index d9c5e1686..d59c9a22a 100644 --- a/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts +++ b/packages/agent-core-v2/src/agent/tools/os/read/readTool.ts @@ -4,6 +4,8 @@ import { RuntimeWorkspaceView } from '#/runtime/runtimeWorkspaceView'; import { unwrapErrorCause } from '#/_base/errors/errors'; import { ISessionSkillCatalog } from '#/features/skill/session/skillCatalog'; import { ISessionWorkspaceContext } from '#/session/workspaceContext/workspaceContext'; +import { IConfigService } from '#/app/config/config'; +import { renderToolResultForModel } from '#/agent/contextMemory/toolResultRender'; import { ToolAccesses, type ExecutableToolResult, @@ -19,19 +21,18 @@ import { MEDIA_SNIFF_BYTES, detectFileType } from '#/agent/media/file-type'; import { toInputJsonSchema } from '#/tool/input-schema'; import { literalRulePattern, matchesPathRuleSubject } from '#/tool/rule-match'; import { makeCarriageReturnsVisible, splitLinesKeepingTerminator, type LineEndingStyle } from '#/_base/text/line-endings'; -import { decodeUtfText, detectTextEncoding, type UtfTextEncoding } from '#/_base/text/encoding'; +import { detectTextEncoding, type UtfTextEncoding } from '#/_base/text/encoding'; import { renderPrompt } from '#/_base/utils/render-prompt'; import { - EVENT_LOG_MAX_LINE_LENGTH, + DEFAULT_MAX_CHARS, + DEFAULT_MAX_CHARS_LIMIT, IReadTool, - MAX_BYTES, - MAX_LINE_LENGTH, - MAX_LINES, ReadInputSchema, TRANSCODE_MAX_BYTES, type ReadInput, } from './read'; import { IAgentToolResultTruncationService } from '#/agent/toolResultTruncation/toolResultTruncation'; +import { READ_SECTION, type ReadConfig } from './configSection'; import readDescriptionTemplate from './read.md?raw'; interface LineEndingFlags { @@ -45,39 +46,40 @@ interface ReadLineEntry { readonly rawContent: string; } -interface RenderedLine { - readonly line: string; - readonly wasTruncated: boolean; +interface ReadTailEntry extends ReadLineEntry { + readonly minChars: number; } -interface FinishReadResultInput { - readonly renderedLines: readonly string[]; - readonly truncatedLineNumbers: readonly number[]; - readonly maxLinesReached: boolean; - readonly maxBytesReached: boolean; - readonly lineEndingStyle: LineEndingStyle; - readonly startLine: number; - readonly totalLines: number; - readonly requestedLines: number; +interface ReadRequest { + readonly args: ReadInput; + readonly maxChars: number; + readonly maxCharsLimit: number; readonly detectedEncoding?: UtfTextEncoding; + readonly lossyDecoding: boolean; readonly eventLog: boolean; } -function lineLengthLimit(eventLog: boolean): number { - return eventLog ? EVENT_LOG_MAX_LINE_LENGTH : MAX_LINE_LENGTH; -} - -function truncateLine(line: string, maxLength: number): string { - if (line.length <= maxLength) return line; - const marker = '...'; - const target = Math.max(maxLength, marker.length); - return line.slice(0, target - marker.length) + marker; +interface ReadPage { + readonly request: ReadRequest; + readonly renderedLines: readonly string[]; + readonly startLine: number; + readonly rangeStart: number; + readonly rangeEnd: number; + readonly totalLines: number; + readonly fromTail: boolean; + readonly lineEndingStyle: LineEndingStyle; } function stripTrailingLf(line: string): string { return line.endsWith('\n') ? line.slice(0, -1) : line; } +function splitsSurrogatePair(text: string, offset: number): boolean { + const previous = text.codePointAt(offset - 1); + const next = text.codePointAt(offset); + return previous >= 0xd800 && previous <= 0xdbff && next >= 0xdc00 && next <= 0xdfff; +} + function updateLineEndingFlags(flags: LineEndingFlags, text: string): void { for (let i = 0; i < text.length; i += 1) { const code = text.codePointAt(i); @@ -100,62 +102,14 @@ function lineEndingStyleFromFlags(flags: LineEndingFlags): LineEndingStyle { return 'lf'; } -function renderLine( - entry: ReadLineEntry, - lineEndingStyle: LineEndingStyle, - maxLineLength: number, -): RenderedLine { +function renderLine(entry: ReadLineEntry, lineEndingStyle: LineEndingStyle): string { const modelContent = lineEndingStyle === 'crlf' && entry.rawContent.endsWith('\r') ? entry.rawContent.slice(0, -1) : entry.rawContent; - const truncated = truncateLine(modelContent, maxLineLength); const renderedContent = - lineEndingStyle === 'mixed' ? makeCarriageReturnsVisible(truncated) : truncated; - return { - line: `${String(entry.lineNo)}\t${renderedContent}`, - wasTruncated: truncated !== modelContent, - }; -} - -function renderedLineBytes(renderedLine: string, isFirst: boolean): number { - return (isFirst ? 0 : 1) + Buffer.byteLength(renderedLine, 'utf8'); -} - -function renderEntries( - entries: readonly ReadLineEntry[], - lineEndingStyle: LineEndingStyle, - maxLineLength: number, -): { - renderedLines: string[]; - truncatedLineNumbers: number[]; - maxBytesReached: boolean; -} { - const renderedLines: string[] = []; - const truncatedLineNumbers: number[] = []; - let bytes = 0; - let maxBytesReached = false; - - for (const entry of entries) { - const rendered = renderLine(entry, lineEndingStyle, maxLineLength); - const lineBytes = renderedLineBytes(rendered.line, renderedLines.length === 0); - if (renderedLines.length > 0 && bytes + lineBytes > MAX_BYTES) { - maxBytesReached = true; - break; - } - - if (rendered.wasTruncated) { - truncatedLineNumbers.push(entry.lineNo); - } - renderedLines.push(rendered.line); - bytes += lineBytes; - if (bytes >= MAX_BYTES) { - maxBytesReached = true; - break; - } - } - - return { renderedLines, truncatedLineNumbers, maxBytesReached }; + lineEndingStyle === 'mixed' ? makeCarriageReturnsVisible(modelContent) : modelContent; + return `${String(entry.lineNo)}\t${renderedContent}`; } function isFileNotFoundError(error: unknown): boolean { @@ -205,29 +159,42 @@ function notUtf8DecodableFileOutput(path: string): string { ); } -const READ_DESCRIPTION = renderPrompt(readDescriptionTemplate, { - MAX_LINES, - MAX_BYTES_KB: MAX_BYTES / 1024, - MAX_LINE_LENGTH, -}); - export class ReadTool implements IReadTool { declare readonly _serviceBrand: undefined; readonly name = 'Read' as const; - readonly description = READ_DESCRIPTION; + get description(): string { + const limits = this.limits(); + return renderPrompt(readDescriptionTemplate, { + DEFAULT_MAX_CHARS: limits.defaultMaxChars, + MAX_CHARS: limits.maxChars, + }); + } readonly parameters: Record = toInputJsonSchema(ReadInputSchema); constructor( @IAgentRuntimeService private readonly runtime: IAgentRuntimeService, @ISessionWorkspaceContext private readonly workspaceCtx: ISessionWorkspaceContext, @ISessionSkillCatalog private readonly skillCatalog: ISessionSkillCatalog, @IAgentToolResultTruncationService private readonly resultTruncation: IAgentToolResultTruncationService, + @IConfigService private readonly config: IConfigService, ) {} + private limits(): { defaultMaxChars: number; maxChars: number } { + const section = this.config.get(READ_SECTION); + const maxChars = section?.maxChars ?? DEFAULT_MAX_CHARS_LIMIT; + return { + defaultMaxChars: Math.min(section?.defaultMaxChars ?? DEFAULT_MAX_CHARS, maxChars), + maxChars, + }; + } + private workspaceConfig(view: RuntimeWorkspaceView): WorkspaceConfig { return { workspaceDir: view.workDir, additionalDirs: view.additionalDirs }; } resolveExecution(args: ReadInput): ToolExecution { + if (args.column_offset !== undefined && (args.line_offset ?? 1) < 0) { + return { isError: true, output: 'column_offset is only supported for forward reads. Use a positive line_offset or the forward Next Read arguments.' }; + } const inspected = inspectAgentRuntime(this.runtime); const view = new RuntimeWorkspaceView(inspected, { workDir: this.workspaceCtx.workDir, @@ -261,9 +228,7 @@ export class ReadTool implements IReadTool { if (denied !== undefined) return { isError: true, output: denied }; const eventLog = this.resultTruncation.isWireJournalPath(path); const result = await this.execution(lease.runtime.fs!, args, path, eventLog); - return eventLog || this.resultTruncation.isSpillFilePath(path) - ? { ...result, spillExempt: true as const } - : result; + return { ...result, spillExempt: true }; } finally { lease.dispose(); } @@ -301,8 +266,9 @@ export class ReadTool implements IReadTool { } const detection = detectTextEncoding(header); - let lines: AsyncIterable; + let readLines: () => AsyncIterable; let detectedEncoding: UtfTextEncoding | undefined; + let lossyDecoding = false; if (!detection.seemsBinary && detection.encoding !== 'utf-8') { if (stat.size > TRANSCODE_MAX_BYTES) { return { @@ -313,42 +279,48 @@ export class ReadTool implements IReadTool { 'Convert it to UTF-8 first (e.g. with `iconv`).', }; } - const decoded = decodeUtfText(await fs.readBytes(safePath), detection.encoding); + const bytes = await fs.readBytes(safePath); + let decoded: string; + try { + decoded = new TextDecoder(detection.encoding, { fatal: true }).decode(bytes); + } catch (error) { + if (!isTextDecodeError(error)) throw error; + decoded = new TextDecoder(detection.encoding, { fatal: false }).decode(bytes); + lossyDecoding = true; + } detectedEncoding = detection.encoding; - lines = decodedLines(splitLinesKeepingTerminator(decoded)); + const decodedContent = splitLinesKeepingTerminator(decoded); + readLines = () => decodedLines(decodedContent); } else if (fileType.kind === 'unknown') { return { isError: true, output: notReadableFileOutput(args.path), }; } else { - lines = fs.readLines(safePath, { errors: 'strict', maxLineBytes: MAX_LINE_LENGTH * 4 }); + readLines = () => fs.readLines(safePath, { errors: 'strict' }); } + const limits = this.limits(); + const request: ReadRequest = { + args, + maxChars: Math.min(args.max_chars ?? limits.defaultMaxChars, limits.maxChars), + maxCharsLimit: limits.maxChars, + detectedEncoding, + lossyDecoding, + eventLog, + }; const lineOffset = args.line_offset ?? 1; - const requestedLines = args.n_lines ?? MAX_LINES; - const effectiveLimit = Math.min(requestedLines, MAX_LINES); - - if (lineOffset < 0) { - return await this.readTail( - args.path, - lines, - lineOffset, - effectiveLimit, - requestedLines, - eventLog, - detectedEncoding, - ); + if (lineOffset >= 0) return await this.readForward(readLines(), request); + const rereadsFile = detectedEncoding === undefined && (args.n_lines ?? Infinity) < -lineOffset; + const result = await this.readTail(readLines, request); + if (!result.isError && rereadsFile) { + const currentStat = await fs.stat(safePath); + if (!currentStat.isFile || currentStat.size !== stat.size || + currentStat.mtimeMs !== stat.mtimeMs || currentStat.ino !== stat.ino) { + return { isError: true, output: 'File changed while reading its tail. Retry Read with the updated file.' }; + } } - return await this.readForward( - args.path, - lines, - lineOffset, - effectiveLimit, - requestedLines, - eventLog, - detectedEncoding, - ); + return result; } catch (error) { if (isTextDecodeError(error)) { return { isError: true, output: notUtf8DecodableFileOutput(args.path) }; @@ -361,212 +333,250 @@ export class ReadTool implements IReadTool { } private async readForward( - displayPath: string, lines: AsyncIterable, - lineOffset: number, - effectiveLimit: number, - requestedLines: number, - eventLog: boolean, - detectedEncoding?: UtfTextEncoding, + request: ReadRequest, ): Promise { + const { args, maxChars } = request; + const lineOffset = args.line_offset ?? 1; + const columnOffset = args.column_offset ?? 0; + const requestedLines = args.n_lines ?? Infinity; const selectedEntries: ReadLineEntry[] = []; const flags: LineEndingFlags = { hasCrLf: false, hasLf: false, hasLoneCr: false }; let currentLineNo = 0; - let maxLinesReached = false; let collectionClosed = false; + let minimumChars = 0; for await (const rawLine of lines) { if (containsNulByte(rawLine)) { - return { isError: true, output: notReadableFileOutput(displayPath) }; + return { isError: true, output: notReadableFileOutput(args.path) }; } currentLineNo += 1; updateLineEndingFlags(flags, rawLine); - if (collectionClosed) { - if (effectiveLimit >= MAX_LINES && currentLineNo >= lineOffset) { - maxLinesReached = true; - } + if (collectionClosed) continue; + if (currentLineNo < lineOffset) continue; + if (selectedEntries.length >= requestedLines) { + collectionClosed = true; continue; } - if (currentLineNo < lineOffset) continue; - if (selectedEntries.length >= effectiveLimit) { - if (effectiveLimit >= MAX_LINES) { - maxLinesReached = true; - } + const rawContent = stripTrailingLf(rawLine); + const lineChars = String(currentLineNo).length + 1 + Math.max( + 0, + rawContent.length - (rawContent.endsWith('\r') ? 1 : 0) - + (currentLineNo === lineOffset ? columnOffset : 0), + ) + (selectedEntries.length === 0 ? 0 : 1); + if (minimumChars + lineChars > maxChars && selectedEntries.length > 0) { collectionClosed = true; continue; } selectedEntries.push({ lineNo: currentLineNo, - rawContent: stripTrailingLf(rawLine), + rawContent, }); - if (selectedEntries.length >= effectiveLimit) { + minimumChars += lineChars; + if (selectedEntries.length >= requestedLines || minimumChars >= maxChars) { collectionClosed = true; } } const lineEndingStyle = lineEndingStyleFromFlags(flags); - const rendered = renderEntries(selectedEntries, lineEndingStyle, lineLengthLimit(eventLog)); - - return this.finishReadResult({ - renderedLines: rendered.renderedLines, - truncatedLineNumbers: rendered.truncatedLineNumbers, - maxLinesReached, - maxBytesReached: rendered.maxBytesReached, - lineEndingStyle, - startLine: selectedEntries.length > 0 ? lineOffset : 0, + const renderedLines = selectedEntries.map((entry) => renderLine(entry, lineEndingStyle)); + const firstLine = renderedLines[0]; + if (columnOffset > 0) { + const prefix = `${String(lineOffset)}\t`; + const text = firstLine?.slice(prefix.length); + if (text === undefined || columnOffset > text.length) { + return { isError: true, output: `column_offset=${String(columnOffset)} is past the end of the starting line ${String(lineOffset)}. Read the line from column 0 to inspect its current contents.` }; + } + if (splitsSurrogatePair(text, columnOffset)) { + return { isError: true, output: `column_offset=${String(columnOffset)} splits a Unicode character in line ${String(lineOffset)}. Use a character boundary or the Next Read arguments.` }; + } + renderedLines[0] = prefix + text.slice(columnOffset); + } + return this.finishPage({ + request, + renderedLines, + startLine: lineOffset, + rangeStart: lineOffset, + rangeEnd: Math.min(currentLineNo, lineOffset + requestedLines - 1), totalLines: currentLineNo, - requestedLines, - detectedEncoding, - eventLog, + fromTail: false, + lineEndingStyle, }); } + private finishPage(page: ReadPage): ExecutableToolResult { + const { args, maxChars, maxCharsLimit, eventLog, detectedEncoding, lossyDecoding } = page.request; + let first = 0; + let end = page.renderedLines.length; + let contentChars = page.renderedLines.reduce((sum, line) => sum + line.length + 1, -1); + const firstColumn = page.fromTail ? 0 : args.column_offset ?? 0; + const firstPrefix = `${String(page.startLine)}\t`; + const firstText = page.renderedLines[0]?.slice(firstPrefix.length) ?? ''; + let fragmentEnd: number | undefined; + + while (true) { + const count = end - first; + const startLine = page.startLine + first; + const endLine = page.startLine + end - 1; + const lineIncomplete = fragmentEnd !== undefined; + const complete = page.rangeStart > page.rangeEnd || + (count > 0 && startLine === page.rangeStart && endLine === page.rangeEnd && !lineIncomplete); + const parts = [ + count > 0 + ? `${String(count)} ${count === 1 ? 'line' : 'lines'} read from file starting from line ${String(startLine)}.` + : 'No lines read from file.', + `Total lines in file: ${String(page.totalLines)}.`, + complete ? 'Requested range complete.' : 'Character limit reached.', + `Effective max_chars: ${String(maxChars)}.`, + ]; + if (!lineIncomplete && ((count > 0 && endLine === page.totalLines) || page.rangeStart > page.totalLines)) { + parts.push('End of file reached.'); + } + if (count > 0 && !page.fromTail && (firstColumn > 0 || fragmentEnd !== undefined)) { + parts.push(`Line ${String(startLine)} fragment: columns [${String(firstColumn)}, ${String(firstColumn + (fragmentEnd ?? firstText.length))}) of ${String(firstColumn + firstText.length)}. ${lineIncomplete ? 'Line continues.' : 'Line complete.'}`); + } + if (args.max_chars !== undefined && args.max_chars > maxCharsLimit) { + parts.push(`Requested max_chars=${String(args.max_chars)} was capped at the configured maximum ${String(maxCharsLimit)}.`); + } + if (!complete && (count > 0 || page.fromTail)) { + const nextStart = page.fromTail ? page.rangeStart : lineIncomplete ? startLine : endLine + 1; + const nextEnd = page.fromTail && count > 0 ? startLine - 1 : page.rangeEnd; + const next = { + path: args.path, + line_offset: nextStart, + column_offset: fragmentEnd !== undefined ? firstColumn + fragmentEnd : undefined, + n_lines: page.fromTail || args.n_lines !== undefined ? nextEnd - nextStart + 1 : undefined, + max_chars: maxChars, + }; + parts.push(`Next Read: ${JSON.stringify(next)}`); + } + if (eventLog) { + parts.push('Pythinker Code agent event log: read one record at a time (n_lines=1); increase max_chars for a longer record or extract fields with Bash.'); + } + if (page.lineEndingStyle === 'mixed') { + parts.push('Mixed or lone carriage-return line endings are shown as \\r. Use exact \\r\\n or \\r escapes in Edit.old_string for those lines.'); + } + if (detectedEncoding !== undefined) { + parts.push(`Detected file encoding: ${encodingDisplayName(detectedEncoding)}; content transcoded to UTF-8 for display. Edit and Write expect UTF-8 — convert the file's encoding first (e.g. \`iconv\` via Bash).`); + } + if (lossyDecoding) { + parts.push('Lossy UTF-16 decoding: malformed sequences were replaced with U+FFFD. The decoded text may differ from the original file.'); + } + const note = `${parts.join(' ')}`; + const renderedChars = count === 0 + ? renderToolResultForModel({ output: '', note }).reduce( + (sum, part) => sum + (part.type === 'text' ? part.text.length : 0), + 0, + ) + : contentChars + 1 + note.length; + if (renderedChars <= maxChars && (complete || count > 0)) { + return { + output: fragmentEnd !== undefined + ? firstPrefix + firstText.slice(0, fragmentEnd) + : page.renderedLines.slice(first, end).join('\n'), + note, + truncated: complete ? undefined : true, + }; + } + if (count === 1 && !page.fromTail) { + const previousEnd = fragmentEnd ?? firstText.length; + fragmentEnd = Math.min(previousEnd - 1, previousEnd - (renderedChars - maxChars)); + if (splitsSurrogatePair(firstText, fragmentEnd)) fragmentEnd -= 1; + if (fragmentEnd <= 0) { + return { isError: true, output: `max_chars=${String(maxChars)} is too small for file text and the Read status. Increase max_chars.` }; + } + contentChars = firstPrefix.length + fragmentEnd; + continue; + } + if (count === 0) { + if (!complete) { + const recovery: ExecutableToolResult = { + isError: true, + output: 'No complete line fits. Continue with the forward Next Read.', + note, + truncated: true, + }; + const recoveryChars = renderToolResultForModel(recovery).reduce( + (sum, part) => sum + (part.type === 'text' ? part.text.length : 0), + 0, + ); + if (recoveryChars <= maxChars) return recovery; + } + return { isError: true, output: `max_chars=${String(maxChars)} is too small for the Read status. Increase max_chars.` }; + } + const dropped = page.fromTail ? first++ : --end; + contentChars -= page.renderedLines[dropped]!.length + 1; + } + } + private async readTail( - displayPath: string, - lines: AsyncIterable, - lineOffset: number, - effectiveLimit: number, - requestedLines: number, - eventLog: boolean, - detectedEncoding?: UtfTextEncoding, + readLines: () => AsyncIterable, + request: ReadRequest, ): Promise { - const tailCount = Math.abs(lineOffset); - const entries: ReadLineEntry[] = []; + const { args, maxChars } = request; + const lineOffset = args.line_offset ?? 1; + const requestedLines = args.n_lines ?? Infinity; + const tailCount = -lineOffset; + const singlePass = requestedLines >= tailCount; + let entries: (ReadTailEntry | undefined)[] = []; + let first = 0; + let chars = 0; + const retainLine = (rawLine: string, lineNo: number): void => { + const rawContent = stripTrailingLf(rawLine); + const minChars = String(lineNo).length + 2 + rawContent.length - (rawContent.endsWith('\r') ? 1 : 0); + entries.push({ lineNo, rawContent, minChars }); + chars += minChars; + while (first < entries.length && (chars - 1 > maxChars || entries.length - first > tailCount)) { + chars -= entries[first]!.minChars; + entries[first++] = undefined; + } + if (first > 1024 && first >= entries.length / 2) { + entries = entries.slice(first); + first = 0; + } + }; const flags: LineEndingFlags = { hasCrLf: false, hasLf: false, hasLoneCr: false }; - let currentLineNo = 0; - - for await (const rawLine of lines) { + let totalLines = 0; + for await (const rawLine of readLines()) { if (containsNulByte(rawLine)) { - return { isError: true, output: notReadableFileOutput(displayPath) }; + return { isError: true, output: notReadableFileOutput(args.path) }; } - currentLineNo += 1; + totalLines += 1; updateLineEndingFlags(flags, rawLine); - entries.push({ - lineNo: currentLineNo, - rawContent: stripTrailingLf(rawLine), - }); - if (entries.length > tailCount) { - entries.shift(); - } - } - - return this.finishTailEntries({ - entries, - lineEndingFlags: flags, - effectiveLimit, - totalLines: currentLineNo, - requestedLines, - eventLog, - detectedEncoding, - }); - } - - private finishTailEntries(input: { - entries: readonly ReadLineEntry[]; - lineEndingFlags: LineEndingFlags; - effectiveLimit: number; - totalLines: number; - requestedLines: number; - eventLog: boolean; - detectedEncoding?: UtfTextEncoding; - }): ExecutableToolResult { - const lineEndingStyle = lineEndingStyleFromFlags(input.lineEndingFlags); - const maxLineLength = lineLengthLimit(input.eventLog); - let renderedCandidates = input.entries.slice(0, input.effectiveLimit).map((entry) => { - return { entry, rendered: renderLine(entry, lineEndingStyle, maxLineLength) }; - }); - - let totalBytes = 0; - for (const [index, candidate] of renderedCandidates.entries()) { - totalBytes += renderedLineBytes(candidate.rendered.line, index === 0); + if (singlePass) retainLine(rawLine, totalLines); } - let maxBytesReached = false; - if (totalBytes > MAX_BYTES) { - maxBytesReached = true; - const kept: typeof renderedCandidates = []; - let bytes = 0; - for (let i = renderedCandidates.length - 1; i >= 0; i -= 1) { - const candidate = renderedCandidates[i]; - if (candidate === undefined) continue; - const lineBytes = renderedLineBytes(candidate.rendered.line, kept.length === 0); - if (kept.length > 0 && bytes + lineBytes > MAX_BYTES) break; - kept.unshift(candidate); - bytes += lineBytes; + const rangeStart = Math.max(1, totalLines + lineOffset + 1); + const rangeEnd = Math.min(totalLines, rangeStart + requestedLines - 1); + const lineEndingStyle = lineEndingStyleFromFlags(flags); + if (!singlePass) { + let currentLine = 0; + for await (const rawLine of readLines()) { + currentLine += 1; + if (currentLine > rangeEnd) break; + if (currentLine < rangeStart) continue; + if (containsNulByte(rawLine)) { + return { isError: true, output: notReadableFileOutput(args.path) }; + } + retainLine(rawLine, currentLine); } - renderedCandidates = kept; - } - - const renderedLines: string[] = []; - const truncatedLineNumbers: number[] = []; - for (const candidate of renderedCandidates) { - renderedLines.push(candidate.rendered.line); - if (candidate.rendered.wasTruncated) { - truncatedLineNumbers.push(candidate.entry.lineNo); + if (currentLine < rangeEnd || currentLine > totalLines) { + return { isError: true, output: 'File changed while reading its tail. Retry Read with the updated file.' }; } } - - return this.finishReadResult({ - renderedLines, - truncatedLineNumbers, - maxLinesReached: false, - maxBytesReached, + const selected = entries.slice(first).map((entry) => renderLine(entry!, lineEndingStyle)); + return this.finishPage({ + request, + renderedLines: selected, + startLine: rangeEnd - selected.length + 1, + rangeStart, + rangeEnd, + totalLines, + fromTail: true, lineEndingStyle, - startLine: renderedCandidates[0]?.entry.lineNo ?? 0, - totalLines: input.totalLines, - requestedLines: input.requestedLines, - detectedEncoding: input.detectedEncoding, - eventLog: input.eventLog, }); } - private finishReadResult(input: FinishReadResultInput): ExecutableToolResult { - return { - output: input.renderedLines.join('\n'), - note: `${this.finishMessage(input)}`, - }; - } - - private finishMessage(input: FinishReadResultInput): string { - const lineCount = input.renderedLines.length; - const lineWord = lineCount === 1 ? 'line' : 'lines'; - const parts = - lineCount > 0 - ? [ - `${String(lineCount)} ${lineWord} read from file starting from line ${String(input.startLine)}.`, - ] - : ['No lines read from file.']; - - parts.push(`Total lines in file: ${String(input.totalLines)}.`); - if (input.maxLinesReached) { - parts.push(`Max ${String(MAX_LINES)} lines reached.`); - } else if (input.maxBytesReached) { - parts.push(`Max ${String(MAX_BYTES)} bytes reached.`); - } else if (lineCount < input.requestedLines) { - parts.push('End of file reached.'); - } - if (input.truncatedLineNumbers.length > 0) { - parts.push( - `Lines [${input.truncatedLineNumbers.join(', ')}] were truncated to ${String(lineLengthLimit(input.eventLog))} characters; use Bash (e.g. cut or sed) to read the elided content of those lines.`, - ); - } - if (input.eventLog) { - parts.push( - `Agent event log: records are returned whole up to ${String(EVENT_LOG_MAX_LINE_LENGTH)} characters per line; read one record at a time (n_lines=1). For a longer record, extract fields with Bash: sed -n 'Np' | jq. A primer on this format appears in your compaction note once a compaction has run.`, - ); - } - if (input.lineEndingStyle === 'mixed') { - parts.push( - 'Mixed or lone carriage-return line endings are shown as \\r. Use exact \\r\\n or \\r escapes in Edit.old_string for those lines.', - ); - } - if (input.detectedEncoding !== undefined) { - parts.push( - `Detected file encoding: ${encodingDisplayName(input.detectedEncoding)}; content transcoded to UTF-8 for display. Edit and Write expect UTF-8 — convert the file's encoding first (e.g. \`iconv\` via Bash).`, - ); - } - return parts.join(' '); - } } registerAgentToolService(IReadTool, ReadTool, { diff --git a/packages/agent-core-v2/test/agent/loop/loop.test.ts b/packages/agent-core-v2/test/agent/loop/loop.test.ts index 26ce39b75..aae843beb 100644 --- a/packages/agent-core-v2/test/agent/loop/loop.test.ts +++ b/packages/agent-core-v2/test/agent/loop/loop.test.ts @@ -162,8 +162,8 @@ describe('Agent loop', () => { [emit] turn.step.started { "time": "