diff --git a/CHANGELOG.md b/CHANGELOG.md index 843055d..f666aa4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,6 +11,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - taskcut is released under the MIT License, in place of Apache-2.0. +### Fixed + +- **Nothing was compacted on Claude Code 2.1.280.** `$.model.complete` now + resolves `{ isAnswered, text, usage }` rather than the reply's text; taskcut + read the object as text, every judgement failed, and the failure was logged + only to the debug log. The reply is now read in either shape, so 2.1.278 + keeps working, and a failed call is logged with its reason. + ## [0.8.0] - 2026-09-22 ### Changed diff --git a/docs/compatibility.md b/docs/compatibility.md index a49188a..d9bf63e 100644 --- a/docs/compatibility.md +++ b/docs/compatibility.md @@ -11,6 +11,7 @@ declarations it is written against are generated per Claude Code version by | `session.start`, `turn.step`, `turn.complete` | hook events | taskcut stops working; `claude plugin validate` reports the unknown event before a session loads it | | `$.session.usage`, `$.session.messages`, `$.session.compact`, `$.fs.read`, `$.env.get`, `$.model.complete`, `$.turn.abort`, `$.prompt.submit`, `$.ui.log` | engine calls | same: refused at load, named in the validation output | | a hook's budget counting only its own code, not its `$` calls | engine rule | the judgement inside a step would overrun it, and the engine would skip the hook: taskcut would compact only at the end of a turn, and say so in the transcript | +| what `$.model.complete` resolves to | data shape | `claude plugin validate` does not see it; only `tsc` against regenerated types does. 2.1.278, which taskcut was measured on, resolved the reply's text; 2.1.280 resolves `{ isAnswered, text, usage }`, and 0.8.0 read that object as text, so every judgement failed and nothing was ever compacted. The reply is now taken as `unknown` and read by `readReply`, which knows both shapes and treats anything else as no reply | | `session.start`'s `isInteractive` | data shape | taskcut would stop ending turns early; a compaction at the end of a turn still works | | `CLAUDE_CODE_ENABLE_FUNCTION_HOOKS` | early-access flag | **nothing**, by design — see below | diff --git a/hooks/judge.ts b/hooks/judge.ts index fc02a12..98e0ac8 100644 --- a/hooks/judge.ts +++ b/hooks/judge.ts @@ -155,6 +155,24 @@ export function judgePrompt(messages: readonly SessionMessage[], step: Step, mem * the emphasis a model sometimes puts round it. Anything else, including no * answer, is not. */ +export type Reply = { text: string } | { reason: string } + +/** + * The judge's reply, whichever shape `$.model.complete` resolved: the reply's + * text up to Claude Code 2.1.278, `{ isAnswered, text }` or + * `{ isAnswered: false, reason }` from 2.1.280. Either is read here, where the + * engine's answer enters, and nothing past this point knows there were two. + */ +export function readReply(answer: unknown): Reply { + if (typeof answer === 'string') return { text: answer } + if (typeof answer === 'object' && answer !== null && 'isAnswered' in answer) { + const { isAnswered, text, reason } = answer as { isAnswered: unknown; text?: unknown; reason?: unknown } + if (isAnswered === true && typeof text === 'string') return { text } + if (isAnswered === false) return { reason: String(reason) } + } + return { reason: 'a reply of no shape taskcut knows' } +} + export function saysDone(answer: string): boolean { const lines = answer.trim().split('\n') const last = lines[lines.length - 1] ?? '' diff --git a/hooks/register.ts b/hooks/register.ts index 6268f19..bf475b6 100644 --- a/hooks/register.ts +++ b/hooks/register.ts @@ -27,7 +27,7 @@ import type { EngineInterface, Register } from 'claude-code' import { ENV_VAR, INERT, decideActivation, type Activation } from './activation' import { readConfig, type Config } from './config' -import { ANSWER_TOKENS, JUDGE_SYSTEM, judgePrompt, saysDone, type Step } from './judge' +import { ANSWER_TOKENS, JUDGE_SYSTEM, judgePrompt, readReply, saysDone, type Step } from './judge' /** * Resolved once, at `session.start`. It starts inert so that a session in which @@ -105,8 +105,15 @@ async function memoryFiles($: EngineInterface): Promise { async function judge($: EngineInterface, step: Step, config: Config): Promise { try { const prompt = judgePrompt(await $.session.messages(), step, await memoryFiles($)) - const answer = await $.model.complete({ model: config.model, system: JUDGE_SYSTEM, prompt, maxTokens: ANSWER_TOKENS }) - return saysDone(answer) + // `unknown`: what the call resolves to has changed between releases, and + // readReply takes every shape it has had. + const answer: unknown = await $.model.complete({ model: config.model, system: JUDGE_SYSTEM, prompt, maxTokens: ANSWER_TOKENS }) + const reply = readReply(answer) + if ('reason' in reply) { + $.ui.log(`could not judge the step (${reply.reason})`, { to: 'debug' }) + return false + } + return saysDone(reply.text) } catch (error) { $.ui.log(`could not judge the step (${String(error)})`, { to: 'debug' }) return false diff --git a/test/judge.test.ts b/test/judge.test.ts index e6fc08e..9906b74 100644 --- a/test/judge.test.ts +++ b/test/judge.test.ts @@ -12,6 +12,7 @@ import { beforeStep, conversationLines, judgePrompt, + readReply, saysDone, } from '../hooks/judge.ts' @@ -126,6 +127,24 @@ describe('judgePrompt', () => { }) }) +describe('readReply', () => { + test('reads the text $.model.complete resolved up to 2.1.278', () => { + assert.deepEqual(readReply('It reports task 2 complete.\nDONE'), { text: 'It reports task 2 complete.\nDONE' }) + }) + + test('reads the result it resolves from 2.1.280', () => { + const usage = { input_tokens: 1, output_tokens: 1, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 } + assert.deepEqual(readReply({ isAnswered: true, text: 'WORKING', usage }), { text: 'WORKING' }) + assert.deepEqual(readReply({ isAnswered: false, reason: 'api-error', status: 529, error: 'overloaded', usage }), { reason: 'api-error' }) + }) + + test('anything else is no reply, never a verdict', () => { + for (const odd of [undefined, null, 42, {}, { isAnswered: true }, { text: 'DONE' }]) { + assert.ok('reason' in readReply(odd), JSON.stringify(odd)) + } + }) +}) + describe('saysDone', () => { test('reads the verdict off the last line', () => { assert.equal(saysDone('It reports task 2 complete and starts task 3.\nDONE'), true)