Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,14 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

- taskcut is released under the MIT License, in place of Apache-2.0.

### Fixed

- **Nothing was compacted on Claude Code 2.1.280.** `$.model.complete` now
resolves `{ isAnswered, text, usage }` rather than the reply's text; taskcut
read the object as text, every judgement failed, and the failure was logged
only to the debug log. The reply is now read in either shape, so 2.1.278
keeps working, and a failed call is logged with its reason.

## [0.8.0] - 2026-09-22

### Changed
Expand Down
1 change: 1 addition & 0 deletions docs/compatibility.md
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ declarations it is written against are generated per Claude Code version by
| `session.start`, `turn.step`, `turn.complete` | hook events | taskcut stops working; `claude plugin validate` reports the unknown event before a session loads it |
| `$.session.usage`, `$.session.messages`, `$.session.compact`, `$.fs.read`, `$.env.get`, `$.model.complete`, `$.turn.abort`, `$.prompt.submit`, `$.ui.log` | engine calls | same: refused at load, named in the validation output |
| a hook's budget counting only its own code, not its `$` calls | engine rule | the judgement inside a step would overrun it, and the engine would skip the hook: taskcut would compact only at the end of a turn, and say so in the transcript |
| what `$.model.complete` resolves to | data shape | `claude plugin validate` does not see it; only `tsc` against regenerated types does. 2.1.278, which taskcut was measured on, resolved the reply's text; 2.1.280 resolves `{ isAnswered, text, usage }`, and 0.8.0 read that object as text, so every judgement failed and nothing was ever compacted. The reply is now taken as `unknown` and read by `readReply`, which knows both shapes and treats anything else as no reply |
| `session.start`'s `isInteractive` | data shape | taskcut would stop ending turns early; a compaction at the end of a turn still works |
| `CLAUDE_CODE_ENABLE_FUNCTION_HOOKS` | early-access flag | **nothing**, by design — see below |

Expand Down
18 changes: 18 additions & 0 deletions hooks/judge.ts
Original file line number Diff line number Diff line change
Expand Up @@ -155,6 +155,24 @@ export function judgePrompt(messages: readonly SessionMessage[], step: Step, mem
* the emphasis a model sometimes puts round it. Anything else, including no
* answer, is not.
*/
export type Reply = { text: string } | { reason: string }

/**
* The judge's reply, whichever shape `$.model.complete` resolved: the reply's
* text up to Claude Code 2.1.278, `{ isAnswered, text }` or
* `{ isAnswered: false, reason }` from 2.1.280. Either is read here, where the
* engine's answer enters, and nothing past this point knows there were two.
*/
export function readReply(answer: unknown): Reply {
if (typeof answer === 'string') return { text: answer }
if (typeof answer === 'object' && answer !== null && 'isAnswered' in answer) {
const { isAnswered, text, reason } = answer as { isAnswered: unknown; text?: unknown; reason?: unknown }
if (isAnswered === true && typeof text === 'string') return { text }
if (isAnswered === false) return { reason: String(reason) }
}
return { reason: 'a reply of no shape taskcut knows' }
}

export function saysDone(answer: string): boolean {
const lines = answer.trim().split('\n')
const last = lines[lines.length - 1] ?? ''
Expand Down
13 changes: 10 additions & 3 deletions hooks/register.ts
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,7 @@ import type { EngineInterface, Register } from 'claude-code'

import { ENV_VAR, INERT, decideActivation, type Activation } from './activation'
import { readConfig, type Config } from './config'
import { ANSWER_TOKENS, JUDGE_SYSTEM, judgePrompt, saysDone, type Step } from './judge'
import { ANSWER_TOKENS, JUDGE_SYSTEM, judgePrompt, readReply, saysDone, type Step } from './judge'

/**
* Resolved once, at `session.start`. It starts inert so that a session in which
Expand Down Expand Up @@ -105,8 +105,15 @@ async function memoryFiles($: EngineInterface): Promise<string[]> {
async function judge($: EngineInterface, step: Step, config: Config): Promise<boolean> {
try {
const prompt = judgePrompt(await $.session.messages(), step, await memoryFiles($))
const answer = await $.model.complete({ model: config.model, system: JUDGE_SYSTEM, prompt, maxTokens: ANSWER_TOKENS })
return saysDone(answer)
// `unknown`: what the call resolves to has changed between releases, and
// readReply takes every shape it has had.
const answer: unknown = await $.model.complete({ model: config.model, system: JUDGE_SYSTEM, prompt, maxTokens: ANSWER_TOKENS })
const reply = readReply(answer)
if ('reason' in reply) {
$.ui.log(`could not judge the step (${reply.reason})`, { to: 'debug' })
return false
}
return saysDone(reply.text)
} catch (error) {
$.ui.log(`could not judge the step (${String(error)})`, { to: 'debug' })
return false
Expand Down
19 changes: 19 additions & 0 deletions test/judge.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ import {
beforeStep,
conversationLines,
judgePrompt,
readReply,
saysDone,
} from '../hooks/judge.ts'

Expand Down Expand Up @@ -126,6 +127,24 @@ describe('judgePrompt', () => {
})
})

describe('readReply', () => {
test('reads the text $.model.complete resolved up to 2.1.278', () => {
assert.deepEqual(readReply('It reports task 2 complete.\nDONE'), { text: 'It reports task 2 complete.\nDONE' })
})

test('reads the result it resolves from 2.1.280', () => {
const usage = { input_tokens: 1, output_tokens: 1, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 }
assert.deepEqual(readReply({ isAnswered: true, text: 'WORKING', usage }), { text: 'WORKING' })
assert.deepEqual(readReply({ isAnswered: false, reason: 'api-error', status: 529, error: 'overloaded', usage }), { reason: 'api-error' })
})

test('anything else is no reply, never a verdict', () => {
for (const odd of [undefined, null, 42, {}, { isAnswered: true }, { text: 'DONE' }]) {
assert.ok('reason' in readReply(odd), JSON.stringify(odd))
}
})
})

describe('saysDone', () => {
test('reads the verdict off the last line', () => {
assert.equal(saysDone('It reports task 2 complete and starts task 3.\nDONE'), true)
Expand Down
Loading