From f62278e16ddf6ba165f5b2b8418d6a2982e918bb Mon Sep 17 00:00:00 2001 From: mattshax Date: Tue, 15 Sep 2026 19:07:50 +0000 Subject: [PATCH] fix(models): the provider probe no longer marks a healthy family unavailable The probe pinged each provider with max_tokens: 1. One model family rejects every spelling of a token cap, explicitly or behind the gateway's masked 400, so the ping failed twice on its own parameter and the probe reported a reachable, unlocked provider as down, with the mark on every model of the family and a banner telling the user to unlock a key that was never locked. The ping now carries no cap. Its cost is bounded another way: the response is read only until its first frame, which is proof enough that the provider is answering, and the request is then abandoned. A body that names a request parameter as the problem is read as a reachable provider whatever else it says, since a provider that argues about the request's shape is up. Tests cover the parameter rejection, the first-frame read, and the unchanged locked and unavailable verdicts. --- server/src/chat/gateway.ts | 81 ++++++++++++++++++++++++------ server/test/providerProbe.test.mjs | 36 ++++++++++++- 2 files changed, 101 insertions(+), 16 deletions(-) diff --git a/server/src/chat/gateway.ts b/server/src/chat/gateway.ts index 5aa9e3d..5f3f5f7 100644 --- a/server/src/chat/gateway.ts +++ b/server/src/chat/gateway.ts @@ -214,30 +214,81 @@ export function extractUnlockUrl(text: string): string | null { return /unlock_url[\\":\s]*(https?:\/\/[^"\\\s]+)/.exec(text)?.[1] ?? null } +/** A body that names a request parameter as the problem: the provider is up. */ +function rejectedParameter(text: string): boolean { + return /unsupported|unknown|unrecognized|invalid|not supported/i.test(text) + && /parameter|max_tokens|max_output_tokens|max_completion_tokens|temperature|top_p/i.test(text) +} + +/** + * The response up to its first SSE frame, or the whole body when it is + * not a stream. An error body is one JSON object and arrives complete; a + * healthy stream is abandoned after the first data frame proves the + * provider is answering, so an uncapped ping costs the provider a few + * tokens rather than a paragraph. Bodies without a readable stream (the + * test double, a proxy that buffers) fall back to text(). + */ +async function firstFrame(res: Response, ctl: AbortController): Promise { + if (!res.ok || !res.body || typeof (res.body as { getReader?: unknown }).getReader !== 'function') return res.text() + const reader = (res.body as ReadableStream).getReader() + const dec = new TextDecoder() + let buf = '' + try { + for (;;) { + const { value, done } = await reader.read() + if (done) break + buf += dec.decode(value, { stream: true }) + if (/^data: /m.test(buf) || /^\s*\{"error"/.test(buf)) break + if (buf.length > 16_384) break + } + } finally { + try { ctl.abort() } catch { /* already closed */ } + try { reader.releaseLock() } catch { /* already released */ } + } + return buf +} + export async function probeProvider(prefix: string, sampleModelId: string, key?: string | null): Promise { const cacheKey = `${prefix}:${(key ?? '').slice(-6)}` const hit = providerProbes.get(cacheKey) if (hit && Date.now() - hit.at < PROBE_TTL_MS) return hit.v const ping = async (): Promise<{ ok: boolean; status: number; text: string }> => { - const res = await fetch(`${GATEWAY_BASE}/chat/completions`, { - method: 'POST', - headers: { Authorization: `Bearer ${key || gatewayKey()}`, 'Content-Type': 'application/json' }, - // stream:true, deliberately: the gateway masks provider errors on the - // non-streaming path and passes them through on the streaming one - // (measured for both the locked-key 401 and parameter rejections), - // so only a streaming probe can see "API key locked". A healthy - // model answers with SSE frames, which the error check ignores. - body: JSON.stringify({ model: sampleModelId, messages: [{ role: 'user', content: 'ping' }], max_tokens: 1, stream: true }), - signal: AbortSignal.timeout(6000), - }) - const text = await res.text() - // A streaming success arrives as SSE data frames; an error arrives as - // one JSON error body whatever the transport asked for. - return { ok: res.ok && !/^\s*\{"error"/.test(text), status: res.status, text: text.slice(0, 4000) } + // No token cap, deliberately. Some provider families reject every + // spelling of one (max_tokens, max_completion_tokens, max_output_tokens), + // and a probe that carries one then fails on its own parameter and + // reports a healthy provider as down; the mark then sits on every + // model of that family. The cost of an uncapped ping is bounded + // another way: the response is read only until its first frame. + const ctl = new AbortController() + const timer = setTimeout(() => ctl.abort(), 6000) + try { + const res = await fetch(`${GATEWAY_BASE}/chat/completions`, { + method: 'POST', + headers: { Authorization: `Bearer ${key || gatewayKey()}`, 'Content-Type': 'application/json' }, + // stream:true, deliberately: the gateway masks provider errors on the + // non-streaming path and passes them through on the streaming one + // (measured for both the locked-key 401 and parameter rejections), + // so only a streaming probe can see "API key locked". A healthy + // model answers with SSE frames, which the error check ignores. + body: JSON.stringify({ model: sampleModelId, messages: [{ role: 'user', content: 'ping' }], stream: true }), + signal: ctl.signal, + }) + const text = await firstFrame(res, ctl) + // A streaming success arrives as SSE data frames; an error arrives as + // one JSON error body whatever the transport asked for. + return { ok: res.ok && !/^\s*\{"error"/.test(text), status: res.status, text: text.slice(0, 4000) } + } finally { + clearTimeout(timer) + } } let v: ProviderVerdict = { ok: true, kind: null, unlockUrl: null, message: '' } try { let r = await ping() + if (!r.ok && rejectedParameter(r.text)) { + // The provider answered, in detail, about the request's shape. That + // is a reachable, unlocked provider; whatever it disliked is ours. + r = { ...r, ok: true } + } if (!r.ok) { const unlockUrl = extractUnlockUrl(r.text) const credentialish = unlockUrl !== null || r.status === 401 || r.status === 403 || /key locked|locked|unauthorized|api key/i.test(r.text) diff --git a/server/test/providerProbe.test.mjs b/server/test/providerProbe.test.mjs index fb02480..842bd95 100644 --- a/server/test/providerProbe.test.mjs +++ b/server/test/providerProbe.test.mjs @@ -22,7 +22,7 @@ test('the probe streams, and a locked key read through the streaming path is a l responses.push({ status: 400, text: '{"error":{"message":"API key locked - visit the unlock URL to re-enable your key","type":"error"}}' }) const v = await probeProvider('me:genaimil', 'me:genaimil/gemini', 'k1') assert.equal(calls[0].body.stream, true, 'stream:true is the path the gateway does not mask') - assert.equal(calls[0].body.max_tokens, 1) + assert.equal('max_tokens' in calls[0].body, false, 'no token cap: some families reject every spelling of one') assert.deepEqual([v.ok, v.kind], [false, 'locked']) }) test('an unlock url in the body is carried on the verdict', async () => { @@ -66,3 +66,37 @@ test('aiHealth explains a 401 and carries the unlock url when the provider gives assert.equal(ok.status, 'ok'); assert.equal(ok.models, 2) assert.equal(extractUnlockUrl('none'), null) }) + +test('a provider that rejects the token-cap parameter is up, not unavailable', async () => { + // The gpt-5.6 family answers any max_tokens with a parameter rejection, + // explicitly or behind the gateway's masked 400. A probe that carried + // one marked every model of the family [unavailable] while all of them + // answered; the probe now sends no cap, and a parameter complaint is + // read as a reachable provider either way. + reset() + responses.push({ status: 400, text: '{"error":{"message":"{\\"detail\\":\\"Unsupported parameter: max_output_tokens\\"}","type":"error"}}' }) + const v = await probeProvider('me:army', 'me:army/gpt-5.6-luna-gov', 'k7') + assert.deepEqual([v.ok, v.kind], [true, null]) + assert.equal(calls.length, 1, 'no second ping needed for a parameter complaint') +}) + +test('a healthy stream is read only to its first frame', async () => { + reset() + let pulls = 0 + const frames = ['data: {"choices":[{"delta":{"content":"p"}}]}\n\n', 'data: {"choices":[{"delta":{"content":"ong"}}]}\n\n', 'data: [DONE]\n'] + const enc = new TextEncoder() + // highWaterMark 0: chunks are produced only when read, so the count is the probe's reads. + const body = new ReadableStream({ pull(c) { if (pulls < frames.length) c.enqueue(enc.encode(frames[pulls++])); else c.close() } }, { highWaterMark: 0 }) + globalThis.fetch = async (url, init) => { + calls.push({ url: String(url), body: init?.body ? JSON.parse(init.body) : null }) + return { ok: true, status: 200, body, text: async () => frames.join('') } + } + const v = await probeProvider('me:army', 'me:army/gpt-5.6-terra-gov', 'k8') + assert.equal(v.ok, true) + assert.ok(pulls <= 1, `read ${pulls} chunks; one frame is proof enough`) + globalThis.fetch = async (url, init) => { + calls.push({ url: String(url), body: init?.body ? JSON.parse(init.body) : null }) + const r = responses.shift() ?? { status: 200, text: 'data: {"choices":[{"delta":{"content":"pong"}}]}\n\ndata: [DONE]\n' } + return { ok: r.status < 400, status: r.status, text: async () => r.text } + } +})