From 4debf5b425c4be02eb3787ae276fab3fd159ee00 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 03:19:04 +0200 Subject: [PATCH 01/17] feat(provider): read model facts from the served catalog instead of the shipped table Add a catalog client for /ai/models: cached per API base URL and project, single-flight, 5-minute TTL with stale-while-revalidate, never throws. Every model fact reader now reads the cached catalog: operations decide native Responses support, reasoning_budget_tokens the thinking budget, and the two chat_completions capability fields the chat flags. Provider aliases, short aliases, the Mistral check and the default model follow the served catalog too. Model construction stays synchronous; the catalog loads on the first async step and a model built before it loaded is rebuilt when the facts differ. The shipped table is imported only by a deprecation shim that keeps the table-backed exports for one release. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- CHANGELOG.md | 29 ++ docs/api-reference/veryfront/provider.md | 67 ++-- src/agent/hosted/default-chat-runtime.test.ts | 4 + src/agent/hosted/default-chat-runtime.ts | 9 + .../hosted/executor-runtime-prepare.test.ts | 12 +- .../veryfront-cloud-agent-service.test.ts | 2 + .../runtime/default-provider-options.test.ts | 6 +- src/agent/runtime/model-resolution.test.ts | 6 +- src/agent/runtime/model-resolution.ts | 7 +- src/agent/runtime/model-transport.test.ts | 6 +- src/agent/runtime/model-transport.ts | 11 +- src/platform/cloud/resolver.test.ts | 20 + src/platform/cloud/resolver.ts | 10 +- src/provider/index.ts | 7 +- .../catalog-client.test-helpers.ts | 377 ++++++++++++++++++ .../veryfront-cloud/catalog-client.ts | 295 ++++++++++++++ .../veryfront-cloud/gateway-routing.test.ts | 6 +- .../model-catalog.deprecated.test.ts | 61 +++ .../model-catalog.deprecated.ts | 111 ++++++ .../model-catalog.served.test.ts | 331 +++++++++++++++ .../veryfront-cloud/model-catalog.test.ts | 7 +- src/provider/veryfront-cloud/model-catalog.ts | 373 +++++++++++------ src/provider/veryfront-cloud/provider.test.ts | 131 +++++- src/provider/veryfront-cloud/provider.ts | 207 +++++++++- src/provider/veryfront-cloud/shared.test.ts | 6 +- .../model-call-context-request.test.ts | 6 +- .../hosted-application-model-resolver.test.ts | 6 +- .../run-scoped-inference-credential.test.ts | 6 +- .../veryfront-cloud-catalog-client.test.ts | 252 ++++++++++++ .../veryfront-cloud-model-id-rule.test.ts | 8 +- ...yfront-cloud-model-table-importers.test.ts | 36 ++ .../issue-1834-recorded-context.test.ts | 2 + 32 files changed, 2208 insertions(+), 209 deletions(-) create mode 100644 src/provider/veryfront-cloud/catalog-client.test-helpers.ts create mode 100644 src/provider/veryfront-cloud/catalog-client.ts create mode 100644 src/provider/veryfront-cloud/model-catalog.deprecated.test.ts create mode 100644 src/provider/veryfront-cloud/model-catalog.deprecated.ts create mode 100644 src/provider/veryfront-cloud/model-catalog.served.test.ts create mode 100644 tests/integration/provider/veryfront-cloud-catalog-client.test.ts create mode 100644 tests/integration/provider/veryfront-cloud-model-table-importers.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 574ec82643..a527cdfc3d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,35 @@ For finish-required streams in the active lifecycle, a completed turn containing only unavailable tool calls now permits the same recovery turn as the legacy lifecycle. Rejected tools are not executed, and malformed or empty handoff requests still fail. +### Changed: Veryfront Cloud model facts come from the served model catalog + +`veryfront-cloud/*` models now read their facts from the model catalog your +Veryfront API serves at `/ai/models`, instead of from a table shipped in +this package. The facts are the wire protocol, whether the Responses operation +is served, the default thinking budget, the two chat completions capability +flags, short model aliases such as `opus`, and the default model. For the +models this package listed, the served facts are the same, so requests are +unchanged. + +- Model construction stays synchronous and makes no network call. The catalog + loads on the first async step of a model (`prepare`, `doGenerate` or + `doStream`), with the same credentials and project as inference, and is + cached for five minutes. +- When the catalog cannot be loaded, calls still go out. Each model then uses + its protocol defaults: a provider named `openai`, `anthropic` or `google` + speaks its own protocol, any other provider speaks the OpenAI protocol, and + no thinking defaults apply. A short alias such as `opus` resolves only once + the catalog is loaded. +- A Mistral model ID the catalog does not list is refused only once the + catalog is loaded. Before that, the platform answers for it. +- `resolveVeryfrontCloudDefaultModelId()` returns the default model the + catalog names, or the built-in default before it loads. + `VeryfrontCloudModelId` types a model ID as `/`. +- `VERYFRONT_CLOUD_CHAT_MODELS`, `findVeryfrontCloudModel`, + `findVeryfrontCloudModelByModelId` and `groupVeryfrontCloudModelsByProvider` + are deprecated. They still return the list shipped with this package, and a + later release removes them. + ### Changed: Veryfront Cloud models call the vendor-neutral endpoints `veryfront-cloud/*` models that speak the OpenAI protocol now send requests to diff --git a/docs/api-reference/veryfront/provider.md b/docs/api-reference/veryfront/provider.md index 47078d366b..12e7287967 100644 --- a/docs/api-reference/veryfront/provider.md +++ b/docs/api-reference/veryfront/provider.md @@ -63,44 +63,47 @@ Clear all registered model providers and reset lazy built-ins (for testing). ### Components -| Name | Description | Source | -| ---------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `DEFAULT_VERYFRONT_CLOUD_MODEL_ID` | Default Veryfront Cloud model ID used when no model is configured. Update this when the current default is deprecated - otherwise the default path silently breaks for users who have not set an explicit model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VERYFRONT_CLOUD_CHAT_MODELS` | Shared Veryfront Cloud chat models value. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VERYFRONT_CLOUD_MODEL_PREFIX` | Shared Veryfront Cloud model prefix value. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ---------------------------------- | -------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | +| `DEFAULT_VERYFRONT_CLOUD_MODEL_ID` | Short ID of the built-in default model, used when no model is configured and the served catalog has not been loaded. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VERYFRONT_CLOUD_CHAT_MODELS` | Chat models shipped with this package. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `VERYFRONT_CLOUD_MODEL_PREFIX` | Shared Veryfront Cloud model prefix value. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Functions -| Name | Description | Source | -| ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `findVeryfrontCloudModel` | Find Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `findVeryfrontCloudModelByModelId` | Find Veryfront Cloud model by model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `groupVeryfrontCloudModelsByProvider` | Group Veryfront Cloud models by provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelId` | Resolves Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | +| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `findVeryfrontCloudModel` | Find a shipped chat model by its short id. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `findVeryfrontCloudModelByModelId` | Find a shipped chat model by its provider-qualified id, in any provider spelling. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `groupVeryfrontCloudModelsByProvider` | Group the shipped chat models by provider, in display order. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveVeryfrontCloudDefaultModelId` | Provider-qualified ID of the default model: the one the served catalog names once it is loaded, otherwise the built-in default. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelId` | Resolve a model ID or short alias to a provider-qualified model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Types -| Name | Description | Source | -| ----------------------------------- | ------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `ModelProviderFactory` | Public API contract for model provider factory. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ModelProviderRegistrationDisposer` | | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ModelRuntime` | Public API contract for model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/types.ts) | -| `VeryfrontCloudChatModel` | Public API contract for Veryfront Cloud chat model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VeryfrontCloudModelThinkingConfig` | Configuration used by Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VeryfrontCloudProviderId` | A Veryfront Cloud provider ID: a listed provider, or any other well-formed provider string. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ----------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | +| `ModelProviderFactory` | Public API contract for model provider factory. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ModelProviderRegistrationDisposer` | | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ModelRuntime` | Public API contract for model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/types.ts) | +| `VeryfrontCloudChatModel` | Public API contract for Veryfront Cloud chat model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudModelId` | A provider-qualified Veryfront Cloud model ID, for example `anthropic/claude-sonnet-4-6`. Any provider and model the platform serves fits, so a new model needs no release of this package. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudModelThinkingConfig` | Configuration used by Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudProviderId` | A Veryfront Cloud provider ID: a listed provider, or any other well-formed provider string. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudRuntimeModelId` | A model ID routed through Veryfront Cloud: `veryfront-cloud//`. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Constants diff --git a/src/agent/hosted/default-chat-runtime.test.ts b/src/agent/hosted/default-chat-runtime.test.ts index 2f86f0288c..3769630ad5 100644 --- a/src/agent/hosted/default-chat-runtime.test.ts +++ b/src/agent/hosted/default-chat-runtime.test.ts @@ -8,6 +8,7 @@ import { assertStringIncludes, } from "#veryfront/testing/assert.ts"; import { it } from "#veryfront/testing/bdd.ts"; +import { useServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; import { deleteEnv, getEnv, setEnv } from "#veryfront/compat/process.ts"; import { refreshEnvironmentConfig } from "#veryfront/config/environment-config.ts"; import { clearModelProviders, type ModelRuntime, registerModelProvider } from "#veryfront/provider"; @@ -465,6 +466,7 @@ it("applies refreshed structured system messages in hosted chat", async () => { }); Deno.test("createDefaultHostedChatRuntime builds a cloud-backed hosted runtime", async () => { + using _catalog = useServedCatalogForTests(); let capturedContext: DefaultHostedChatRuntimeTaskContext | undefined; let capturedCapability: unknown; const runEventWriterCapability = createHostedRunEventWriterCapability({ @@ -1027,6 +1029,7 @@ Deno.test("hosted first provider call filters skill tools for every tool selecto }); Deno.test("createDefaultHostedChatRuntime forwards hosted project slug to integration discovery", async () => { + using _catalog = useServedCatalogForTests(); const previousApiBaseUrl = getEnv("VERYFRONT_API_BASE_URL"); const previousApiToken = getEnv("VERYFRONT_API_TOKEN"); const previousProjectSlug = getEnv("VERYFRONT_PROJECT_SLUG"); @@ -1106,6 +1109,7 @@ Deno.test("createDefaultHostedChatRuntime forwards hosted project slug to integr }); Deno.test("createDefaultHostedChatRuntime keeps per-run host tools out of the global registry", async () => { + using _catalog = useServedCatalogForTests(); try { const createRuntime = (description: string) => createDefaultHostedChatRuntime({ diff --git a/src/agent/hosted/default-chat-runtime.ts b/src/agent/hosted/default-chat-runtime.ts index 7476324247..32cecd58f9 100644 --- a/src/agent/hosted/default-chat-runtime.ts +++ b/src/agent/hosted/default-chat-runtime.ts @@ -17,6 +17,7 @@ import { resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; +import { loadVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { runWithVeryfrontCloudContext, runWithVeryfrontCloudContextAsync, @@ -533,6 +534,14 @@ export async function createDefaultHostedChatRuntime( return await runWithEffectiveSourceIntegrationPolicy( input.sourceIntegrationPolicy, async () => { + // A short model alias resolves through the served catalog. + if (input.config.apiUrl && input.options.authToken) { + await loadVeryfrontCloudCatalog({ + apiBaseUrl: input.config.apiUrl, + apiToken: input.options.authToken, + ...(input.options.projectSlug ? { projectSlug: input.options.projectSlug } : {}), + }); + } const modelId = resolveVeryfrontCloudModelId(input.options.model); const cloudContext = createCloudContext({ config: input.config, diff --git a/src/agent/hosted/executor-runtime-prepare.test.ts b/src/agent/hosted/executor-runtime-prepare.test.ts index 4d9b3e7946..a80e1a30d9 100644 --- a/src/agent/hosted/executor-runtime-prepare.test.ts +++ b/src/agent/hosted/executor-runtime-prepare.test.ts @@ -1,7 +1,9 @@ import "#veryfront/schemas/_test-setup.ts"; import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; import { PERMISSION_DENIED } from "#veryfront/errors"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; import type { ModelRuntime, ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { defineSchema } from "#veryfront/schemas/index.ts"; @@ -166,6 +168,8 @@ async function prepare( } describe("executor runtime preparation", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("extracts inline tools under the source policy in the full-runtime profile", async () => { const policy = { schemaVersion: 1 as const, mode: "allowlist" as const, integrations: {} }; const observed: unknown[] = []; @@ -847,6 +851,8 @@ function syntheticRemoteTool(name: string): ToolDefinition { } describe("executor runtime preparation review regressions", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); const outputCases: { name: string; request: Record; @@ -2289,14 +2295,14 @@ Synthetic source instructions.`, const selectedModel of [ modelId, "veryfront-cloud/anthropic/claude-sonnet-4-6", - "veryfront-cloud/anthropic/claude-opus-4-7", + "veryfront-cloud/anthropic/claude-opus-4-8", ] ) { for (const thinking of [{ enabled: false }, { enabled: true, budgetTokens: 4096 }]) { it(`carries explicit thinking ${thinking.enabled} into ${selectedModel} call data`, async () => { let captured: ModelRuntimeCallOptions | undefined; let sourceTransportCalls = 0; - const adaptive = selectedModel.endsWith("claude-opus-4-7") && thinking.enabled; + const adaptive = selectedModel.endsWith("claude-opus-4-8") && thinking.enabled; const f = fixture({ config: { thinking: { enabled: !thinking.enabled }, diff --git a/src/agent/hosted/veryfront-cloud-agent-service.test.ts b/src/agent/hosted/veryfront-cloud-agent-service.test.ts index f8fa333cf6..ca22f6a161 100644 --- a/src/agent/hosted/veryfront-cloud-agent-service.test.ts +++ b/src/agent/hosted/veryfront-cloud-agent-service.test.ts @@ -8,6 +8,7 @@ import { assertStrictEquals, } from "#veryfront/testing/assert.ts"; import { describe, it } from "#veryfront/testing/bdd.ts"; +import { useServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; import { resolve } from "node:path"; import { pathToFileURL } from "node:url"; import type { CreateSandboxBashTool } from "#veryfront/sandbox"; @@ -83,6 +84,7 @@ Deno.test("public agent service options expose deployment-owned remote MCP compo }); Deno.test("root and child runtimes use the deployment-owned remote MCP factory", async () => { + using _catalog = useServedCatalogForTests(); const createdConfigs: RemoteMCPToolSourceConfig[] = []; let failStudioListing = false; let modelCallCount = 0; diff --git a/src/agent/runtime/default-provider-options.test.ts b/src/agent/runtime/default-provider-options.test.ts index a34ed6aae6..4c1ddeaf51 100644 --- a/src/agent/runtime/default-provider-options.test.ts +++ b/src/agent/runtime/default-provider-options.test.ts @@ -1,9 +1,13 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; describe("resolveProviderOptionsWithDefaults", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("enables Anthropic thinking by default for Anthropic models", () => { const result = resolveProviderOptionsWithDefaults( "anthropic/claude-sonnet-4-6", diff --git a/src/agent/runtime/model-resolution.test.ts b/src/agent/runtime/model-resolution.test.ts index da04d817e9..e74cbca9b8 100644 --- a/src/agent/runtime/model-resolution.test.ts +++ b/src/agent/runtime/model-resolution.test.ts @@ -7,7 +7,9 @@ import { } from "#veryfront/testing/assert.ts"; import { VeryfrontError } from "#veryfront/errors"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; -import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { VERYFRONT_CLOUD_CHAT_MODELS } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { AUTO_AGENT_MODEL, @@ -41,6 +43,8 @@ function clearModelEnv(): void { } describe("agent/runtime/model-resolution", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { clearModelEnv(); }); diff --git a/src/agent/runtime/model-resolution.ts b/src/agent/runtime/model-resolution.ts index d769cd018f..52758b750a 100644 --- a/src/agent/runtime/model-resolution.ts +++ b/src/agent/runtime/model-resolution.ts @@ -4,7 +4,7 @@ import { getMistralEnvConfig, getOpenAIEnvConfig, } from "#veryfront/config/env.ts"; -import { findVeryfrontCloudModelByModelId } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; +import { isSupportedMistralModelId } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { DEFAULT_MODEL_CREDENTIAL_MISMATCH, NOT_SUPPORTED } from "#veryfront/errors"; import { getDefaultVeryfrontCloudModel, @@ -132,12 +132,11 @@ function listAvailableDirectProviders(): string[] { } function isSupportedHostedMistralModel(modelId: string): boolean { - return Boolean(findVeryfrontCloudModelByModelId(`mistral/${modelId}`)); + return isSupportedMistralModelId(`mistral/${modelId}`); } function isUnsupportedVeryfrontCloudMistralModel(modelId: string): boolean { - return modelId.startsWith("veryfront-cloud/mistral/") && - !findVeryfrontCloudModelByModelId(modelId); + return modelId.startsWith("veryfront-cloud/mistral/") && !isSupportedMistralModelId(modelId); } function normalizeVeryfrontCloudRuntimeModel(modelId: string): string { diff --git a/src/agent/runtime/model-transport.test.ts b/src/agent/runtime/model-transport.test.ts index e7914c6990..e71f86e254 100644 --- a/src/agent/runtime/model-transport.test.ts +++ b/src/agent/runtime/model-transport.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertStrictEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { ModelRuntime } from "#veryfront/provider"; import type { AgentConfig, ModelTransportRequest } from "../types.ts"; import { resolveAgentModelTransport } from "./model-transport.ts"; @@ -23,6 +25,8 @@ function createModel(modelId: string): ModelRuntime { } describe("resolveAgentModelTransport", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("resolves the configured runtime model when no host transport hook is present", async () => { const config: AgentConfig = { model: "local/qwen3.5-0.8b", diff --git a/src/agent/runtime/model-transport.ts b/src/agent/runtime/model-transport.ts index 3556c03e3c..9c5df2afe4 100644 --- a/src/agent/runtime/model-transport.ts +++ b/src/agent/runtime/model-transport.ts @@ -7,6 +7,7 @@ import { type AgentConfig, type RuntimeReasoningOption } from "../types.ts"; import { type ModelRuntime, resolveModel } from "#veryfront/provider"; import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; +import { warmVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/provider.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; import { resolveConfiguredAgentModel, @@ -211,10 +212,16 @@ export async function resolveAgentModelTransport( ): Promise { const requestedModel = resolveConfiguredAgentModel(input.modelOverride || input.config.model); const resolvedModelString = resolveRuntimeModel(input.modelOverride || input.config.model); - const privatelyResolvedModel = input.resolveModelRuntime && - IntrinsicReflectApply(StringStartsWith, resolvedModelString, [VERYFRONT_CLOUD_MODEL_PREFIX]) + const usesVeryfrontCloud = IntrinsicReflectApply(StringStartsWith, resolvedModelString, [ + VERYFRONT_CLOUD_MODEL_PREFIX, + ]) as boolean; + const privatelyResolvedModel = input.resolveModelRuntime && usesVeryfrontCloud ? input.resolveModelRuntime(resolvedModelString) : undefined; + // Thinking defaults below read the served catalog. A privately resolved + // model's run was prepared from the catalog as it stood then, and its call + // must keep what that preparation reserved, so only ambient runs load it here. + if (usesVeryfrontCloud && !privatelyResolvedModel) await warmVeryfrontCloudCatalog(); const transport = privatelyResolvedModel ? undefined : await input.config.resolveModelTransport?.({ diff --git a/src/platform/cloud/resolver.test.ts b/src/platform/cloud/resolver.test.ts index 010f035d54..985a18f3d7 100644 --- a/src/platform/cloud/resolver.test.ts +++ b/src/platform/cloud/resolver.test.ts @@ -13,6 +13,10 @@ import { createTestConfig, } from "#veryfront/config/runtime-config.ts"; import { runWithVeryfrontCloudContext } from "#veryfront/provider/veryfront-cloud/context.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { runWithProjectEnv } from "#veryfront/server/project-env"; import { runWithRequestContext } from "#veryfront/platform/adapters/fs/veryfront/request-context.ts"; import { __resetEnvLoaderForTests } from "#veryfront/utils/env-loader.ts"; @@ -116,6 +120,22 @@ describe("platform/cloud/resolver", () => { ); }); + it("defaults to the served default model once the catalog loads, unless overridden", () => { + assertEquals(getDefaultVeryfrontCloudModel(), "veryfront-cloud/mistral/mistral-small-2503"); + + __setVeryfrontCloudCatalogForTests({ + models: [], + defaultModelId: "anthropic/claude-sonnet-4-6", + }); + try { + assertEquals(getDefaultVeryfrontCloudModel(), "veryfront-cloud/anthropic/claude-sonnet-4-6"); + setEnv("VERYFRONT_DEFAULT_MODEL", "openai/gpt-5.2"); + assertEquals(getDefaultVeryfrontCloudModel(), "veryfront-cloud/openai/gpt-5.2"); + } finally { + __resetVeryfrontCloudCatalogForTests(); + } + }); + it("lets scoped cloud context override env bootstrap values", () => { setEnv("VERYFRONT_API_TOKEN", "vf_env_token"); setEnv("VERYFRONT_PROJECT_SLUG", "env-project"); diff --git a/src/platform/cloud/resolver.ts b/src/platform/cloud/resolver.ts index 18d03b91b4..be2be79644 100644 --- a/src/platform/cloud/resolver.ts +++ b/src/platform/cloud/resolver.ts @@ -1,4 +1,5 @@ import { getRuntimeRequestContext } from "#veryfront/platform/runtime-request-context.ts"; +import { peekVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { getHostEnv, getHostEnvExcludingEnvFile, @@ -92,6 +93,7 @@ function resolveHostCredentialApiBaseUrl(): string { DEFAULT_API_BASE_URL; } +/** Built-in default model, used until the served catalog names one. */ export const DEFAULT_VERYFRONT_CLOUD_MODEL = "veryfront-cloud/mistral/mistral-small-2503"; export const DEFAULT_VERYFRONT_CLOUD_EMBEDDING_MODEL = "veryfront-cloud/openai/text-embedding-3-small"; @@ -229,10 +231,16 @@ export function isVeryfrontCloudEnabled(): boolean { return Boolean(bootstrap.apiToken && hasProjectContext); } +/** + * Default Veryfront Cloud model: `VERYFRONT_DEFAULT_MODEL` when set, otherwise + * the default the served catalog names once it is loaded, otherwise the + * built-in default. + */ export function getDefaultVeryfrontCloudModel(): string { + const served = peekVeryfrontCloudCatalog()?.defaultModelId; return normalizeCloudModelString( getHostEnv("VERYFRONT_DEFAULT_MODEL"), - DEFAULT_VERYFRONT_CLOUD_MODEL, + served?.includes("/") ? served : DEFAULT_VERYFRONT_CLOUD_MODEL, ); } diff --git a/src/provider/index.ts b/src/provider/index.ts index 5030799b3d..457039ab49 100644 --- a/src/provider/index.ts +++ b/src/provider/index.ts @@ -21,7 +21,11 @@ export { } from "./model-registry.ts"; export type { ModelProviderFactory, ModelProviderRegistrationDisposer } from "./model-registry.ts"; export type { ModelRuntime } from "./types.ts"; -export type { VeryfrontCloudProviderId } from "./veryfront-cloud/model-catalog.ts"; +export type { + VeryfrontCloudModelId, + VeryfrontCloudProviderId, + VeryfrontCloudRuntimeModelId, +} from "./veryfront-cloud/model-catalog.ts"; export { DEFAULT_VERYFRONT_CLOUD_MODEL_ID, findVeryfrontCloudModel, @@ -30,6 +34,7 @@ export { groupVeryfrontCloudModelsByProvider, normalizeVeryfrontCloudModelId, resolveHostedVeryfrontCloudModelId, + resolveVeryfrontCloudDefaultModelId, resolveVeryfrontCloudGatewayModelId, resolveVeryfrontCloudModelId, resolveVeryfrontCloudModelThinking, diff --git a/src/provider/veryfront-cloud/catalog-client.test-helpers.ts b/src/provider/veryfront-cloud/catalog-client.test-helpers.ts new file mode 100644 index 0000000000..d3c399d91e --- /dev/null +++ b/src/provider/veryfront-cloud/catalog-client.test-helpers.ts @@ -0,0 +1,377 @@ +/** + * Served model catalog fixtures for tests. + * + * `SERVED_MODEL_ROWS` copies the `/ai/models` rows the platform serves today + * for the models this package's shipped table lists, reduced to the fields + * the catalog client reads. `UNSERVED_TABLE_MODEL_ROWS` describes the models + * the shipped table still lists but the platform no longer serves, with the + * facts the table carries, so tests written against those IDs keep working. + */ +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "./catalog-client.ts"; + +/** Served rows for the models the shipped table lists. */ +export const SERVED_MODEL_ROWS = [ + { + "id": "claude-opus-4-8", + "modelId": "anthropic/claude-opus-4-8", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "opus", + "anthropic/claude-opus-4-8", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "adaptive", + }, + }, + { + "id": "claude-opus-4-6", + "modelId": "anthropic/claude-opus-4-6", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "anthropic/claude-opus-4-6", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "budget", + "reasoning_budget_tokens": 2048, + }, + }, + { + "id": "claude-sonnet-4-6", + "modelId": "anthropic/claude-sonnet-4-6", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "sonnet", + "anthropic/claude-sonnet-4-6", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "budget", + "reasoning_budget_tokens": 2048, + }, + }, + { + "id": "claude-haiku-4-5-20251001", + "modelId": "anthropic/claude-haiku-4-5-20251001", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "haiku", + "anthropic/claude-haiku-4-5-20251001", + "claude-haiku-4-5", + "anthropic/claude-haiku-4-5", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "budget", + "reasoning_budget_tokens": 1024, + }, + }, + { + "id": "gpt-5.5", + "modelId": "openai/gpt-5.5", + "provider": "openai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.5", + "gpt-5.5", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "transport": "chat-completions", + "chat_completions_reasoning_with_function_tools": false, + }, + }, + { + "id": "gpt-5.4-mini", + "modelId": "openai/gpt-5.4-mini", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.4-mini", + "gpt-5.4-mini", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gpt-5.4", + "modelId": "openai/gpt-5.4", + "provider": "openai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.4", + "gpt-5.4", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "transport": "chat-completions", + "chat_completions_reasoning_with_function_tools": false, + }, + }, + { + "id": "gpt-5-nano", + "modelId": "openai/gpt-5-nano", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5-nano", + "gpt-5-nano", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gemini-3.5-flash", + "modelId": "google-ai-studio/gemini-3.5-flash", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-3.5-flash", + "gemini-3.5-flash", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + }, + }, + { + "id": "gemini-2.5-pro", + "modelId": "google-ai-studio/gemini-2.5-pro", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-2.5-pro", + "gemini-2.5-pro", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gemini-2.5-flash", + "modelId": "google-ai-studio/gemini-2.5-flash", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-2.5-flash", + "gemini-2.5-flash", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + }, + }, + { + "id": "mistral-small-2503", + "modelId": "mistral/mistral-small-2503", + "provider": "mistral", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "mistral/mistral-small-2503", + "mistral-small-2503", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + "chat_completions_consecutive_system_messages": true, + }, + }, + { + "id": "kimi-k2.6", + "modelId": "moonshotai/kimi-k2.6", + "provider": "moonshotai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "moonshotai/kimi-k2.6", + "kimi-k2.6", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "kimi-k2.5", + "modelId": "moonshotai/kimi-k2.5", + "provider": "moonshotai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "moonshotai/kimi-k2.5", + "kimi-k2.5", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, +] as const; + +/** Rows for models the shipped table lists that the platform no longer serves. */ +export const UNSERVED_TABLE_MODEL_ROWS = [ + { + "id": "gpt-5.4-nano", + "modelId": "openai/gpt-5.4-nano", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.4-nano", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gpt-5.2", + "modelId": "openai/gpt-5.2", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.2", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gemini-3.1-pro-preview", + "modelId": "google-ai-studio/gemini-3.1-pro-preview", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-3.1-pro-preview", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "mistral-large-2512", + "modelId": "mistral/mistral-large-2512", + "provider": "mistral", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "mistral/mistral-large-2512", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + }, + }, +] as const; + +/** Default model the platform serves. */ +export const SERVED_DEFAULT_MODEL_ID = "mistral/mistral-small-2503"; + +/** A `/ai/models` payload with every row above. */ +export function servedCatalogPayload(): Record { + return { + models: [...SERVED_MODEL_ROWS, ...UNSERVED_TABLE_MODEL_ROWS], + defaultModelId: SERVED_DEFAULT_MODEL_ID, + }; +} + +/** Serve {@link servedCatalogPayload} to every catalog read until reset. */ +export function seedServedCatalogForTests(): void { + __setVeryfrontCloudCatalogForTests(servedCatalogPayload()); +} + +/** + * Serve {@link servedCatalogPayload} until the returned handle is disposed: + * `using _catalog = useServedCatalogForTests();`. + */ +export function useServedCatalogForTests(): Disposable { + seedServedCatalogForTests(); + return { [Symbol.dispose]: __resetVeryfrontCloudCatalogForTests }; +} diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts new file mode 100644 index 0000000000..673d122164 --- /dev/null +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -0,0 +1,295 @@ +/** + * Client for the model catalog Veryfront Cloud serves at `/ai/models`. + * + * Model facts (wire protocol, operations, thinking defaults and transport + * capabilities) come from the served catalog, not from a table shipped in this + * package. Loading is asynchronous and happens on the first async step of a + * model call; every synchronous reader uses {@link peekVeryfrontCloudCatalog} + * and degrades when nothing is loaded yet. + * + * - Entries are cached per API base URL and project, because the served list + * is filtered by project policy. + * - Concurrent loads for one key share a single request. + * - An entry is fresh for {@link VERYFRONT_CLOUD_CATALOG_TTL_MS}. A stale entry + * is returned at once while one refresh runs in the background, and it is + * kept when that refresh fails. + * - A failed load never throws. It is logged once and retried after + * {@link VERYFRONT_CLOUD_CATALOG_RETRY_MS}. + */ +import { createVeryfrontApiOriginBoundOutboundFetch } from "#veryfront/security/http/outbound-fetch.ts"; +import { logger } from "#veryfront/utils/logger/logger.ts"; + +/** How long a loaded catalog is used before it is refreshed. */ +export const VERYFRONT_CLOUD_CATALOG_TTL_MS = 5 * 60_000; +/** How long a failed load waits before the next attempt for the same key. */ +export const VERYFRONT_CLOUD_CATALOG_RETRY_MS = 30_000; +/** Upper bound on one catalog request. */ +const VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS = 10_000; +/** Path of the served catalog, relative to the API base URL. */ +const VERYFRONT_CLOUD_CATALOG_PATH = "ai/models"; + +/** One model of the served catalog, reduced to the facts this package reads. */ +export interface VeryfrontCloudCatalogModel { + /** Short model id, for example `claude-sonnet-4-6`. */ + readonly id: string; + /** Provider-qualified model id, for example `anthropic/claude-sonnet-4-6`. */ + readonly modelId: string; + /** Canonical provider the model belongs to. */ + readonly provider: string; + /** Other ids that select this model. */ + readonly aliases: readonly string[]; + /** Wire protocol the model is served on. Absent when the platform declares none. */ + readonly surface?: string; + /** Operations of `surface` the model is served on. Absent on an older API. */ + readonly operations?: readonly string[]; + /** Whether the model takes thinking controls. */ + readonly thinking?: boolean; + /** Which reasoning control the model takes, for example `budget` or `adaptive`. */ + readonly reasoningMode?: string; + /** OpenAI wire API the model must use, when the platform constrains it. */ + readonly transport?: string; + /** Thinking budget to send when the caller sets none. */ + readonly reasoningBudgetTokens?: number; + /** Whether `chat-completions` accepts reasoning together with function tools. */ + readonly chatCompletionsReasoningWithFunctionTools?: boolean; + /** Whether `chat-completions` accepts adjacent system messages separately. */ + readonly chatCompletionsConsecutiveSystemMessages?: boolean; +} + +/** A loaded catalog. */ +export interface VeryfrontCloudCatalog { + readonly models: readonly VeryfrontCloudCatalogModel[]; + /** Provider-qualified id of the default model, when the platform names one. */ + readonly defaultModelId?: string; +} + +/** Credentials and scope a catalog load uses: the same ones inference uses. */ +export interface VeryfrontCloudCatalogLoadOptions { + readonly apiBaseUrl: string; + readonly apiToken: string; + readonly projectSlug?: string; + readonly signal?: AbortSignal; +} + +interface CatalogEntry { + catalog: VeryfrontCloudCatalog; + fetchedAt: number; +} + +const entries = new Map(); +const inflight = new Map>(); +const failedAt = new Map(); +let latest: VeryfrontCloudCatalog | undefined; +let seeded: VeryfrontCloudCatalog | undefined; +let failureLogged = false; +let now: () => number = Date.now; +/** Bumped by a test reset, so a load that settles afterwards changes nothing. */ +let generation = 0; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function optionalString(value: unknown): string | undefined { + return typeof value === "string" && value.length > 0 ? value : undefined; +} + +function optionalBoolean(value: unknown): boolean | undefined { + return typeof value === "boolean" ? value : undefined; +} + +function optionalPositiveInteger(value: unknown): number | undefined { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined; +} + +function stringList(value: unknown): string[] | undefined { + if (!Array.isArray(value)) return undefined; + return value.filter((item): item is string => typeof item === "string" && item.length > 0); +} + +function parseModel(value: unknown): VeryfrontCloudCatalogModel | undefined { + if (!isRecord(value)) return undefined; + const id = optionalString(value.id); + const modelId = optionalString(value.modelId); + const provider = optionalString(value.provider); + if (!id || !modelId || !provider) return undefined; + const capabilities = isRecord(value.capabilities) ? value.capabilities : {}; + const model: VeryfrontCloudCatalogModel = { + id, + modelId, + provider, + aliases: Object.freeze(stringList(value.aliases) ?? []), + surface: optionalString(value.surface), + operations: ((operations) => operations && Object.freeze(operations))( + stringList(value.operations), + ), + thinking: optionalBoolean(capabilities.thinking), + reasoningMode: optionalString(capabilities.reasoning_mode), + transport: optionalString(capabilities.transport), + reasoningBudgetTokens: optionalPositiveInteger(capabilities.reasoning_budget_tokens), + chatCompletionsReasoningWithFunctionTools: optionalBoolean( + capabilities.chat_completions_reasoning_with_function_tools, + ), + chatCompletionsConsecutiveSystemMessages: optionalBoolean( + capabilities.chat_completions_consecutive_system_messages, + ), + }; + return Object.freeze(model); +} + +/** + * Read a served `/ai/models` payload. Rows without an id, model id or provider + * are skipped, and a field an older API does not serve reads as absent. + * Returns undefined when the payload carries no model list at all. + */ +export function parseVeryfrontCloudCatalog(payload: unknown): VeryfrontCloudCatalog | undefined { + if (!isRecord(payload) || !Array.isArray(payload.models)) return undefined; + const models: VeryfrontCloudCatalogModel[] = []; + for (const row of payload.models) { + const model = parseModel(row); + if (model) models.push(model); + } + return Object.freeze({ + models: Object.freeze(models), + defaultModelId: optionalString(payload.defaultModelId), + }); +} + +function cacheKey(apiBaseUrl: string, projectSlug: string | undefined): string { + return `${apiBaseUrl}\n${projectSlug ?? ""}`; +} + +function catalogUrl(apiBaseUrl: string): string { + const url = new URL(apiBaseUrl); + url.pathname = `${url.pathname.replace(/\/+$/, "")}/${VERYFRONT_CLOUD_CATALOG_PATH}`; + url.hash = ""; + url.search = ""; + return url.toString(); +} + +async function fetchCatalog( + options: VeryfrontCloudCatalogLoadOptions, +): Promise { + const headers = new Headers({ + Accept: "application/json", + Authorization: `Bearer ${options.apiToken}`, + }); + if (options.projectSlug) headers.set("x-veryfront-project-slug", options.projectSlug); + const timeout = new AbortController(); + const timer = setTimeout(() => timeout.abort(), VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS); + const signal = options.signal + ? AbortSignal.any([options.signal, timeout.signal]) + : timeout.signal; + try { + const response = await createVeryfrontApiOriginBoundOutboundFetch(options.apiBaseUrl)( + catalogUrl(options.apiBaseUrl), + { method: "GET", headers, signal }, + ); + if (!response.ok) { + await response.body?.cancel(); + throw new Error( + `Veryfront Cloud model catalog request failed with status ${response.status}`, + ); + } + const catalog = parseVeryfrontCloudCatalog(await response.json()); + if (!catalog) throw new Error("Veryfront Cloud model catalog response has no model list"); + return catalog; + } finally { + clearTimeout(timer); + } +} + +function refresh( + key: string, + options: VeryfrontCloudCatalogLoadOptions, +): Promise { + const pending = inflight.get(key); + if (pending) return pending; + const started = generation; + const request = fetchCatalog(options).then( + (catalog) => { + if (started !== generation) return catalog; + entries.set(key, { catalog, fetchedAt: now() }); + failedAt.delete(key); + latest = catalog; + failureLogged = false; + return catalog; + }, + (error: unknown) => { + if (started !== generation) return undefined; + failedAt.set(key, now()); + if (!failureLogged) { + failureLogged = true; + logger.warn( + "Veryfront Cloud model catalog is unavailable; model facts fall back to protocol defaults", + { error: error instanceof Error ? error.message : String(error) }, + ); + } + return entries.get(key)?.catalog; + }, + ).finally(() => { + if (started === generation) inflight.delete(key); + }); + inflight.set(key, request); + return request; +} + +/** + * Load the served catalog for an API base URL and project. Resolves to the + * cached catalog when it is fresh, to a stale one while a refresh runs, and to + * undefined when no catalog could be loaded. Never rejects. + */ +export function loadVeryfrontCloudCatalog( + options: VeryfrontCloudCatalogLoadOptions, +): Promise { + if (seeded) return Promise.resolve(seeded); + const key = cacheKey(options.apiBaseUrl, options.projectSlug); + const entry = entries.get(key); + const current = now(); + if (entry && current - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS) { + return Promise.resolve(entry.catalog); + } + const lastFailure = failedAt.get(key); + if (lastFailure !== undefined && current - lastFailure < VERYFRONT_CLOUD_CATALOG_RETRY_MS) { + return Promise.resolve(entry?.catalog); + } + const request = refresh(key, options); + // Stale while revalidate: the stale entry answers now, the refresh replaces it. + return entry ? Promise.resolve(entry.catalog) : request; +} + +/** Whether a load for these credentials would answer from a fresh cache entry, without a request. */ +export function isVeryfrontCloudCatalogFresh( + options: Pick, +): boolean { + if (seeded) return true; + const entry = entries.get(cacheKey(options.apiBaseUrl, options.projectSlug)); + return entry !== undefined && now() - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS; +} + +/** The most recently loaded catalog, stale or not, or undefined before any load. */ +export function peekVeryfrontCloudCatalog(): VeryfrontCloudCatalog | undefined { + return seeded ?? latest; +} + +/** @internal Serve a fixed catalog for every key, as if freshly loaded. `undefined` clears it. */ +export function __setVeryfrontCloudCatalogForTests(payload: unknown): void { + seeded = payload === undefined ? undefined : parseVeryfrontCloudCatalog(payload); +} + +/** @internal Forget every loaded catalog, pending load and failure. */ +export function __resetVeryfrontCloudCatalogForTests(): void { + generation++; + entries.clear(); + inflight.clear(); + failedAt.clear(); + latest = undefined; + seeded = undefined; + failureLogged = false; + now = Date.now; +} + +/** @internal Replace the clock the cache reads. */ +export function __setVeryfrontCloudCatalogClockForTests(clock: () => number): void { + now = clock; +} diff --git a/src/provider/veryfront-cloud/gateway-routing.test.ts b/src/provider/veryfront-cloud/gateway-routing.test.ts index 58024a523f..f659ee958c 100644 --- a/src/provider/veryfront-cloud/gateway-routing.test.ts +++ b/src/provider/veryfront-cloud/gateway-routing.test.ts @@ -7,7 +7,9 @@ */ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { resolveGenAiProviderName } from "#veryfront/agent/hosted/trace-attributes.ts"; import { getProviderToolProfile } from "#veryfront/agent/runtime/provider-tool-compat.ts"; import { @@ -48,6 +50,8 @@ function routingRow(modelId: string): RoutingRow { } describe("provider/veryfront-cloud gateway routing", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("keeps the routing facts of every catalog model", () => { assertEquals(VERYFRONT_CLOUD_CHAT_MODELS.map((model) => routingRow(model.modelId)), [ { diff --git a/src/provider/veryfront-cloud/model-catalog.deprecated.test.ts b/src/provider/veryfront-cloud/model-catalog.deprecated.test.ts new file mode 100644 index 0000000000..3d506184e9 --- /dev/null +++ b/src/provider/veryfront-cloud/model-catalog.deprecated.test.ts @@ -0,0 +1,61 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import * as barrel from "#veryfront/provider"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "./catalog-client.ts"; +import * as catalog from "./model-catalog.ts"; +import * as shim from "./model-catalog.deprecated.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, +} from "./model-catalog.data.ts"; + +describe("provider/veryfront-cloud/model-catalog deprecated exports", () => { + afterEach(__resetVeryfrontCloudCatalogForTests); + + it("keeps the shipped model list whatever the served catalog says", () => { + __setVeryfrontCloudCatalogForTests({ models: [], defaultModelId: "acme/acme-1" }); + + assertEquals( + shim.VERYFRONT_CLOUD_CHAT_MODELS.map((model) => model.modelId), + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => model.modelId), + ); + assertEquals(shim.DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL.id, TABLE_DEFAULT_MODEL_ID); + assertEquals(shim.findVeryfrontCloudModel("opus")?.modelId, "anthropic/claude-opus-4-8"); + assertEquals( + shim.findVeryfrontCloudModelByModelId("veryfront-cloud/google/gemini-2.5-pro")?.id, + "gemini-2.5-pro", + ); + assertEquals( + shim.groupVeryfrontCloudModelsByProvider().map((group) => group.provider), + ["anthropic", "openai", "google", "mistral", "moonshotai"], + ); + }); + + it("is what the catalog module and the public barrel export under the same names", () => { + for ( + const name of [ + "VERYFRONT_CLOUD_CHAT_MODELS", + "DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL", + "findVeryfrontCloudModel", + "findVeryfrontCloudModelByModelId", + "groupVeryfrontCloudModelsByProvider", + ] as const + ) { + assertEquals(catalog[name], shim[name], name); + } + for ( + const name of [ + "VERYFRONT_CLOUD_CHAT_MODELS", + "findVeryfrontCloudModel", + "findVeryfrontCloudModelByModelId", + "groupVeryfrontCloudModelsByProvider", + ] as const + ) { + assertEquals(barrel[name], shim[name], name); + } + }); +}); diff --git a/src/provider/veryfront-cloud/model-catalog.deprecated.ts b/src/provider/veryfront-cloud/model-catalog.deprecated.ts new file mode 100644 index 0000000000..c4d96216a2 --- /dev/null +++ b/src/provider/veryfront-cloud/model-catalog.deprecated.ts @@ -0,0 +1,111 @@ +/** + * Deprecated model list exports, backed by the table shipped in this package. + * + * Model facts now come from the served catalog (`catalog-client.ts`), and no + * resolution logic reads this module. It keeps the public exports that only + * make sense with a shipped list working for one release, and it is the only + * module that imports the shipped table. + */ +import type { KnownVeryfrontCloudProviderId, VeryfrontCloudChatModel } from "./model-catalog.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, + VERYFRONT_CLOUD_PROVIDER_ALIASES, + VERYFRONT_CLOUD_PROVIDER_LABELS as PROVIDER_LABELS, + VERYFRONT_CLOUD_PROVIDER_ORDER as PROVIDER_ORDER, +} from "./model-catalog.data.ts"; + +const MODEL_PREFIX = "veryfront-cloud/"; +const providerAliases: ReadonlyMap = new Map(VERYFRONT_CLOUD_PROVIDER_ALIASES); + +function isPositiveSafeInteger(value: unknown): value is number { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0; +} + +/** `/` by the shipped alias list, prefix removed. */ +function tableModelKey(modelId: string): string { + const normalized = modelId.startsWith(MODEL_PREFIX) + ? modelId.slice(MODEL_PREFIX.length) + : modelId; + const slashIndex = normalized.indexOf("/"); + if (slashIndex <= 0) return normalized; + const provider = normalized.slice(0, slashIndex); + return `${providerAliases.get(provider) ?? provider}/${normalized.slice(slashIndex + 1)}`; +} + +/** + * Chat models shipped with this package. + * + * @deprecated Read the served catalog instead. This list is removed in a later release. + */ +export const VERYFRONT_CLOUD_CHAT_MODELS: readonly VeryfrontCloudChatModel[] = Object.freeze( + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => { + if ( + model.thinkingBudgetTokens !== undefined && + !isPositiveSafeInteger(model.thinkingBudgetTokens) + ) { + throw new TypeError( + `Veryfront Cloud model "${model.id}" thinkingBudgetTokens must be a positive safe integer`, + ); + } + return Object.freeze(model); + }), +); + +const defaultVeryfrontCloudChatModel = VERYFRONT_CLOUD_CHAT_MODELS.find( + (model) => model.id === TABLE_DEFAULT_MODEL_ID, +); +if (!defaultVeryfrontCloudChatModel) { + throw new Error( + `Veryfront Cloud default model "${TABLE_DEFAULT_MODEL_ID}" is missing from the catalog`, + ); +} + +/** + * Shipped descriptor of the built-in default model. + * + * @deprecated Read the served catalog instead. This descriptor is removed in a later release. + */ +export const DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL = defaultVeryfrontCloudChatModel; + +/** + * Find a shipped chat model by its short id. + * + * @deprecated Read the served catalog instead. This lookup is removed in a later release. + */ +export function findVeryfrontCloudModel( + id: string, +): VeryfrontCloudChatModel | undefined { + return VERYFRONT_CLOUD_CHAT_MODELS.find((model) => model.id === id); +} + +/** + * Find a shipped chat model by its provider-qualified id, in any provider spelling. + * + * @deprecated Read the served catalog instead. This lookup is removed in a later release. + */ +export function findVeryfrontCloudModelByModelId( + modelId: string, +): VeryfrontCloudChatModel | undefined { + const key = tableModelKey(modelId); + return VERYFRONT_CLOUD_CHAT_MODELS.find((model) => tableModelKey(model.modelId) === key); +} + +/** + * Group the shipped chat models by provider, in display order. + * + * @deprecated Read the served catalog instead. This grouping is removed in a later release. + */ +export function groupVeryfrontCloudModelsByProvider(): Array<{ + readonly provider: KnownVeryfrontCloudProviderId; + readonly label: string; + readonly models: readonly VeryfrontCloudChatModel[]; +}> { + return PROVIDER_ORDER.map((provider) => ({ + provider, + label: PROVIDER_LABELS[provider], + models: Object.freeze( + VERYFRONT_CLOUD_CHAT_MODELS.filter((model) => model.provider === provider), + ), + })).filter((group) => group.models.length > 0); +} diff --git a/src/provider/veryfront-cloud/model-catalog.served.test.ts b/src/provider/veryfront-cloud/model-catalog.served.test.ts new file mode 100644 index 0000000000..22b4a6195e --- /dev/null +++ b/src/provider/veryfront-cloud/model-catalog.served.test.ts @@ -0,0 +1,331 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "./catalog-client.ts"; +import { + seedServedCatalogForTests, + SERVED_MODEL_ROWS, + servedCatalogPayload, +} from "./catalog-client.test-helpers.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, + isSupportedMistralModelId, + resolveVeryfrontCloudDefaultModelId, + resolveVeryfrontCloudModelId, + resolveVeryfrontCloudModelThinking, + resolveVeryfrontCloudOpenAIChatFunctionToolReasoning, + resolveVeryfrontCloudOpenAIChatSystemMessages, + resolveVeryfrontCloudOpenAITransport, + resolveVeryfrontCloudOpenAITransportPlan, + resolveVeryfrontCloudProviderId, + resolveVeryfrontCloudProviderRouting, + resolveVeryfrontCloudReasoningOption, + resolveVeryfrontCloudThinkingProviderOptions, +} from "./model-catalog.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, + VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, + VERYFRONT_CLOUD_PROVIDER_ALIASES, + VERYFRONT_CLOUD_PROVIDER_ROUTING, +} from "./model-catalog.data.ts"; +import { isOpenAIReasoningModel } from "../shared/openai-reasoning.ts"; + +type ServedRow = { + id: string; + modelId: string; + provider: string; + surface?: string; + operations?: readonly string[]; + aliases: readonly string[]; + capabilities: Record; +}; + +/** A served payload of the given rows, each with the fields a row always carries. */ +function payload(rows: readonly ServedRow[], defaultModelId?: string): Record { + return { models: rows, ...(defaultModelId ? { defaultModelId } : {}) }; +} + +function row(modelId: string, fields: Partial = {}): ServedRow { + const [provider = "", id = ""] = modelId.split("/"); + return { id, modelId, provider, aliases: [], capabilities: {}, ...fields }; +} + +describe("provider/veryfront-cloud/model-catalog served facts", () => { + afterEach(__resetVeryfrontCloudCatalogForTests); + + describe("before the catalog is loaded", () => { + it("routes providers named after a protocol natively and any other on the default surface", () => { + assertEquals(resolveVeryfrontCloudProviderRouting("openai"), { + surface: "openai", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("anthropic"), { + surface: "anthropic", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("google"), { + surface: "google", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("mistral"), { surface: "openai" }); + }); + + it("declares no model facts and uses the built-in default model", () => { + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), undefined); + assertEquals(resolveVeryfrontCloudOpenAITransport("openai/gpt-5.5"), undefined); + assertEquals( + resolveVeryfrontCloudOpenAIChatSystemMessages("mistral/mistral-small-2503"), + undefined, + ); + assertEquals( + resolveVeryfrontCloudDefaultModelId(), + DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, + ); + assertEquals(resolveVeryfrontCloudModelId(), DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID); + // The platform refuses a model it does not serve; nothing is refused locally. + assertEquals(isSupportedMistralModelId("mistral/not-served"), true); + }); + }); + + describe("once the catalog is loaded", () => { + it("reads native providers from the operations each model is served on", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-chat-only", { surface: "openai", operations: ["chat-completions"] }), + row("acme/acme-1", { surface: "openai", operations: ["responses", "chat-completions"] }), + row("anthropic/claude-x", { surface: "anthropic", operations: ["messages"] }), + ])); + + assertEquals(resolveVeryfrontCloudProviderRouting("openai"), { + surface: "openai", + native: false, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("acme"), { + surface: "openai", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("anthropic"), { + surface: "anthropic", + native: true, + }); + }); + + it("keeps a model the platform does not serve on Responses on chat completions", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-5.9", { + surface: "openai", + operations: ["responses", "chat-completions"], + capabilities: { thinking: true }, + }), + row("openai/gpt-5.9-chat", { + surface: "openai", + operations: ["chat-completions"], + capabilities: { thinking: true }, + }), + ])); + + assertEquals(resolveVeryfrontCloudOpenAITransportPlan("openai", "gpt-5.9"), { + transport: "responses", + pinned: true, + }); + assertEquals(resolveVeryfrontCloudOpenAITransportPlan("openai", "gpt-5.9-chat"), { + transport: "chat-completions", + pinned: true, + }); + }); + + it("reads the thinking budget from reasoning_budget_tokens", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("anthropic/claude-sonnet-4-6", { + surface: "anthropic", + aliases: ["sonnet"], + capabilities: { thinking: true, reasoning_mode: "budget", reasoning_budget_tokens: 4096 }, + }), + row("openai/gpt-think", { surface: "openai", capabilities: { thinking: true } }), + row("openai/gpt-plain", { surface: "openai", capabilities: { thinking: false } }), + ])); + + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), { + enabled: true, + budgetTokens: 4096, + }); + assertEquals(resolveVeryfrontCloudModelThinking("sonnet"), { + enabled: true, + budgetTokens: 4096, + }); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-think"), { enabled: true }); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-plain"), undefined); + }); + + it("reads both chat completions flags from their capability fields", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-5.4", { + surface: "openai", + capabilities: { + transport: "chat-completions", + chat_completions_reasoning_with_function_tools: true, + }, + }), + row("mistral/mistral-small-2503", { + surface: "openai", + capabilities: { chat_completions_consecutive_system_messages: false }, + }), + ])); + + assertEquals(resolveVeryfrontCloudOpenAITransport("openai/gpt-5.4"), "chat-completions"); + assertEquals(resolveVeryfrontCloudOpenAIChatFunctionToolReasoning("openai/gpt-5.4"), true); + assertEquals( + resolveVeryfrontCloudOpenAIChatSystemMessages("mistral/mistral-small-2503"), + false, + ); + }); + + it("resolves a provider alias from the served provider field", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("vendor-studio/vendor-model", { provider: "vendor", surface: "google" }), + ])); + + assertEquals(resolveVeryfrontCloudProviderId("vendor-studio"), "vendor"); + assertEquals(resolveVeryfrontCloudProviderRouting("vendor-studio").surface, "google"); + }); + + it("uses the default model the catalog names", () => { + __setVeryfrontCloudCatalogForTests( + payload([row("anthropic/claude-sonnet-4-6")], "anthropic/claude-sonnet-4-6"), + ); + + assertEquals(resolveVeryfrontCloudDefaultModelId(), "anthropic/claude-sonnet-4-6"); + assertEquals(resolveVeryfrontCloudModelId(), "anthropic/claude-sonnet-4-6"); + }); + + it("refuses a Mistral model the catalog does not list", () => { + seedServedCatalogForTests(); + + assertEquals(isSupportedMistralModelId("mistral/mistral-small-2503"), true); + assertEquals(isSupportedMistralModelId("mistral/not-served"), false); + }); + }); + + describe("parity with the shipped table for today's models", () => { + const tableAliases = new Map(VERYFRONT_CLOUD_PROVIDER_ALIASES); + const tableKey = (modelId: string): string => { + const slash = modelId.indexOf("/"); + const provider = modelId.slice(0, slash); + return `${tableAliases.get(provider) ?? provider}/${modelId.slice(slash + 1)}`; + }; + const tableRouting = new Map(VERYFRONT_CLOUD_PROVIDER_ROUTING); + const tableCapabilities = new Map(VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES); + const servedRows = SERVED_MODEL_ROWS as readonly ServedRow[]; + + /** The transport plan the table-backed resolver chose, from the table's own inputs. */ + function tablePlan(provider: string, upstreamModelId: string, modelId: string) { + const routing = tableRouting.get(provider); + if (routing?.surface !== "openai" || routing.native !== true) { + return { transport: "chat-completions", pinned: true }; + } + const declared = tableCapabilities.get(tableKey(modelId))?.openAITransport; + if (declared !== undefined) return { transport: declared, pinned: true }; + const entry = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + tableKey(model.modelId) === tableKey(modelId) + ); + if (entry?.thinking === true || entry?.thinkingBudgetTokens !== undefined) { + return { transport: "responses", pinned: true }; + } + if (isOpenAIReasoningModel(upstreamModelId, "veryfront-cloud")) { + return { transport: "responses", pinned: true }; + } + return { transport: "chat-completions", pinned: false }; + } + + it("covers every served row with a shipped table entry", () => { + for (const served of servedRows) { + const entry = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + tableKey(model.modelId) === tableKey(served.modelId) + ); + assertEquals(entry !== undefined, true, `${served.modelId} is not in the shipped table`); + } + }); + + for (const served of servedRows) { + const entry = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + tableKey(model.modelId) === tableKey(served.modelId) + ); + if (!entry) continue; + + it(`serves the shipped facts of ${entry.modelId}`, () => { + __setVeryfrontCloudCatalogForTests(servedCatalogPayload()); + const key = tableKey(entry.modelId); + const [provider = "", upstreamModelId = ""] = [ + key.slice(0, key.indexOf("/")), + key.slice(key.indexOf("/") + 1), + ]; + const capabilities = tableCapabilities.get(key); + const routing = tableRouting.get(provider); + + assertEquals(resolveVeryfrontCloudProviderId(entry.modelId.split("/")[0] ?? ""), provider); + assertEquals(resolveVeryfrontCloudProviderRouting(provider).surface, routing?.surface); + assertEquals( + resolveVeryfrontCloudProviderRouting(provider).native, + routing?.native === true, + ); + assertEquals(resolveVeryfrontCloudModelId(entry.id), entry.modelId); + assertEquals( + resolveVeryfrontCloudOpenAITransport(entry.modelId), + capabilities?.openAITransport, + ); + assertEquals( + resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(entry.modelId), + capabilities?.openAIChatReasoningWithFunctionTools, + ); + assertEquals( + resolveVeryfrontCloudOpenAIChatSystemMessages(entry.modelId), + capabilities?.openAIChatPreserveSystemMessages, + ); + if (routing?.surface === "openai") { + assertEquals( + resolveVeryfrontCloudOpenAITransportPlan(provider, upstreamModelId), + tablePlan(provider, upstreamModelId, entry.modelId), + ); + } + + const tableThinking = entry.thinking === true || entry.thinkingBudgetTokens !== undefined + ? { + enabled: true, + ...(entry.thinkingBudgetTokens === undefined + ? {} + : { budgetTokens: entry.thinkingBudgetTokens }), + } + : undefined; + const servedThinking = resolveVeryfrontCloudModelThinking(entry.modelId); + if (capabilities?.anthropicThinkingMode === "adaptive") { + // An adaptive model takes no budget, so the served catalog declares + // none; what is sent is the same with or without the shipped one. + assertEquals(servedThinking?.enabled, true); + assertEquals( + resolveVeryfrontCloudThinkingProviderOptions(entry.modelId, servedThinking), + resolveVeryfrontCloudThinkingProviderOptions(entry.modelId, tableThinking), + ); + assertEquals( + resolveVeryfrontCloudReasoningOption(entry.modelId, servedThinking), + resolveVeryfrontCloudReasoningOption(entry.modelId, tableThinking), + ); + } else { + assertEquals(servedThinking, tableThinking); + } + }); + } + + it("serves the shipped default model", () => { + seedServedCatalogForTests(); + const tableDefault = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + model.id === TABLE_DEFAULT_MODEL_ID + ); + + assertEquals(resolveVeryfrontCloudDefaultModelId(), tableDefault?.modelId); + assertEquals(DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, tableDefault?.modelId); + }); + }); +}); diff --git a/src/provider/veryfront-cloud/model-catalog.test.ts b/src/provider/veryfront-cloud/model-catalog.test.ts index 957b37e6d7..6cff1961a7 100644 --- a/src/provider/veryfront-cloud/model-catalog.test.ts +++ b/src/provider/veryfront-cloud/model-catalog.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertExists, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { canonicalVeryfrontCloudModelKey, DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL, @@ -28,6 +30,8 @@ import { import { VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES } from "./model-catalog.data.ts"; describe("provider/veryfront-cloud/model-catalog", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("retires DeepSeek from managed selections while retaining Mistral as default", () => { assertEquals(findVeryfrontCloudModelByModelId("deepseek/deepseek-v4-flash"), undefined); assertEquals( @@ -649,7 +653,6 @@ describe("provider/veryfront-cloud/model-catalog", () => { it("keeps adaptive Anthropic thinking out of provider-neutral reasoning", () => { for ( const modelId of [ - "anthropic/claude-opus-4-7", "anthropic/claude-opus-4-8", "veryfront-cloud/anthropic/claude-opus-4-8", ] diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index 1e4dea170f..bb9826b874 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -1,20 +1,18 @@ import { INVALID_ARGUMENT, NOT_SUPPORTED } from "#veryfront/errors"; import { isOpenAIReasoningModel } from "../shared/openai-reasoning.ts"; import { - DEFAULT_VERYFRONT_CLOUD_GATEWAY_API_VERSION, - DEFAULT_VERYFRONT_CLOUD_MODEL_ID as CATALOG_DEFAULT_MODEL_ID, - DEFAULT_VERYFRONT_CLOUD_SURFACE, - VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, - VERYFRONT_CLOUD_GATEWAY_PATH_PREFIX, - VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, - VERYFRONT_CLOUD_PROVIDER_ALIASES, - VERYFRONT_CLOUD_PROVIDER_LABELS as PROVIDER_LABELS, - VERYFRONT_CLOUD_PROVIDER_ORDER as PROVIDER_ORDER, - VERYFRONT_CLOUD_PROVIDER_ROUTING, - VERYFRONT_CLOUD_SURFACE_GATEWAY_API_VERSIONS, - type VeryfrontCloudModelTransportCapabilities, - type VeryfrontCloudProviderRouting, -} from "./model-catalog.data.ts"; + peekVeryfrontCloudCatalog, + type VeryfrontCloudCatalog, + type VeryfrontCloudCatalogModel, +} from "./catalog-client.ts"; + +export { + DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL, + findVeryfrontCloudModel, + findVeryfrontCloudModelByModelId, + groupVeryfrontCloudModelsByProvider, + VERYFRONT_CLOUD_CHAT_MODELS, +} from "./model-catalog.deprecated.ts"; /** * Veryfront Cloud providers listed in the catalog of this package. @@ -43,6 +41,16 @@ export type VeryfrontCloudProviderId = | KnownVeryfrontCloudProviderId | (string & Record); +/** + * A provider-qualified Veryfront Cloud model ID, for example + * `anthropic/claude-sonnet-4-6`. Any provider and model the platform serves + * fits, so a new model needs no release of this package. + */ +export type VeryfrontCloudModelId = `${string}/${string}`; + +/** A model ID routed through Veryfront Cloud: `veryfront-cloud//`. */ +export type VeryfrontCloudRuntimeModelId = `veryfront-cloud/${string}/${string}`; + /** Wire format a Veryfront Cloud gateway endpoint speaks. */ export type VeryfrontCloudWireSurface = "openai" | "anthropic" | "google"; @@ -55,6 +63,24 @@ export type VeryfrontCloudSurfaceId = | VeryfrontCloudWireSurface | (string & Record); +/** + * Gateway routing for one provider: the wire format its endpoint speaks, and + * whether it implements that format natively. On the OpenAI surface, only a + * native provider can use the Responses transport. + */ +export type VeryfrontCloudProviderRouting = { + readonly surface: VeryfrontCloudSurfaceId; + readonly native?: boolean; +}; + +/** Model-specific transport capabilities that cannot be inferred from the provider family. */ +type VeryfrontCloudModelTransportCapabilities = { + readonly anthropicThinkingMode?: "adaptive"; + readonly openAITransport?: "chat-completions" | "responses"; + readonly openAIChatReasoningWithFunctionTools?: boolean; + readonly openAIChatPreserveSystemMessages?: boolean; +}; + /** Configuration used by Veryfront Cloud model thinking. */ export type VeryfrontCloudModelThinkingConfig = { enabled: boolean; @@ -88,32 +114,135 @@ function requireThinkingBudgetTokens(value: unknown): number | undefined { } /** - * Default Veryfront Cloud model ID used when no model is configured. - * Update this when the current default is deprecated — otherwise the default - * path silently breaks for users who have not set an explicit model. + * Short ID of the built-in default model, used when no model is configured + * and the served catalog has not been loaded. */ -export const DEFAULT_VERYFRONT_CLOUD_MODEL_ID = CATALOG_DEFAULT_MODEL_ID; +export const DEFAULT_VERYFRONT_CLOUD_MODEL_ID = "mistral-small-2503"; /** Shared Veryfront Cloud model prefix value. */ export const VERYFRONT_CLOUD_MODEL_PREFIX = "veryfront-cloud/"; +/** Provider-qualified ID of the built-in default model. */ +export const DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID: VeryfrontCloudModelId = + "mistral/mistral-small-2503"; +/** Veryfront Cloud runtime ID of the built-in default model. */ +export const DEFAULT_VERYFRONT_CLOUD_RUNTIME_MODEL_ID: VeryfrontCloudRuntimeModelId = + `veryfront-cloud/${DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID}`; + +/** + * Provider-qualified ID of the default model: the one the served catalog + * names once it is loaded, otherwise the built-in default. + */ +export function resolveVeryfrontCloudDefaultModelId(): VeryfrontCloudModelId { + const served = peekVeryfrontCloudCatalog()?.defaultModelId; + return served !== undefined && served.includes("/") + ? served as VeryfrontCloudModelId + : DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID; +} + +/** Leading gateway path segments of a vendor-scoped route, shared by every surface. */ +const VENDOR_GATEWAY_PATH_PREFIX = "ai/gateway"; +/** Vendor-scoped gateway API version per wire protocol. */ +const VENDOR_GATEWAY_API_VERSIONS: ReadonlyMap = new Map([ + ["anthropic", "v1"], + ["openai", "v1"], + ["google", "v1beta"], +]); +/** Vendor-scoped gateway API version for a protocol without its own entry. */ +const DEFAULT_VENDOR_GATEWAY_API_VERSION = "v1"; +/** Surface used for a provider the served catalog does not describe. */ +const DEFAULT_VERYFRONT_CLOUD_SURFACE = "openai"; +/** + * Providers named after the wire protocol they implement. When the served + * catalog does not describe a provider, one of these speaks its own protocol + * natively and any other provider speaks the default surface. + */ +const PROTOCOL_NAMED_PROVIDERS: ReadonlySet = new Set(["openai", "anthropic", "google"]); + +/** Lookups built once per loaded catalog. */ +interface ServedCatalogIndex { + /** Provider segment of a served model ID -> the canonical provider it belongs to. */ + readonly providerAliases: ReadonlyMap; + /** Canonical provider -> routing derived from its served models. */ + readonly routing: ReadonlyMap>; + /** `/` -> served model. */ + readonly byKey: ReadonlyMap; + /** Short ID or bare alias -> served model. */ + readonly byShortId: ReadonlyMap; + /** Exact provider-qualified model ID -> served model. */ + readonly byModelId: ReadonlyMap; +} + +const servedIndexes = new WeakMap(); -/** Private runtime Map for alias lookups, built from the frozen data entries. */ -const _providerAliasMap = new Map(VERYFRONT_CLOUD_PROVIDER_ALIASES); -/** Private runtime Map for transport-capability lookups, built from the frozen data entries. */ -const _transportCapabilitiesMap = new Map( - VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, -); -/** Private runtime Map for provider routing lookups, built from the frozen data entries. */ -const _providerRoutingMap = new Map(VERYFRONT_CLOUD_PROVIDER_ROUTING); -/** Private runtime Map for gateway API version lookups, built from the frozen data entries. */ -const _surfaceGatewayApiVersionMap = new Map( - VERYFRONT_CLOUD_SURFACE_GATEWAY_API_VERSIONS, -); - -/** Resolve a supported gateway provider alias without consulting object prototypes. */ +function modelSegment(modelId: string): string { + return modelId.slice(modelId.indexOf("/") + 1); +} + +function buildServedIndex(catalog: VeryfrontCloudCatalog): ServedCatalogIndex { + const providerAliases = new Map(); + const surfaces = new Map(); + const operationsKnown = new Set(); + const servesResponses = new Set(); + const byKey = new Map(); + const byShortId = new Map(); + const byModelId = new Map(); + + for (const model of catalog.models) { + const slashIndex = model.modelId.indexOf("/"); + if (slashIndex <= 0) continue; + const segment = model.modelId.slice(0, slashIndex); + if (!providerAliases.has(segment)) providerAliases.set(segment, model.provider); + if (!providerAliases.has(model.provider)) providerAliases.set(model.provider, model.provider); + if (model.surface !== undefined && !surfaces.has(model.provider)) { + surfaces.set(model.provider, model.surface); + } + if (model.operations !== undefined) { + operationsKnown.add(model.provider); + if (model.operations.includes("responses")) servesResponses.add(model.provider); + } + const key = `${model.provider}/${modelSegment(model.modelId)}`; + if (!byKey.has(key)) byKey.set(key, model); + if (!byModelId.has(model.modelId)) byModelId.set(model.modelId, model); + for (const shortId of [model.id, ...model.aliases]) { + if (!shortId.includes("/") && !byShortId.has(shortId)) byShortId.set(shortId, model); + } + } + + const routing = new Map>(); + for (const [provider, surface] of surfaces) { + // A provider is native to the OpenAI surface when the platform serves the + // Responses operation for one of its models. An API that serves no + // operations yet leaves the protocol-named rule in place. + const native = surface !== "openai" + ? true + : operationsKnown.has(provider) + ? servesResponses.has(provider) + : PROTOCOL_NAMED_PROVIDERS.has(provider); + routing.set(provider, Object.freeze({ surface, native })); + } + + return { providerAliases, routing, byKey, byShortId, byModelId }; +} + +function servedIndex(): ServedCatalogIndex | undefined { + const catalog = peekVeryfrontCloudCatalog(); + if (!catalog) return undefined; + let index = servedIndexes.get(catalog); + if (!index) { + index = buildServedIndex(catalog); + servedIndexes.set(catalog, index); + } + return index; +} + +/** + * Canonical provider for a provider segment the served catalog spells, for + * example `google-ai-studio` for `google`. Undefined before the catalog is + * loaded, or when the catalog does not name the segment. + */ export function normalizeVeryfrontCloudProviderAlias( provider: string, -): KnownVeryfrontCloudProviderId | undefined { - return _providerAliasMap.get(provider); +): VeryfrontCloudProviderId | undefined { + return servedIndex()?.providerAliases.get(provider); } /** @@ -147,18 +276,33 @@ export function resolveVeryfrontCloudProviderId( : undefined; } -/** Routing used for a provider the catalog data does not list. */ +/** Routing used for a provider the served catalog does not describe. */ const DEFAULT_PROVIDER_ROUTING: Readonly = Object .freeze({ surface: DEFAULT_VERYFRONT_CLOUD_SURFACE, }); -/** Gateway routing declared for a provider, or the default for an unlisted one. */ +/** Routing of a provider named after its protocol, used until the catalog describes it. */ +const PROTOCOL_NAMED_ROUTING: ReadonlyMap> = + new Map( + [...PROTOCOL_NAMED_PROVIDERS].map(( + protocol, + ) => [protocol, Object.freeze({ surface: protocol, native: true })]), + ); + +/** + * Gateway routing for a provider, as the served catalog describes it. Before + * the catalog is loaded, and for a provider it does not describe, a provider + * named after a protocol speaks that protocol natively and any other provider + * speaks the default surface. + */ export function resolveVeryfrontCloudProviderRouting( provider: string, ): Readonly { const canonical = normalizeVeryfrontCloudProviderAlias(provider) ?? provider; - return _providerRoutingMap.get(canonical) ?? DEFAULT_PROVIDER_ROUTING; + return servedIndex()?.routing.get(canonical) ?? + PROTOCOL_NAMED_ROUTING.get(canonical) ?? + DEFAULT_PROVIDER_ROUTING; } /** Wire format the given provider's gateway endpoint speaks. */ @@ -200,10 +344,10 @@ export function resolveVeryfrontCloudGatewayPath( ): string | undefined { const providerId = resolveVeryfrontCloudProviderId(provider); if (!providerId) return undefined; - const apiVersion = _surfaceGatewayApiVersionMap.get( + const apiVersion = VENDOR_GATEWAY_API_VERSIONS.get( resolveVeryfrontCloudSurface(providerId), - ) ?? DEFAULT_VERYFRONT_CLOUD_GATEWAY_API_VERSION; - return `${VERYFRONT_CLOUD_GATEWAY_PATH_PREFIX}/${providerId}/${apiVersion}`; + ) ?? DEFAULT_VENDOR_GATEWAY_API_VERSION; + return `${VENDOR_GATEWAY_PATH_PREFIX}/${providerId}/${apiVersion}`; } /** @@ -241,12 +385,39 @@ export function canonicalVeryfrontCloudModelKey(modelId: string): string { : `${provider}/${normalizedModelId.slice(slashIndex + 1)}`; } +/** + * The served model a model ID names: by provider-qualified ID in any provider + * spelling, or by short ID or bare alias. Undefined before the catalog is + * loaded, or when the catalog does not list the model. + */ +function findServedModel(modelId: string): VeryfrontCloudCatalogModel | undefined { + const index = servedIndex(); + if (!index) return undefined; + return index.byKey.get(canonicalVeryfrontCloudModelKey(modelId)) ?? + index.byShortId.get(normalizeVeryfrontCloudModelId(modelId)); +} + +function isOpenAITransport(value: string | undefined): value is "chat-completions" | "responses" { + return value === "chat-completions" || value === "responses"; +} + function getVeryfrontCloudModelTransportCapabilities( modelId: string, ): Readonly | undefined { - return _transportCapabilitiesMap.get( - canonicalVeryfrontCloudModelKey(modelId), - ); + const model = findServedModel(modelId); + if (!model) return undefined; + return { + ...(model.surface === "anthropic" && model.reasoningMode === "adaptive" + ? { anthropicThinkingMode: "adaptive" as const } + : {}), + ...(isOpenAITransport(model.transport) ? { openAITransport: model.transport } : {}), + ...(model.chatCompletionsReasoningWithFunctionTools === undefined ? {} : { + openAIChatReasoningWithFunctionTools: model.chatCompletionsReasoningWithFunctionTools, + }), + ...(model.chatCompletionsConsecutiveSystemMessages === undefined ? {} : { + openAIChatPreserveSystemMessages: model.chatCompletionsConsecutiveSystemMessages, + }), + }; } /** Resolves a model-specific OpenAI transport override for Veryfront Cloud. */ @@ -324,6 +495,12 @@ export function resolveVeryfrontCloudOpenAITransportPlan( if (declared !== undefined) { return declared === "responses" ? RESPONSES_PINNED : CHAT_COMPLETIONS_PINNED; } + // A model the platform does not serve on Responses keeps to chat completions, + // even when its provider serves Responses for other models. + const operations = findServedModel(catalogModelId)?.operations; + if (operations !== undefined && !operations.includes("responses")) { + return CHAT_COMPLETIONS_PINNED; + } if (resolveVeryfrontCloudModelThinking(catalogModelId)?.enabled === true) { return RESPONSES_PINNED; } @@ -349,13 +526,6 @@ export function resolveVeryfrontCloudOpenAICallTransport( return usesHostedTool ? "responses" : "chat-completions"; } -/** - * Returns true if the given model ID is a Mistral model in the catalog. - * - * Compared by canonical key on both sides, so a catalog entry served under a - * provider alias and a request spelling the canonical provider (or carrying - * the gateway prefix) still meet. - */ /** * Whether an id is a Mistral id under any spelling the runtime accepts: the * gateway prefix stripped and the provider segment resolved through the alias @@ -365,52 +535,15 @@ function isMistralModelId(modelId: string): boolean { return canonicalVeryfrontCloudModelKey(modelId).startsWith("mistral/"); } +/** + * Whether a Mistral model ID is one the served catalog lists. Before the + * catalog is loaded every ID passes, and the platform refuses one it does not + * serve. + */ export function isSupportedMistralModelId(modelId: string): boolean { - const key = canonicalVeryfrontCloudModelKey(modelId); - return VERYFRONT_CLOUD_CHAT_MODELS.some( - (model) => - model.provider === "mistral" && - canonicalVeryfrontCloudModelKey(model.modelId) === key, - ); -} - -/** Shared Veryfront Cloud chat models value. */ -export const VERYFRONT_CLOUD_CHAT_MODELS: readonly VeryfrontCloudChatModel[] = Object.freeze( - VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => { - if ( - model.thinkingBudgetTokens !== undefined && - !isPositiveSafeInteger(model.thinkingBudgetTokens) - ) { - throw new TypeError( - `Veryfront Cloud model "${model.id}" thinkingBudgetTokens must be a positive safe integer`, - ); - } - return Object.freeze(model); - }), -); - -const defaultVeryfrontCloudChatModel = VERYFRONT_CLOUD_CHAT_MODELS.find( - (model) => model.id === DEFAULT_VERYFRONT_CLOUD_MODEL_ID, -); -if (!defaultVeryfrontCloudChatModel) { - throw new Error( - `Veryfront Cloud default model "${DEFAULT_VERYFRONT_CLOUD_MODEL_ID}" is missing from the catalog`, - ); -} - -/** Catalog-backed default model descriptor. */ -export const DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL = defaultVeryfrontCloudChatModel; -/** Canonical direct provider/model ID for the default chat model. */ -export const DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID = DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL.modelId; -/** Canonical hosted runtime ID for the default chat model. */ -export const DEFAULT_VERYFRONT_CLOUD_RUNTIME_MODEL_ID = - `${VERYFRONT_CLOUD_MODEL_PREFIX}${DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID}`; - -/** Find Veryfront Cloud model. */ -export function findVeryfrontCloudModel( - id: string, -): VeryfrontCloudChatModel | undefined { - return VERYFRONT_CLOUD_CHAT_MODELS.find((model) => model.id === id); + const index = servedIndex(); + if (!index) return true; + return index.byKey.get(canonicalVeryfrontCloudModelKey(modelId))?.provider === "mistral"; } /** Normalizes Veryfront Cloud model ID. */ @@ -420,19 +553,6 @@ export function normalizeVeryfrontCloudModelId(modelId: string): string { : modelId; } -/** Find Veryfront Cloud model by model ID. */ -export function findVeryfrontCloudModelByModelId( - modelId: string, -): VeryfrontCloudChatModel | undefined { - // Compared by canonical key on both sides: the catalog may publish a model - // under a provider alias while a caller spells the canonical provider, or - // the other way round once the catalog moves on and the alias is retained. - const key = canonicalVeryfrontCloudModelKey(modelId); - return VERYFRONT_CLOUD_CHAT_MODELS.find( - (model) => canonicalVeryfrontCloudModelKey(model.modelId) === key, - ); -} - /** * Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. * @@ -465,12 +585,17 @@ export function tryGetVeryfrontCloudProviderFromModelId( } } -/** Resolves Veryfront Cloud model ID. */ +/** + * Resolve a model ID or short alias to a provider-qualified model ID. + * + * No value resolves to the default model. A provider-qualified ID is returned + * as written. A short ID or alias resolves through the served catalog, so it + * resolves only once the catalog is loaded. + */ export function resolveVeryfrontCloudModelId(alias?: string): string { - const requestedModel = alias || DEFAULT_VERYFRONT_CLOUD_MODEL_ID; - const catalogModel = VERYFRONT_CLOUD_CHAT_MODELS.find((model) => - model.modelId === requestedModel - ); + const requestedModel = alias || resolveVeryfrontCloudDefaultModelId(); + const index = servedIndex(); + const catalogModel = index?.byModelId.get(requestedModel); if (catalogModel) { return catalogModel.modelId; } @@ -489,7 +614,7 @@ export function resolveVeryfrontCloudModelId(alias?: string): string { return requestedModel; } - const model = findVeryfrontCloudModel(requestedModel); + const model = index?.byShortId.get(requestedModel); if (!model) { throw INVALID_ARGUMENT.create({ detail: `Unknown model alias "${requestedModel}"`, @@ -552,9 +677,8 @@ export function resolveVeryfrontCloudModelThinking( return undefined; } - const model = findVeryfrontCloudModelByModelId(modelId) ?? - findVeryfrontCloudModel(modelId); - const budgetTokens = requireThinkingBudgetTokens(model?.thinkingBudgetTokens); + const model = findServedModel(modelId); + const budgetTokens = requireThinkingBudgetTokens(model?.reasoningBudgetTokens); if (model?.thinking !== true && budgetTokens === undefined) { return undefined; } @@ -643,21 +767,6 @@ export function resolveVeryfrontCloudThinkingProviderOptions( }; } -/** Group Veryfront Cloud models by provider. */ -export function groupVeryfrontCloudModelsByProvider(): Array<{ - readonly provider: KnownVeryfrontCloudProviderId; - readonly label: string; - readonly models: readonly VeryfrontCloudChatModel[]; -}> { - return PROVIDER_ORDER.map((provider) => ({ - provider, - label: PROVIDER_LABELS[provider], - models: Object.freeze( - VERYFRONT_CLOUD_CHAT_MODELS.filter((model) => model.provider === provider), - ), - })).filter((group) => group.models.length > 0); -} - /** * Prefix a model ID for a hosted run. Alias of * {@link resolveVeryfrontCloudGatewayModelId}, with the same contract: call it diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index 8eb913b3c1..756b8e4c44 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -1,7 +1,9 @@ import "#veryfront/schemas/_test-setup.ts"; import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { assertEquals, assertThrows } from "#veryfront/testing/assert.ts"; -import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests, servedCatalogPayload } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { agent } from "#veryfront/agent"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; import { clearEmbeddingProviders, resolveEmbeddingModel } from "#veryfront/embedding/index.ts"; @@ -9,7 +11,12 @@ import { ensureBuiltinLLMProviders } from "#veryfront/extensions/builtin-extensi import { clearModelProviders, resolveModel } from "#veryfront/provider"; import type { ModelRuntime } from "#veryfront/provider/types.ts"; import { getVeryfrontCloudAuthToken } from "#veryfront/platform/cloud/resolver.ts"; -import { createVeryfrontCloudInferenceModel, createVeryfrontCloudModel } from "./provider.ts"; +import { + createVeryfrontCloudInferenceModel, + createVeryfrontCloudModel, + warmVeryfrontCloudCatalog, +} from "./provider.ts"; +import { resolveVeryfrontCloudModelThinking } from "./model-catalog.ts"; import { createVeryfrontCloudFetch, getVeryfrontCloudGatewayBaseUrl, @@ -113,6 +120,8 @@ function setCloudBootstrap(): void { } describe("provider/veryfront-cloud", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { restoreMockFetch(); clearCloudEnv(); @@ -1383,6 +1392,8 @@ describe("provider/veryfront-cloud", () => { }); describe("provider/veryfront-cloud vendor-neutral routes", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { restoreMockFetch(); clearCloudEnv(); @@ -1751,3 +1762,119 @@ describe("provider/veryfront-cloud vendor-neutral routes", () => { assertEquals(sent, ['{"input":"x"}', "[1,2]", "not json", '{"model":7}']); }); }); + +describe("provider/veryfront-cloud served catalog loading", () => { + beforeEach(__resetVeryfrontCloudCatalogForTests); + afterEach(() => { + __resetVeryfrontCloudCatalogForTests(); + restoreMockFetch(); + clearCloudEnv(); + clearModelProviders(); + }); + + type CapturedRequest = { + method: string; + url: string; + authorization: string | null; + projectSlug: string | null; + }; + + /** Answer the catalog request from `catalog` and every other request with a finished chat stream. */ + function installGateway(catalog: () => Response): CapturedRequest[] { + const requests: CapturedRequest[] = []; + const encoder = new TextEncoder(); + installMockFetch( + ((input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + requests.push({ + method: request.method, + url: request.url, + authorization: request.headers.get("authorization"), + projectSlug: request.headers.get("x-veryfront-project-slug"), + }); + if (request.url.endsWith("/ai/models")) return Promise.resolve(catalog()); + return Promise.resolve( + new Response( + readableStreamFrom([ + encoder.encode('data: {"choices":[{"delta":{"content":"Hi"}}]}\n\n'), + encoder.encode('data: {"choices":[{"finish_reason":"stop"}]}\n\n'), + encoder.encode("data: [DONE]\n\n"), + ]), + { status: 200, headers: { "content-type": "text/event-stream" } }, + ), + ); + }) as typeof fetch, + ); + return requests; + } + + async function streamOnce(model: ModelRuntime): Promise { + const result = await model.doStream({ + prompt: [{ role: "user", content: [{ type: "text", text: "Hi" }] }], + } as never); + await drainStream(result.stream); + } + + it("builds synchronously, then loads the catalog on the first call and follows it", async () => { + setCloudBootstrap(); + const requests = installGateway(() => Response.json(servedCatalogPayload())); + + // Before the catalog loads, a reasoning-style OpenAI id would use Responses. + const model = resolveModel("veryfront-cloud/openai/gpt-5.5") as ModelRuntime; + assertEquals(requests.length, 0); + await streamOnce(model); + await streamOnce(model); + + assertEquals(requests.map(({ method, url }) => `${method} ${url}`), [ + "GET https://api.veryfront.com/ai/models", + "POST https://api.veryfront.com/ai/v1/chat/completions", + "POST https://api.veryfront.com/ai/v1/chat/completions", + ]); + assertEquals(requests[0]?.authorization, "Bearer vf_test_provider"); + assertEquals(requests[0]?.projectSlug, "provider-test-project"); + }); + + it("loads the catalog in prepare, before the first call", async () => { + setCloudBootstrap(); + const requests = installGateway(() => Response.json(servedCatalogPayload())); + + const model = resolveModel("veryfront-cloud/mistral/mistral-small-2503") as ModelRuntime; + await model.prepare?.(); + + assertEquals(requests.map(({ url }) => url), ["https://api.veryfront.com/ai/models"]); + }); + + it("still calls the model when the catalog cannot be loaded", async () => { + setCloudBootstrap(); + const requests = installGateway(() => Response.json({ error: "unavailable" }, { status: 503 })); + + await streamOnce(resolveModel("veryfront-cloud/mistral/mistral-small-2503") as ModelRuntime); + + assertEquals(requests.map(({ method, url }) => `${method} ${url}`), [ + "GET https://api.veryfront.com/ai/models", + "POST https://api.veryfront.com/ai/v1/chat/completions", + ]); + }); + + it("loads the catalog with the ambient credentials before thinking defaults are read", async () => { + setCloudBootstrap(); + const requests = installGateway(() => Response.json(servedCatalogPayload())); + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), undefined); + + await warmVeryfrontCloudCatalog(); + + assertEquals(requests.map(({ url }) => url), ["https://api.veryfront.com/ai/models"]); + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), { + enabled: true, + budgetTokens: 2048, + }); + }); + + it("skips the ambient load without credentials", async () => { + const requests = installGateway(() => Response.json(servedCatalogPayload())); + + await warmVeryfrontCloudCatalog(); + + assertEquals(requests, []); + }); +}); diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index 47432ee3fd..d17cf14109 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -24,6 +24,7 @@ import { resolveVeryfrontCloudOpenAITransportPlan, resolveVeryfrontCloudProviderRouting, } from "./model-catalog.ts"; +import { isVeryfrontCloudCatalogFresh, loadVeryfrontCloudCatalog } from "./catalog-client.ts"; const IntrinsicReflectApply = Reflect.apply; const HostCrypto = globalThis.crypto; @@ -82,34 +83,84 @@ function wrapVeryfrontCloudModel( return wrapped; } +/** + * @internal Load the served catalog with the ambient Veryfront Cloud + * credentials, before model facts are read synchronously. Never throws: without + * credentials or a catalog, the readers fall back to protocol defaults. + */ +export async function warmVeryfrontCloudCatalog(abortSignal?: AbortSignal): Promise { + let bootstrap: ReturnType; + try { + bootstrap = requireVeryfrontCloudBootstrap(); + } catch { + return; + } + await loadVeryfrontCloudCatalog({ + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + ...(abortSignal ? { signal: abortSignal } : {}), + }); +} + +/** + * Wrap a built model so its first async step loads the served catalog. When + * the catalog changes how the model is built, the calls go to a model rebuilt + * from it; the metadata stays that of the model built at construction. Once + * the catalog is settled, calls go straight to the current model. + */ +function withServedCatalog( + model: ModelRuntime, + settled: () => ModelRuntime | undefined, + ready: (abortSignal?: AbortSignal) => Promise, +): ModelRuntime { + const readSignal = (options: unknown): AbortSignal | undefined => + options !== null && typeof options === "object" + ? (options as { abortSignal?: AbortSignal }).abortSignal + : undefined; + return ObjectCreate(model, { + prepare: { + value: async (abortSignal?: AbortSignal): Promise => { + const current = settled() ?? await ready(abortSignal); + if (current.prepare) await current.prepare(abortSignal); + }, + }, + doGenerate: { + value: (options: unknown) => { + const current = settled(); + if (current) return current.doGenerate(options); + return (async () => await (await ready(readSignal(options))).doGenerate(options))(); + }, + }, + doStream: { + value: (options: unknown) => { + const current = settled(); + if (current) return current.doStream(options); + return (async () => await (await ready(readSignal(options))).doStream(options))(); + }, + }, + }); +} + +type VeryfrontCloudModelOptions = { + apiBaseUrl?: string; + assertInferenceCredentialActive?: () => void; + credentialSource?: "application"; + providerSelection?: "first-party"; + assertCredentialActive?: () => void; +}; + function createVeryfrontCloudModelInternal( modelId: string, inferenceCredential?: string, - options: { - apiBaseUrl?: string; - assertInferenceCredentialActive?: () => void; - credentialSource?: "application"; - providerSelection?: "first-party"; - assertCredentialActive?: () => void; - } = {}, + options: VeryfrontCloudModelOptions = {}, ): ModelRuntime { - const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language"); + // Parsed here so a malformed ID fails at construction; the provider is + // resolved again at build time, once the served catalog may name its alias. + parseVeryfrontCloudModelId(modelId, "language"); const { apiBaseUrl, apiToken, projectSlug } = options.credentialSource === "application" ? requireApplicationBootstrap() : requireVeryfrontCloudBootstrap(inferenceCredential, options.apiBaseUrl); - // Builders keep the upstream model id; on a vendor-neutral route the fetch - // wrapper sends it as `/`. - const { baseURL, wireModelProvider } = resolveVeryfrontCloudGatewayRoute(apiBaseUrl, provider); - const fetch = createVeryfrontCloudFetch(apiToken, baseURL, projectSlug, { - inferenceCredential: inferenceCredential !== undefined, - ...(wireModelProvider ? { wireModelProvider } : {}), - ...(options.assertCredentialActive || options.assertInferenceCredentialActive - ? { - assertInferenceCredentialActive: options.assertCredentialActive ?? - options.assertInferenceCredentialActive, - } - : {}), - }); const usesHostPrivateCredential = inferenceCredential === undefined && options.credentialSource !== "application" && getHostSecret("VERYFRONT_API_TOKEN") === apiToken; const usesPrivateCredential = inferenceCredential !== undefined || usesHostPrivateCredential || @@ -125,6 +176,120 @@ function createVeryfrontCloudModelInternal( // credential therefore uses only first-party transports that project code // cannot replace; ordinary project credentials retain extension behavior. const registry = useFirstPartyTransport ? undefined : ensureBuiltinLLMProviders(); + + // The served facts a build reads. A model built before the catalog loaded is + // rebuilt at its first async step when these differ. + function buildFacts(): string { + const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language"); + const catalogModelId = `${provider}/${upstreamModelId}`; + const routing = resolveVeryfrontCloudProviderRouting(provider); + const plan = resolveVeryfrontCloudOpenAITransportPlan(provider, upstreamModelId); + return [ + provider, + routing.surface, + String(routing.native), + plan.transport, + String(plan.pinned), + String(resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogModelId)), + String(resolveVeryfrontCloudOpenAIChatSystemMessages(catalogModelId)), + ].join("\n"); + } + + const build = (): ModelRuntime => + buildVeryfrontCloudModel({ + modelId, + inferenceCredential, + options, + apiBaseUrl, + apiToken, + projectSlug, + providerCredential, + registry, + useFirstPartyTransport, + }); + let facts = buildFacts(); + const built = build(); + let current = built; + let preparing: Promise | undefined; + let isSettled = false; + const rebuildIfChanged = (): ModelRuntime => { + const next = buildFacts(); + if (next !== facts) { + facts = next; + current = build(); + } + isSettled = true; + return current; + }; + // A fresh cached catalog settles the model without waiting on anything. + const settled = (): ModelRuntime | undefined => { + if (isSettled) return current; + if (!isVeryfrontCloudCatalogFresh({ apiBaseUrl, ...(projectSlug ? { projectSlug } : {}) })) { + return undefined; + } + return rebuildIfChanged(); + }; + const prepare = async (abortSignal?: AbortSignal): Promise => { + await loadVeryfrontCloudCatalog({ + apiBaseUrl, + apiToken, + ...(projectSlug ? { projectSlug } : {}), + ...(abortSignal ? { signal: abortSignal } : {}), + }); + return rebuildIfChanged(); + }; + const ready = async (abortSignal?: AbortSignal): Promise => { + preparing ??= prepare(abortSignal); + try { + return await preparing; + } catch (error) { + // A failed build is retried on the next call rather than cached. + preparing = undefined; + throw error; + } + }; + return withServedCatalog(built, settled, ready); +} + +interface VeryfrontCloudModelBuild { + readonly modelId: string; + readonly inferenceCredential: string | undefined; + readonly options: VeryfrontCloudModelOptions; + readonly apiBaseUrl: string; + readonly apiToken: string; + readonly projectSlug: string | undefined; + readonly providerCredential: string; + readonly registry: ReturnType | undefined; + readonly useFirstPartyTransport: boolean; +} + +/** Build a model from the served facts as they stand now. */ +function buildVeryfrontCloudModel(build: VeryfrontCloudModelBuild): ModelRuntime { + const { + modelId, + inferenceCredential, + options, + apiBaseUrl, + apiToken, + projectSlug, + providerCredential, + registry, + useFirstPartyTransport, + } = build; + const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language"); + // Builders keep the upstream model id; on a vendor-neutral route the fetch + // wrapper sends it as `/`. + const { baseURL, wireModelProvider } = resolveVeryfrontCloudGatewayRoute(apiBaseUrl, provider); + const fetch = createVeryfrontCloudFetch(apiToken, baseURL, projectSlug, { + inferenceCredential: inferenceCredential !== undefined, + ...(wireModelProvider ? { wireModelProvider } : {}), + ...(options.assertCredentialActive || options.assertInferenceCredentialActive + ? { + assertInferenceCredentialActive: options.assertCredentialActive ?? + options.assertInferenceCredentialActive, + } + : {}), + }); const routing = resolveVeryfrontCloudProviderRouting(provider); // A provider that only speaks the OpenAI wire format is promised the chat diff --git a/src/provider/veryfront-cloud/shared.test.ts b/src/provider/veryfront-cloud/shared.test.ts index 63d0be4a2d..76079f35b4 100644 --- a/src/provider/veryfront-cloud/shared.test.ts +++ b/src/provider/veryfront-cloud/shared.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { isVeryfrontGatewayResponse } from "#veryfront/provider/runtime-loader/provider-http.ts"; import { @@ -15,6 +17,8 @@ import { } from "./shared.ts"; describe("provider/veryfront-cloud/shared", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("normalizes provider aliases when parsing model IDs", () => { assertEquals( parseVeryfrontCloudModelId("google-ai-studio/gemini-2.0-flash", "embedding"), diff --git a/src/runtime/model-call-context-request.test.ts b/src/runtime/model-call-context-request.test.ts index 10d3833f3f..16c92501ea 100644 --- a/src/runtime/model-call-context-request.test.ts +++ b/src/runtime/model-call-context-request.test.ts @@ -1,5 +1,7 @@ import { assertEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { createWarningCollector } from "#veryfront/provider/shared/index.ts"; import { @@ -34,6 +36,8 @@ const samplingFields = [ ] as const; describe("model call request projection", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("matches OpenAI-compatible Cloud controls including Kimi fixed sampling", () => { for ( const [modelProvider, modelId] of [["mistral", "mistral-large"], [ diff --git a/tests/integration/agent/hosted-application-model-resolver.test.ts b/tests/integration/agent/hosted-application-model-resolver.test.ts index f1dec2a23b..58899d37fd 100644 --- a/tests/integration/agent/hosted-application-model-resolver.test.ts +++ b/tests/integration/agent/hosted-application-model-resolver.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { withEnv } from "#veryfront/testing/deno-compat.ts"; import { @@ -74,6 +76,8 @@ async function drain(stream: ReadableStream) { } describe("hosted ordinary application model resolver", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("audits adaptive Cloud transport for generation and streaming on the same model", async () => { const resolver = createHostedApplicationModelResolver(resolverOptions()); try { diff --git a/tests/integration/agent/run-scoped-inference-credential.test.ts b/tests/integration/agent/run-scoped-inference-credential.test.ts index 393aef0dec..796a0cbc33 100644 --- a/tests/integration/agent/run-scoped-inference-credential.test.ts +++ b/tests/integration/agent/run-scoped-inference-credential.test.ts @@ -24,7 +24,9 @@ import { import { parseAgUiJsonBody } from "#veryfront/agent/ag-ui/request-shared.ts"; import { readBodyWithLimit } from "#veryfront/security/input-validation/limits.ts"; import { assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; -import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { installMockFetch, restoreMockFetch, @@ -73,6 +75,8 @@ function runtimeAgentInvocation(inferenceAuthToken: string): Record { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { restoreMockFetch(); clearModelProviders(); diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts new file mode 100644 index 0000000000..04bf2a5d49 --- /dev/null +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -0,0 +1,252 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogClockForTests, + loadVeryfrontCloudCatalog, + parseVeryfrontCloudCatalog, + peekVeryfrontCloudCatalog, + VERYFRONT_CLOUD_CATALOG_RETRY_MS, + VERYFRONT_CLOUD_CATALOG_TTL_MS, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { servedCatalogPayload } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; + +const API_BASE_URL = "https://api.veryfront.com"; +const LOAD = { apiBaseUrl: API_BASE_URL, apiToken: "vf_catalog_test", projectSlug: "catalog-test" }; + +function catalogPayload(defaultModelId: string): Record { + return { ...servedCatalogPayload(), defaultModelId }; +} + +function jsonResponse(body: unknown, status = 200): Response { + return new Response(JSON.stringify(body), { + status, + headers: { "content-type": "application/json" }, + }); +} + +/** A fetch stub that answers every request from `respond` and records each one. */ +function recordingFetch(respond: () => Response | Promise): { + fetch: typeof fetch; + requests: Request[]; +} { + const requests: Request[] = []; + return { + requests, + fetch: ((input: URL | Request | string, init?: RequestInit) => { + requests.push(new Request(input, init)); + return Promise.resolve(respond()); + }) as typeof fetch, + }; +} + +/** Let a background refresh finish: it settles within one macrotask on a stub fetch. */ +function settleBackgroundRefresh(): Promise { + return new Promise((resolve) => setTimeout(resolve, 0)); +} + +function useClock(start = 1_000_000): { advance(ms: number): void } { + let current = start; + __setVeryfrontCloudCatalogClockForTests(() => current); + return { + advance(ms: number) { + current += ms; + }, + }; +} + +describe("provider/veryfront-cloud/catalog-client", () => { + afterEach(__resetVeryfrontCloudCatalogForTests); + + describe("parseVeryfrontCloudCatalog", () => { + it("reads the served model facts and the default model", () => { + const catalog = parseVeryfrontCloudCatalog(servedCatalogPayload()); + const sonnet = catalog?.models.find((model) => model.id === "claude-sonnet-4-6"); + const gpt = catalog?.models.find((model) => model.id === "gpt-5.5"); + const mistral = catalog?.models.find((model) => model.id === "mistral-small-2503"); + + assertEquals(catalog?.defaultModelId, "mistral/mistral-small-2503"); + assertEquals(sonnet?.surface, "anthropic"); + assertEquals(sonnet?.operations, ["messages"]); + assertEquals(sonnet?.reasoningBudgetTokens, 2048); + assertEquals(sonnet?.aliases.includes("sonnet"), true); + assertEquals(gpt?.transport, "chat-completions"); + assertEquals(gpt?.chatCompletionsReasoningWithFunctionTools, false); + assertEquals(mistral?.chatCompletionsConsecutiveSystemMessages, true); + }); + + it("reads a field an older API does not serve as absent", () => { + const catalog = parseVeryfrontCloudCatalog({ + models: [{ + id: "gpt-x", + modelId: "openai/gpt-x", + provider: "openai", + surface: "openai", + capabilities: { thinking: true }, + }], + }); + + assertEquals(catalog?.defaultModelId, undefined); + assertEquals(catalog?.models.length, 1); + assertEquals(catalog?.models[0]?.operations, undefined); + assertEquals(catalog?.models[0]?.reasoningBudgetTokens, undefined); + assertEquals(catalog?.models[0]?.chatCompletionsConsecutiveSystemMessages, undefined); + assertEquals(catalog?.models[0]?.aliases, []); + }); + + it("skips malformed rows and ignores invalid field values", () => { + const catalog = parseVeryfrontCloudCatalog({ + models: [ + null, + { id: "no-model-id", provider: "openai" }, + { + id: "gpt-y", + modelId: "openai/gpt-y", + provider: "openai", + operations: ["responses", 7], + capabilities: { reasoning_budget_tokens: 1.5, transport: 3 }, + }, + ], + }); + + assertEquals(catalog?.models.map((model) => model.id), ["gpt-y"]); + assertEquals(catalog?.models[0]?.operations, ["responses"]); + assertEquals(catalog?.models[0]?.reasoningBudgetTokens, undefined); + assertEquals(catalog?.models[0]?.transport, undefined); + }); + + it("returns undefined for a payload without a model list", () => { + assertEquals(parseVeryfrontCloudCatalog({ error: "nope" }), undefined); + assertEquals(parseVeryfrontCloudCatalog("[]"), undefined); + }); + }); + + describe("loadVeryfrontCloudCatalog", () => { + it("is cold until the first load, then serves the loaded catalog synchronously", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + assertEquals(peekVeryfrontCloudCatalog(), undefined); + + const loaded = await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(LOAD)); + + assertEquals(loaded?.defaultModelId, "mistral/mistral-small-2503"); + assertEquals(peekVeryfrontCloudCatalog(), loaded); + assertEquals(stub.requests.length, 1); + const [request] = stub.requests; + assertEquals(request?.method, "GET"); + assertEquals(request?.url, `${API_BASE_URL}/ai/models`); + assertEquals(request?.headers.get("authorization"), "Bearer vf_catalog_test"); + assertEquals(request?.headers.get("x-veryfront-project-slug"), "catalog-test"); + }); + + it("shares one request between concurrent loads for the same key", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + const [first, second] = await withMockFetch( + stub.fetch, + () => Promise.all([loadVeryfrontCloudCatalog(LOAD), loadVeryfrontCloudCatalog(LOAD)]), + ); + + assertEquals(stub.requests.length, 1); + assertEquals(first, second); + }); + + it("keeps a separate entry per project", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + await loadVeryfrontCloudCatalog({ ...LOAD, projectSlug: "other-project" }); + await loadVeryfrontCloudCatalog(LOAD); + }); + + assertEquals( + stub.requests.map((request) => request.headers.get("x-veryfront-project-slug")), + ["catalog-test", "other-project"], + ); + }); + + it("serves a fresh entry from cache and refreshes a stale one in the background", async () => { + const clock = useClock(); + let defaultModelId = "mistral/mistral-small-2503"; + const stub = recordingFetch(() => jsonResponse(catalogPayload(defaultModelId))); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + clock.advance(VERYFRONT_CLOUD_CATALOG_TTL_MS - 1); + await loadVeryfrontCloudCatalog(LOAD); + assertEquals(stub.requests.length, 1); + + defaultModelId = "anthropic/claude-sonnet-4-6"; + clock.advance(1); + // Stale: the stale entry answers at once while one refresh runs. + const stale = await loadVeryfrontCloudCatalog(LOAD); + assertEquals(stale?.defaultModelId, "mistral/mistral-small-2503"); + + await settleBackgroundRefresh(); + assertEquals(stub.requests.length, 2); + const refreshed = await loadVeryfrontCloudCatalog(LOAD); + assertEquals(refreshed?.defaultModelId, "anthropic/claude-sonnet-4-6"); + assertEquals(peekVeryfrontCloudCatalog()?.defaultModelId, "anthropic/claude-sonnet-4-6"); + assertEquals(stub.requests.length, 2); + }); + }); + + it("resolves to undefined when the catalog cannot be loaded, and retries later", async () => { + const clock = useClock(); + let status = 503; + const stub = recordingFetch(() => + status === 200 ? jsonResponse(servedCatalogPayload()) : jsonResponse({}, status) + ); + + await withMockFetch(stub.fetch, async () => { + assertEquals(await loadVeryfrontCloudCatalog(LOAD), undefined); + assertEquals(peekVeryfrontCloudCatalog(), undefined); + + status = 200; + clock.advance(VERYFRONT_CLOUD_CATALOG_RETRY_MS - 1); + assertEquals(await loadVeryfrontCloudCatalog(LOAD), undefined); + assertEquals(stub.requests.length, 1); + + clock.advance(1); + assertEquals( + (await loadVeryfrontCloudCatalog(LOAD))?.defaultModelId, + "mistral/mistral-small-2503", + ); + assertEquals(stub.requests.length, 2); + }); + }); + + it("treats a body without a model list as a failed load", async () => { + const stub = recordingFetch(() => jsonResponse({ unexpected: true })); + + const loaded = await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(LOAD)); + + assertEquals(loaded, undefined); + assertEquals(peekVeryfrontCloudCatalog(), undefined); + }); + + it("keeps the stale catalog when a refresh fails", async () => { + const clock = useClock(); + let fail = false; + const stub = recordingFetch(() => { + if (fail) return Promise.reject(new TypeError("network down")); + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + const loaded = await loadVeryfrontCloudCatalog(LOAD); + fail = true; + clock.advance(VERYFRONT_CLOUD_CATALOG_TTL_MS); + await loadVeryfrontCloudCatalog(LOAD); + await settleBackgroundRefresh(); + const afterFailure = await loadVeryfrontCloudCatalog(LOAD); + + assertEquals(stub.requests.length, 2); + assertEquals(afterFailure, loaded); + assertEquals(peekVeryfrontCloudCatalog(), loaded); + }); + }); + }); +}); diff --git a/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts b/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts index 997123d959..e190a9ed73 100644 --- a/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts +++ b/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts @@ -1,5 +1,7 @@ import { assertEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { DEFAULT_VERYFRONT_CLOUD_MODEL_ID, findVeryfrontCloudModel, @@ -62,6 +64,8 @@ function runtimeAccepts(modelId: string): boolean { } describe("veryfront-cloud model id rule", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); for (const modelId of MODEL_IDS) { it(`agrees with the runtime about ${JSON.stringify(modelId)}`, () => { assertEquals( @@ -134,6 +138,8 @@ function representativePayload(): Record { * new way of producing one is caught here rather than found in the catalog. */ describe("veryfront-cloud catalog round trip", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("resolves every shipped entry back to itself", () => { assertEquals(VERYFRONT_CLOUD_CHAT_MODELS.length > 0, true); diff --git a/tests/integration/provider/veryfront-cloud-model-table-importers.test.ts b/tests/integration/provider/veryfront-cloud-model-table-importers.test.ts new file mode 100644 index 0000000000..3ec0c1ad2c --- /dev/null +++ b/tests/integration/provider/veryfront-cloud-model-table-importers.test.ts @@ -0,0 +1,36 @@ +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { describe, it } from "#veryfront/testing/bdd.ts"; + +const SHIPPED_TABLE_MODULE = "model-catalog.data.ts"; +const SHIM_MODULE = "src/provider/veryfront-cloud/model-catalog.deprecated.ts"; + +/** Every non-test source file under `src/`, relative to the repository root. */ +async function sourceFiles(root: URL, dir = "src"): Promise { + const files: string[] = []; + for await (const entry of Deno.readDir(new URL(`${dir}/`, root))) { + const path = `${dir}/${entry.name}`; + if (entry.isDirectory) files.push(...await sourceFiles(root, path)); + else if (/\.tsx?$/.test(entry.name) && !/\.test(-helpers)?\.tsx?$/.test(entry.name)) { + files.push(path); + } + } + return files; +} + +describe("veryfront-cloud shipped model table", () => { + it("is the only source module that imports the shipped table", async () => { + const root = new URL("../../../", import.meta.url); + const importers: string[] = []; + for (const path of await sourceFiles(root)) { + if (path.endsWith(`/${SHIPPED_TABLE_MODULE}`)) continue; + const text = await Deno.readTextFile(new URL(path, root)); + if ( + new RegExp(`^import(?! type)[^;]*["'][^"']*${SHIPPED_TABLE_MODULE}["']`, "m").test(text) + ) { + importers.push(path); + } + } + + assertEquals(importers, [SHIM_MODULE]); + }); +}); diff --git a/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts b/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts index 9e8eceaadf..3e62d89fae 100644 --- a/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts +++ b/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts @@ -1,6 +1,7 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals } from "#veryfront/testing/assert.ts"; import { afterEach, it } from "#veryfront/testing/bdd.ts"; +import { useServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; import { clearModelProviders } from "#veryfront/provider"; @@ -24,6 +25,7 @@ afterEach(() => { }); it("converts the sanitized issue 1834 context into a valid Mistral gateway ingress request", async () => { + using _catalog = useServedCatalogForTests(); setEnv("VERYFRONT_API_TOKEN", "vf_test_issue_1834"); setEnv("VERYFRONT_PROJECT_SLUG", "issue-1834-project"); let capturedUrl = ""; From 1f2c89285d69c925fbe074a1fd1fd8e62b78f3b4 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 03:45:05 +0200 Subject: [PATCH 02/17] fix(provider): scope served model facts per project and credential Review follow-up for the catalog client: - Cache and read the catalog per API base URL, project and credential fingerprint; a model reads its own scope, other readers the ambient one. - Settle a model only once a catalog was obtained; a failed or abandoned load retries on a later call. - A caller's abort signal or wait bound never cancels or fails the shared request. - Read the shipped list while no catalog has loaded for the scope, keep google-ai-studio as a protocol alias, and export loadVeryfrontCloudModelCatalog(). - Record the facts a built model calls with in the model-call context, and settle the model before recording. - Forward metadata to a rebuilt model; bound warm-up waits. - Add an optional non-reserving loadModelCatalog facade to executor preparation, awaited after the grant checks. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- CHANGELOG.md | 20 +- docs/api-reference/veryfront/provider.md | 41 ++-- src/agent/hosted/default-chat-runtime.ts | 4 + .../hosted/executor-runtime-prepare.test.ts | 109 ++++++++- src/agent/hosted/runtime-preparation-core.ts | 18 ++ src/platform/cloud/resolver.ts | 13 +- src/provider/index.ts | 1 + .../veryfront-cloud/catalog-client.ts | 158 +++++++++++--- .../model-catalog.deprecated.ts | 103 ++++++++- .../model-catalog.served.test.ts | 96 +++++++- src/provider/veryfront-cloud/model-catalog.ts | 120 ++++++++-- src/provider/veryfront-cloud/provider.test.ts | 190 +++++++++++++--- src/provider/veryfront-cloud/provider.ts | 206 ++++++++++++------ src/provider/veryfront-cloud/shared.ts | 28 +++ src/runtime/model-call-context-request.ts | 31 ++- src/runtime/runtime-bridge.ts | 9 + .../veryfront-cloud-catalog-client.test.ts | 74 ++++++- 17 files changed, 1013 insertions(+), 208 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a527cdfc3d..2790c242cf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -33,14 +33,18 @@ unchanged. - Model construction stays synchronous and makes no network call. The catalog loads on the first async step of a model (`prepare`, `doGenerate` or `doStream`), with the same credentials and project as inference, and is - cached for five minutes. -- When the catalog cannot be loaded, calls still go out. Each model then uses - its protocol defaults: a provider named `openai`, `anthropic` or `google` - speaks its own protocol, any other provider speaks the OpenAI protocol, and - no thinking defaults apply. A short alias such as `opus` resolves only once - the catalog is loaded. -- A Mistral model ID the catalog does not list is refused only once the - catalog is loaded. Before that, the platform answers for it. + cached for five minutes per API, project and credential. +- Until the catalog has loaded for the credentials in use, and whenever it + cannot be loaded, the facts shipped with this package apply, as in the + previous release. A model whose first load failed tries again on a later + call. +- A model keeps the facts it settled with for its lifetime. A catalog refreshed + later applies to models constructed after the refresh. +- `loadVeryfrontCloudModelCatalog()` loads the catalog for the Veryfront Cloud + credentials in effect, so synchronous helpers such as + `resolveVeryfrontCloudModelId("opus")` and + `resolveVeryfrontCloudModelThinking()` read served facts, including models + the platform added after this release. - `resolveVeryfrontCloudDefaultModelId()` returns the default model the catalog names, or the built-in default before it loads. `VeryfrontCloudModelId` types a model ID as `/`. diff --git a/docs/api-reference/veryfront/provider.md b/docs/api-reference/veryfront/provider.md index 12e7287967..5336335acf 100644 --- a/docs/api-reference/veryfront/provider.md +++ b/docs/api-reference/veryfront/provider.md @@ -71,26 +71,27 @@ Clear all registered model providers and reset lazy built-ins (for testing). ### Functions -| Name | Description | Source | -| ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | -| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `findVeryfrontCloudModel` | Find a shipped chat model by its short id. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | -| `findVeryfrontCloudModelByModelId` | Find a shipped chat model by its provider-qualified id, in any provider spelling. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | -| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `groupVeryfrontCloudModelsByProvider` | Group the shipped chat models by provider, in display order. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | -| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveVeryfrontCloudDefaultModelId` | Provider-qualified ID of the default model: the one the served catalog names once it is loaded, otherwise the built-in default. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelId` | Resolve a model ID or short alias to a provider-qualified model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ---------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | +| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `findVeryfrontCloudModel` | Find a shipped chat model by its short id. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `findVeryfrontCloudModelByModelId` | Find a shipped chat model by its provider-qualified id, in any provider spelling. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `groupVeryfrontCloudModelsByProvider` | Group the shipped chat models by provider, in display order. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `loadVeryfrontCloudModelCatalog` | Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud credentials and project in effect, so model facts read synchronously afterwards (thinking defaults, short aliases such as `opus`, the default model) come from it. Resolves to whether a catalog is available. Never throws: without credentials or a reachable catalog, the facts shipped with this package apply. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/shared.ts) | +| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveVeryfrontCloudDefaultModelId` | Provider-qualified ID of the default model: the one the served catalog names once it is loaded, otherwise the built-in default. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelId` | Resolve a model ID or short alias to a provider-qualified model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Types diff --git a/src/agent/hosted/default-chat-runtime.ts b/src/agent/hosted/default-chat-runtime.ts index 32cecd58f9..8ad2775a65 100644 --- a/src/agent/hosted/default-chat-runtime.ts +++ b/src/agent/hosted/default-chat-runtime.ts @@ -525,6 +525,9 @@ function runWithDefaultHostedRequestContext( ); } +/** Longest a hosted run waits for the served catalog before resolving a short alias. */ +const CATALOG_ALIAS_MAX_WAIT_MS = 3_000; + /** Create default hosted chat runtime. */ export async function createDefaultHostedChatRuntime( input: CreateDefaultHostedChatRuntimeOptions, @@ -540,6 +543,7 @@ export async function createDefaultHostedChatRuntime( apiBaseUrl: input.config.apiUrl, apiToken: input.options.authToken, ...(input.options.projectSlug ? { projectSlug: input.options.projectSlug } : {}), + maxWaitMs: CATALOG_ALIAS_MAX_WAIT_MS, }); } const modelId = resolveVeryfrontCloudModelId(input.options.model); diff --git a/src/agent/hosted/executor-runtime-prepare.test.ts b/src/agent/hosted/executor-runtime-prepare.test.ts index a80e1a30d9..13db959a58 100644 --- a/src/agent/hosted/executor-runtime-prepare.test.ts +++ b/src/agent/hosted/executor-runtime-prepare.test.ts @@ -3,7 +3,10 @@ import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/te import { PERMISSION_DENIED } from "#veryfront/errors"; import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; -import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; import type { ModelRuntime, ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { defineSchema } from "#veryfront/schemas/index.ts"; @@ -2236,6 +2239,110 @@ Synthetic source instructions.`, }); } + it("reads thinking defaults from the catalog the loadModelCatalog facade loads", async () => { + __setVeryfrontCloudCatalogForTests(undefined); + const selectedModel = "veryfront-cloud/anthropic/claude-sonnet-4-6"; + let captured: ModelRuntimeCallOptions | undefined; + let catalogLoads = 0; + const f = fixture({ + grant: { + ...grant, + defaultModelId: selectedModel, + models: new Map([[selectedModel, { maxOutputTokens: 8192, providerToolNames: [] }]]), + }, + facades: { + loadModelCatalog: () => { + catalogLoads++; + __setVeryfrontCloudCatalogForTests({ + models: [{ + id: "claude-sonnet-4-6", + modelId: "anthropic/claude-sonnet-4-6", + provider: "anthropic", + surface: "anthropic", + operations: ["messages"], + capabilities: { + thinking: true, + reasoning_mode: "budget", + reasoning_budget_tokens: 1024, + }, + }], + }); + return Promise.resolve(); + }, + resolveModelRuntime: () => ({ + ...model, + doStream: (options) => { + captured = options as ModelRuntimeCallOptions; + return finishStream(); + }, + }), + }, + }); + try { + await Array.fromAsync(await preparedStream(f)); + assert(captured); + assertEquals(catalogLoads, 1); + assertEquals(captured.reasoning, { enabled: true, budgetTokens: 1024 }); + assertEquals(captured.maxOutputTokens, 7168); + } finally { + await f.owner.close(); + } + }); + + it("calls loadModelCatalog only after every grant check passes", async () => { + let catalogLoads = 0; + const f = fixture({ + facades: { + loadModelCatalog: () => { + catalogLoads++; + return Promise.resolve(); + }, + }, + }); + try { + assertEquals( + await prepare( + f.owner, + { agentId: "coder", modelId: "veryfront-cloud/not-granted/x" } as JsonValue, + ), + { ok: false, code: "EXECUTOR_RUNTIME_NOT_GRANTED" }, + ); + assertEquals(catalogLoads, 0); + } finally { + await f.owner.close(); + } + }); + + it("prepares with the shipped facts when loadModelCatalog fails", async () => { + __setVeryfrontCloudCatalogForTests(undefined); + const selectedModel = "veryfront-cloud/anthropic/claude-sonnet-4-6"; + let captured: ModelRuntimeCallOptions | undefined; + const f = fixture({ + grant: { + ...grant, + defaultModelId: selectedModel, + models: new Map([[selectedModel, { maxOutputTokens: 8192, providerToolNames: [] }]]), + }, + facades: { + loadModelCatalog: () => Promise.reject(new Error("catalog unavailable")), + resolveModelRuntime: () => ({ + ...model, + doStream: (options) => { + captured = options as ModelRuntimeCallOptions; + return finishStream(); + }, + }), + }, + }); + try { + await Array.fromAsync(await preparedStream(f)); + assert(captured); + assertEquals(captured.reasoning, { enabled: true, budgetTokens: 2048 }); + } finally { + await f.owner.close(); + } + }); + it("reserves the catalog thinking budget when the request omits thinking and output limits", async () => { const selectedModel = "veryfront-cloud/anthropic/claude-sonnet-4-6"; let captured: ModelRuntimeCallOptions | undefined; diff --git a/src/agent/hosted/runtime-preparation-core.ts b/src/agent/hosted/runtime-preparation-core.ts index a695c6f9dc..75cbca12c7 100644 --- a/src/agent/hosted/runtime-preparation-core.ts +++ b/src/agent/hosted/runtime-preparation-core.ts @@ -167,6 +167,15 @@ export type ExecutorRuntimePreparationGrant = Omit Promise; hostTools: ReadonlyMap; remoteToolSources: ReadonlyMap; projectSteering?: { @@ -325,6 +334,7 @@ export function createRuntimePreparationCore(input: RuntimePreparationCoreOption const installedModelResolver = input.facades.resolveModelRuntime; const facades: ExecutorRuntimeFacades = { resolveModelRuntime: snapshotFacadeMethod(input.facades, "resolveModelRuntime"), + loadModelCatalog: snapshotFacadeMethod(input.facades, "loadModelCatalog"), cleanup: snapshotFacadeMethod(input.facades, "cleanup"), projectSteering: snapshotSteeringFacade(input.facades.projectSteering), latestConversationUserText: snapshotFacadeMethod(input.facades, "latestConversationUserText"), @@ -476,6 +486,14 @@ export function createRuntimePreparationCore(input: RuntimePreparationCoreOption (request.maxOutputTokens !== undefined && request.maxOutputTokens > modelGrant.maxOutputTokens) ) refuse("EXECUTOR_RUNTIME_NOT_GRANTED"); + if (facades.loadModelCatalog) { + try { + await observePrivatePromise(facades.loadModelCatalog(context.signal)); + } catch { + // The reads below fall back to the shipped facts. + } + assertActive(); + } const thinking = request.thinking ?? definition.thinking ?? resolveVeryfrontCloudModelThinking(modelId); let availableOutputTokens = modelGrant.maxOutputTokens; diff --git a/src/platform/cloud/resolver.ts b/src/platform/cloud/resolver.ts index be2be79644..9fdaef2951 100644 --- a/src/platform/cloud/resolver.ts +++ b/src/platform/cloud/resolver.ts @@ -233,11 +233,18 @@ export function isVeryfrontCloudEnabled(): boolean { /** * Default Veryfront Cloud model: `VERYFRONT_DEFAULT_MODEL` when set, otherwise - * the default the served catalog names once it is loaded, otherwise the - * built-in default. + * the default the served catalog names once it has loaded for the current + * credentials and project, otherwise the built-in default. */ export function getDefaultVeryfrontCloudModel(): string { - const served = peekVeryfrontCloudCatalog()?.defaultModelId; + const bootstrap = getVeryfrontCloudBootstrap(); + const served = (bootstrap.apiToken + ? peekVeryfrontCloudCatalog({ + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + }) + : peekVeryfrontCloudCatalog())?.defaultModelId; return normalizeCloudModelString( getHostEnv("VERYFRONT_DEFAULT_MODEL"), served?.includes("/") ? served : DEFAULT_VERYFRONT_CLOUD_MODEL, diff --git a/src/provider/index.ts b/src/provider/index.ts index 457039ab49..1b30894932 100644 --- a/src/provider/index.ts +++ b/src/provider/index.ts @@ -44,6 +44,7 @@ export { VERYFRONT_CLOUD_CHAT_MODELS, VERYFRONT_CLOUD_MODEL_PREFIX, } from "./veryfront-cloud/model-catalog.ts"; +export { loadVeryfrontCloudModelCatalog } from "./veryfront-cloud/shared.ts"; export type { VeryfrontCloudChatModel, VeryfrontCloudModelThinkingConfig, diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index 673d122164..2b44c51b01 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -7,9 +7,12 @@ * model call; every synchronous reader uses {@link peekVeryfrontCloudCatalog} * and degrades when nothing is loaded yet. * - * - Entries are cached per API base URL and project, because the served list - * is filtered by project policy. - * - Concurrent loads for one key share a single request. + * - Entries are cached per API base URL, project and credential, because the + * served list is filtered by the project the credential or header selects. + * A synchronous read names the same scope, so one project never reads + * another project's list. + * - Concurrent loads for one key share a single request. A caller's abort + * signal only stops that caller waiting; it never cancels the shared request. * - An entry is fresh for {@link VERYFRONT_CLOUD_CATALOG_TTL_MS}. A stale entry * is returned at once while one refresh runs in the background, and it is * kept when that refresh fails. @@ -25,6 +28,8 @@ export const VERYFRONT_CLOUD_CATALOG_TTL_MS = 5 * 60_000; export const VERYFRONT_CLOUD_CATALOG_RETRY_MS = 30_000; /** Upper bound on one catalog request. */ const VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS = 10_000; +/** Header naming the project a catalog request is scoped to. */ +const PROJECT_SLUG_HEADER = "x-veryfront-project-slug"; /** Path of the served catalog, relative to the API base URL. */ const VERYFRONT_CLOUD_CATALOG_PATH = "ai/models"; @@ -63,12 +68,19 @@ export interface VeryfrontCloudCatalog { readonly defaultModelId?: string; } -/** Credentials and scope a catalog load uses: the same ones inference uses. */ -export interface VeryfrontCloudCatalogLoadOptions { +/** Credentials and project a catalog is loaded and read for: the same ones inference uses. */ +export interface VeryfrontCloudCatalogScope { readonly apiBaseUrl: string; readonly apiToken: string; readonly projectSlug?: string; +} + +/** Options for one catalog load. */ +export interface VeryfrontCloudCatalogLoadOptions extends VeryfrontCloudCatalogScope { + /** Stops this caller waiting. The shared request keeps running for other callers. */ readonly signal?: AbortSignal; + /** Longest this caller waits for a request in flight before it goes on without it. */ + readonly maxWaitMs?: number; } interface CatalogEntry { @@ -79,7 +91,8 @@ interface CatalogEntry { const entries = new Map(); const inflight = new Map>(); const failedAt = new Map(); -let latest: VeryfrontCloudCatalog | undefined; +/** Key of the scope a synchronous read uses, while {@link withVeryfrontCloudCatalogScope} runs. */ +let activeKey: string | undefined; let seeded: VeryfrontCloudCatalog | undefined; let failureLogged = false; let now: () => number = Date.now; @@ -155,8 +168,25 @@ export function parseVeryfrontCloudCatalog(payload: unknown): VeryfrontCloudCata }); } -function cacheKey(apiBaseUrl: string, projectSlug: string | undefined): string { - return `${apiBaseUrl}\n${projectSlug ?? ""}`; +/** + * Non-reversible fingerprint of a credential, so entries for different + * credentials never share a key and the key never holds the credential. + */ +function credentialFingerprint(token: string): string { + let a = 0x811c9dc5; + let b = 0x01000193; + for (let index = 0; index < token.length; index++) { + const code = token.charCodeAt(index); + a = Math.imul(a ^ code, 0x01000193) >>> 0; + b = Math.imul(b ^ code, 0x85ebca6b) >>> 0; + } + return `${a.toString(16)}${b.toString(16)}`; +} + +function cacheKey(scope: VeryfrontCloudCatalogScope): string { + return `${scope.apiBaseUrl}\n${scope.projectSlug ?? ""}\n${ + credentialFingerprint(scope.apiToken) + }`; } function catalogUrl(apiBaseUrl: string): string { @@ -168,18 +198,18 @@ function catalogUrl(apiBaseUrl: string): string { } async function fetchCatalog( - options: VeryfrontCloudCatalogLoadOptions, + options: VeryfrontCloudCatalogScope, ): Promise { const headers = new Headers({ Accept: "application/json", Authorization: `Bearer ${options.apiToken}`, }); - if (options.projectSlug) headers.set("x-veryfront-project-slug", options.projectSlug); + if (options.projectSlug) headers.set(PROJECT_SLUG_HEADER, options.projectSlug); + // Only the internal timeout bounds the shared request: one caller giving up + // must not fail the load for every other caller on the same key. const timeout = new AbortController(); const timer = setTimeout(() => timeout.abort(), VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS); - const signal = options.signal - ? AbortSignal.any([options.signal, timeout.signal]) - : timeout.signal; + const signal = timeout.signal; try { const response = await createVeryfrontApiOriginBoundOutboundFetch(options.apiBaseUrl)( catalogUrl(options.apiBaseUrl), @@ -201,7 +231,7 @@ async function fetchCatalog( function refresh( key: string, - options: VeryfrontCloudCatalogLoadOptions, + options: VeryfrontCloudCatalogScope, ): Promise { const pending = inflight.get(key); if (pending) return pending; @@ -211,7 +241,6 @@ function refresh( if (started !== generation) return catalog; entries.set(key, { catalog, fetchedAt: now() }); failedAt.delete(key); - latest = catalog; failureLogged = false; return catalog; }, @@ -221,7 +250,7 @@ function refresh( if (!failureLogged) { failureLogged = true; logger.warn( - "Veryfront Cloud model catalog is unavailable; model facts fall back to protocol defaults", + "Veryfront Cloud model catalog is unavailable; model facts fall back to the built-in list", { error: error instanceof Error ? error.message : String(error) }, ); } @@ -234,16 +263,39 @@ function refresh( return request; } +/** Resolve with what `request` resolves to, or with `fallback` once this caller stops waiting. */ +function waitFor( + request: Promise, + fallback: VeryfrontCloudCatalog | undefined, + signal: AbortSignal | undefined, + maxWaitMs: number | undefined, +): Promise { + if (!signal && maxWaitMs === undefined) return request; + if (signal?.aborted) return Promise.resolve(fallback); + return new Promise((resolve) => { + let timer: ReturnType | undefined; + const finish = (value: VeryfrontCloudCatalog | undefined) => { + if (timer !== undefined) clearTimeout(timer); + signal?.removeEventListener("abort", onAbort); + resolve(value); + }; + const onAbort = () => finish(fallback); + signal?.addEventListener("abort", onAbort, { once: true }); + if (maxWaitMs !== undefined) timer = setTimeout(() => finish(fallback), maxWaitMs); + request.then(finish); + }); +} + /** - * Load the served catalog for an API base URL and project. Resolves to the - * cached catalog when it is fresh, to a stale one while a refresh runs, and to - * undefined when no catalog could be loaded. Never rejects. + * Load the served catalog for a scope. Resolves to the cached catalog when it + * is fresh, to a stale one while a refresh runs, and to undefined when no + * catalog could be loaded or the caller stopped waiting first. Never rejects. */ export function loadVeryfrontCloudCatalog( options: VeryfrontCloudCatalogLoadOptions, ): Promise { if (seeded) return Promise.resolve(seeded); - const key = cacheKey(options.apiBaseUrl, options.projectSlug); + const key = cacheKey(options); const entry = entries.get(key); const current = now(); if (entry && current - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS) { @@ -253,23 +305,57 @@ export function loadVeryfrontCloudCatalog( if (lastFailure !== undefined && current - lastFailure < VERYFRONT_CLOUD_CATALOG_RETRY_MS) { return Promise.resolve(entry?.catalog); } - const request = refresh(key, options); + const request = refresh(key, { + apiBaseUrl: options.apiBaseUrl, + apiToken: options.apiToken, + ...(options.projectSlug ? { projectSlug: options.projectSlug } : {}), + }); // Stale while revalidate: the stale entry answers now, the refresh replaces it. - return entry ? Promise.resolve(entry.catalog) : request; + if (entry) return Promise.resolve(entry.catalog); + return waitFor(request, undefined, options.signal, options.maxWaitMs); } -/** Whether a load for these credentials would answer from a fresh cache entry, without a request. */ -export function isVeryfrontCloudCatalogFresh( - options: Pick, -): boolean { +/** Whether a load for this scope would answer from a fresh cache entry, without a request. */ +export function isVeryfrontCloudCatalogFresh(scope: VeryfrontCloudCatalogScope): boolean { if (seeded) return true; - const entry = entries.get(cacheKey(options.apiBaseUrl, options.projectSlug)); + const entry = entries.get(cacheKey(scope)); return entry !== undefined && now() - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS; } -/** The most recently loaded catalog, stale or not, or undefined before any load. */ -export function peekVeryfrontCloudCatalog(): VeryfrontCloudCatalog | undefined { - return seeded ?? latest; +/** + * Run `fn` synchronously with {@link peekVeryfrontCloudCatalog} reading the + * catalog loaded for `scope`, so a model's facts come from its own project and + * credential whatever the ambient request carries. + */ +export function withVeryfrontCloudCatalogScope( + scope: VeryfrontCloudCatalogScope, + fn: () => T, +): T { + const previous = activeKey; + activeKey = cacheKey(scope); + try { + return fn(); + } finally { + activeKey = previous; + } +} + +/** Whether {@link withVeryfrontCloudCatalogScope} names the scope reads use right now. */ +export function hasActiveVeryfrontCloudCatalogScope(): boolean { + return activeKey !== undefined; +} + +/** + * The catalog loaded for a scope, stale or not, or undefined before any load + * for it. Without a scope, reads the one {@link withVeryfrontCloudCatalogScope} + * names, and undefined outside it. + */ +export function peekVeryfrontCloudCatalog( + scope?: VeryfrontCloudCatalogScope, +): VeryfrontCloudCatalog | undefined { + if (seeded) return seeded; + const key = scope ? cacheKey(scope) : activeKey; + return key === undefined ? undefined : entries.get(key)?.catalog; } /** @internal Serve a fixed catalog for every key, as if freshly loaded. `undefined` clears it. */ @@ -277,13 +363,23 @@ export function __setVeryfrontCloudCatalogForTests(payload: unknown): void { seeded = payload === undefined ? undefined : parseVeryfrontCloudCatalog(payload); } +/** @internal Store a catalog for one scope, as if it had just loaded for it. */ +export function __setVeryfrontCloudCatalogForScopeForTests( + scope: VeryfrontCloudCatalogScope, + payload: unknown, +): void { + const catalog = parseVeryfrontCloudCatalog(payload); + if (!catalog) throw new TypeError("Test catalog payload has no model list"); + entries.set(cacheKey(scope), { catalog, fetchedAt: now() }); +} + /** @internal Forget every loaded catalog, pending load and failure. */ export function __resetVeryfrontCloudCatalogForTests(): void { generation++; entries.clear(); inflight.clear(); failedAt.clear(); - latest = undefined; + activeKey = undefined; seeded = undefined; failureLogged = false; now = Date.now; diff --git a/src/provider/veryfront-cloud/model-catalog.deprecated.ts b/src/provider/veryfront-cloud/model-catalog.deprecated.ts index c4d96216a2..b5fc8fd5d0 100644 --- a/src/provider/veryfront-cloud/model-catalog.deprecated.ts +++ b/src/provider/veryfront-cloud/model-catalog.deprecated.ts @@ -1,18 +1,23 @@ /** * Deprecated model list exports, backed by the table shipped in this package. * - * Model facts now come from the served catalog (`catalog-client.ts`), and no - * resolution logic reads this module. It keeps the public exports that only - * make sense with a shipped list working for one release, and it is the only - * module that imports the shipped table. + * Model facts now come from the served catalog (`catalog-client.ts`). This + * module keeps the public exports that only make sense with a shipped list + * working for one release, and it is the only module that imports the shipped + * table. While it ships, the resolvers also read {@link SHIPPED_VERYFRONT_CLOUD_CATALOG} + * for a scope whose catalog has not loaded, so a process that has not reached + * `/ai/models` keeps the behaviour of the previous release. */ import type { KnownVeryfrontCloudProviderId, VeryfrontCloudChatModel } from "./model-catalog.ts"; +import type { VeryfrontCloudCatalog, VeryfrontCloudCatalogModel } from "./catalog-client.ts"; import { DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, + VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, VERYFRONT_CLOUD_PROVIDER_ALIASES, VERYFRONT_CLOUD_PROVIDER_LABELS as PROVIDER_LABELS, VERYFRONT_CLOUD_PROVIDER_ORDER as PROVIDER_ORDER, + VERYFRONT_CLOUD_PROVIDER_ROUTING, } from "./model-catalog.data.ts"; const MODEL_PREFIX = "veryfront-cloud/"; @@ -109,3 +114,93 @@ export function groupVeryfrontCloudModelsByProvider(): Array<{ ), })).filter((group) => group.models.length > 0); } + +const providerRouting = new Map(VERYFRONT_CLOUD_PROVIDER_ROUTING); +const transportCapabilities = new Map(VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES); + +/** The operations the shipped table implies for a model, in the served catalog's terms. */ +function shippedOperations(provider: string, key: string): readonly string[] | undefined { + const routing = providerRouting.get(provider); + switch (routing?.surface) { + case "anthropic": + return ["messages"]; + case "google": + return ["generate-content", "stream-generate-content"]; + case "openai": + return routing.native === true && + transportCapabilities.get(key)?.openAITransport !== "chat-completions" + ? ["responses", "chat-completions"] + : ["chat-completions"]; + default: + return undefined; + } +} + +function shippedModel( + id: string, + modelId: string, + provider: string, + thinking: boolean | undefined, + budget: number | undefined, +): VeryfrontCloudCatalogModel { + const key = tableModelKey(modelId); + const capabilities = transportCapabilities.get(key); + const surface = providerRouting.get(provider)?.surface; + const operations = shippedOperations(provider, key); + return Object.freeze({ + id, + modelId, + provider, + aliases: Object.freeze([]), + ...(surface === undefined ? {} : { surface }), + ...(operations === undefined ? {} : { operations: Object.freeze([...operations]) }), + ...(thinking === undefined ? {} : { thinking }), + ...(capabilities?.anthropicThinkingMode + ? { reasoningMode: capabilities.anthropicThinkingMode } + : {}), + ...(capabilities?.openAITransport ? { transport: capabilities.openAITransport } : {}), + ...(budget === undefined ? {} : { reasoningBudgetTokens: budget }), + ...(capabilities?.openAIChatReasoningWithFunctionTools === undefined ? {} : { + chatCompletionsReasoningWithFunctionTools: capabilities.openAIChatReasoningWithFunctionTools, + }), + ...(capabilities?.openAIChatPreserveSystemMessages === undefined ? {} : { + chatCompletionsConsecutiveSystemMessages: capabilities.openAIChatPreserveSystemMessages, + }), + }); +} + +/** + * The shipped table in the served catalog's shape. The resolvers read it for a + * scope whose catalog has not loaded. A later release removes it with the table. + * + * @deprecated Internal fallback that is removed with the shipped table. + */ +export const SHIPPED_VERYFRONT_CLOUD_CATALOG: VeryfrontCloudCatalog = Object.freeze({ + models: Object.freeze([ + ...VERYFRONT_CLOUD_CHAT_MODELS.map((model) => + shippedModel( + model.id, + model.modelId, + model.provider, + model.thinking === true || model.thinkingBudgetTokens !== undefined ? true : undefined, + model.thinkingBudgetTokens, + ) + ), + // Transport rows for models without a chat entry keep their facts too. + ...VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES + .filter(([key]) => + !VERYFRONT_CLOUD_CHAT_MODELS.some((model) => tableModelKey(model.modelId) === key) + ) + .map(([key]) => { + const slashIndex = key.indexOf("/"); + return shippedModel( + key.slice(slashIndex + 1), + key, + key.slice(0, slashIndex), + undefined, + undefined, + ); + }), + ]), + defaultModelId: DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL.modelId, +}); diff --git a/src/provider/veryfront-cloud/model-catalog.served.test.ts b/src/provider/veryfront-cloud/model-catalog.served.test.ts index 22b4a6195e..56b3b3255a 100644 --- a/src/provider/veryfront-cloud/model-catalog.served.test.ts +++ b/src/provider/veryfront-cloud/model-catalog.served.test.ts @@ -3,7 +3,10 @@ import { assertEquals } from "#veryfront/testing/assert.ts"; import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; import { __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForScopeForTests, __setVeryfrontCloudCatalogForTests, + peekVeryfrontCloudCatalog, + withVeryfrontCloudCatalogScope, } from "./catalog-client.ts"; import { seedServedCatalogForTests, @@ -58,7 +61,7 @@ describe("provider/veryfront-cloud/model-catalog served facts", () => { afterEach(__resetVeryfrontCloudCatalogForTests); describe("before the catalog is loaded", () => { - it("routes providers named after a protocol natively and any other on the default surface", () => { + it("routes on the shipped list: protocol providers natively, Mistral on the OpenAI protocol", () => { assertEquals(resolveVeryfrontCloudProviderRouting("openai"), { surface: "openai", native: true, @@ -71,23 +74,100 @@ describe("provider/veryfront-cloud/model-catalog served facts", () => { surface: "google", native: true, }); - assertEquals(resolveVeryfrontCloudProviderRouting("mistral"), { surface: "openai" }); + assertEquals(resolveVeryfrontCloudProviderRouting("mistral").surface, "openai"); + assertEquals(resolveVeryfrontCloudProviderRouting("mistral").native, false); + assertEquals(resolveVeryfrontCloudProviderRouting("acme-labs"), { surface: "openai" }); }); - it("declares no model facts and uses the built-in default model", () => { - assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), undefined); - assertEquals(resolveVeryfrontCloudOpenAITransport("openai/gpt-5.5"), undefined); + it("routes google-ai-studio as Google", () => { + assertEquals(resolveVeryfrontCloudProviderId("google-ai-studio"), "google"); + assertEquals(resolveVeryfrontCloudProviderRouting("google-ai-studio"), { + surface: "google", + native: true, + }); + }); + + it("reads the facts shipped with this package, so behaviour matches the previous release", () => { + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), { + enabled: true, + budgetTokens: 2048, + }); + assertEquals(resolveVeryfrontCloudOpenAITransport("openai/gpt-5.5"), "chat-completions"); assertEquals( resolveVeryfrontCloudOpenAIChatSystemMessages("mistral/mistral-small-2503"), - undefined, + true, ); + assertEquals(resolveVeryfrontCloudModelId("opus"), "anthropic/claude-opus-4-8"); assertEquals( resolveVeryfrontCloudDefaultModelId(), DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, ); assertEquals(resolveVeryfrontCloudModelId(), DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID); - // The platform refuses a model it does not serve; nothing is refused locally. - assertEquals(isSupportedMistralModelId("mistral/not-served"), true); + assertEquals(isSupportedMistralModelId("mistral/mistral-small-2503"), true); + assertEquals(isSupportedMistralModelId("mistral/not-listed"), false); + }); + }); + + describe("scope", () => { + const projectA = { + apiBaseUrl: "https://api.example.test", + apiToken: "token-a", + projectSlug: "a", + }; + const projectB = { + apiBaseUrl: "https://api.example.test", + apiToken: "token-b", + projectSlug: "b", + }; + const sameProjectOtherToken = { ...projectA, apiToken: "token-c" }; + + it("reads each project's own catalog, never one another project loaded", () => { + __setVeryfrontCloudCatalogForScopeForTests( + projectA, + payload([ + row("openai/gpt-a", { surface: "openai", operations: ["chat-completions"] }), + ], "openai/gpt-a"), + ); + __setVeryfrontCloudCatalogForScopeForTests( + projectB, + payload([ + row("mistral/mistral-b", { surface: "openai", operations: ["chat-completions"] }), + ], "mistral/mistral-b"), + ); + + withVeryfrontCloudCatalogScope(projectA, () => { + assertEquals(isSupportedMistralModelId("mistral/mistral-b"), false); + assertEquals(resolveVeryfrontCloudDefaultModelId(), "openai/gpt-a"); + assertEquals(resolveVeryfrontCloudProviderRouting("openai").native, false); + }); + withVeryfrontCloudCatalogScope(projectB, () => { + assertEquals(isSupportedMistralModelId("mistral/mistral-b"), true); + assertEquals(resolveVeryfrontCloudDefaultModelId(), "mistral/mistral-b"); + }); + }); + + it("keeps a credential's catalog apart from another credential's for the same project", () => { + __setVeryfrontCloudCatalogForScopeForTests(projectA, payload([], "openai/gpt-a")); + + assertEquals(peekVeryfrontCloudCatalog(projectA)?.defaultModelId, "openai/gpt-a"); + assertEquals(peekVeryfrontCloudCatalog(sameProjectOtherToken), undefined); + withVeryfrontCloudCatalogScope(sameProjectOtherToken, () => { + // Nothing loaded for this credential: the shipped list applies. + assertEquals( + resolveVeryfrontCloudDefaultModelId(), + DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, + ); + assertEquals(resolveVeryfrontCloudModelId("opus"), "anthropic/claude-opus-4-8"); + }); + }); + + it("keeps google-ai-studio as Google when a loaded catalog lists no Google model", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-x", { surface: "openai", operations: ["chat-completions"] }), + ])); + + assertEquals(resolveVeryfrontCloudProviderId("google-ai-studio"), "google"); + assertEquals(resolveVeryfrontCloudProviderRouting("google-ai-studio").surface, "google"); }); }); diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index bb9826b874..1b9846e785 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -1,10 +1,15 @@ import { INVALID_ARGUMENT, NOT_SUPPORTED } from "#veryfront/errors"; import { isOpenAIReasoningModel } from "../shared/openai-reasoning.ts"; +import { getVeryfrontCloudBootstrap } from "#veryfront/platform/cloud/resolver.ts"; +import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; +import type { ModelRuntime } from "../types.ts"; import { + hasActiveVeryfrontCloudCatalogScope, peekVeryfrontCloudCatalog, type VeryfrontCloudCatalog, type VeryfrontCloudCatalogModel, } from "./catalog-client.ts"; +import { SHIPPED_VERYFRONT_CLOUD_CATALOG } from "./model-catalog.deprecated.ts"; export { DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL, @@ -132,7 +137,7 @@ export const DEFAULT_VERYFRONT_CLOUD_RUNTIME_MODEL_ID: VeryfrontCloudRuntimeMode * names once it is loaded, otherwise the built-in default. */ export function resolveVeryfrontCloudDefaultModelId(): VeryfrontCloudModelId { - const served = peekVeryfrontCloudCatalog()?.defaultModelId; + const served = loadedCatalog()?.defaultModelId; return served !== undefined && served.includes("/") ? served as VeryfrontCloudModelId : DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID; @@ -151,11 +156,18 @@ const DEFAULT_VENDOR_GATEWAY_API_VERSION = "v1"; /** Surface used for a provider the served catalog does not describe. */ const DEFAULT_VERYFRONT_CLOUD_SURFACE = "openai"; /** - * Providers named after the wire protocol they implement. When the served - * catalog does not describe a provider, one of these speaks its own protocol - * natively and any other provider speaks the default surface. + * Providers named after the wire protocol they implement. When no catalog + * describes a provider, one of these speaks its own protocol natively and any + * other provider speaks the default surface. */ const PROTOCOL_NAMED_PROVIDERS: ReadonlySet = new Set(["openai", "anthropic", "google"]); +/** + * Provider spellings that name a protocol-named provider. A protocol fact, not + * a model fact: it holds whether or not a catalog has loaded. + */ +const PROTOCOL_PROVIDER_ALIASES: ReadonlyMap = new Map([ + ["google-ai-studio", "google"], +]); /** Lookups built once per loaded catalog. */ interface ServedCatalogIndex { @@ -223,9 +235,41 @@ function buildServedIndex(catalog: VeryfrontCloudCatalog): ServedCatalogIndex { return { providerAliases, routing, byKey, byShortId, byModelId }; } -function servedIndex(): ServedCatalogIndex | undefined { - const catalog = peekVeryfrontCloudCatalog(); - if (!catalog) return undefined; +/** + * The scope synchronous reads use: the one a model build names, otherwise the + * ambient Veryfront Cloud credentials. Undefined without credentials. + */ +function ambientScope(): + | { apiBaseUrl: string; apiToken: string; projectSlug?: string } + | undefined { + let bootstrap: ReturnType; + try { + bootstrap = getVeryfrontCloudBootstrap(); + } catch { + return undefined; + } + if (!bootstrap.apiToken || !bootstrap.apiBaseUrl) return undefined; + return { + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + }; +} + +/** The served catalog loaded for the scope reads use, or undefined before it loads. */ +function loadedCatalog(): VeryfrontCloudCatalog | undefined { + if (hasActiveVeryfrontCloudCatalogScope()) return peekVeryfrontCloudCatalog(); + const scope = ambientScope(); + // A scope-less read still sees a catalog fixed by a test hook. + return scope ? peekVeryfrontCloudCatalog(scope) : peekVeryfrontCloudCatalog(); +} + +/** + * The index reads use: the served catalog loaded for the current scope, or the + * shipped list while none has loaded for it. + */ +function servedIndex(): ServedCatalogIndex { + const catalog = loadedCatalog() ?? SHIPPED_VERYFRONT_CLOUD_CATALOG; let index = servedIndexes.get(catalog); if (!index) { index = buildServedIndex(catalog); @@ -235,14 +279,14 @@ function servedIndex(): ServedCatalogIndex | undefined { } /** - * Canonical provider for a provider segment the served catalog spells, for - * example `google-ai-studio` for `google`. Undefined before the catalog is - * loaded, or when the catalog does not name the segment. + * Canonical provider for a provider segment the catalog spells, for example + * `google-ai-studio` for `google`. Undefined when neither the catalog nor the + * protocol aliases name the segment. */ export function normalizeVeryfrontCloudProviderAlias( provider: string, ): VeryfrontCloudProviderId | undefined { - return servedIndex()?.providerAliases.get(provider); + return servedIndex().providerAliases.get(provider) ?? PROTOCOL_PROVIDER_ALIASES.get(provider); } /** @@ -300,7 +344,7 @@ export function resolveVeryfrontCloudProviderRouting( provider: string, ): Readonly { const canonical = normalizeVeryfrontCloudProviderAlias(provider) ?? provider; - return servedIndex()?.routing.get(canonical) ?? + return servedIndex().routing.get(canonical) ?? PROTOCOL_NAMED_ROUTING.get(canonical) ?? DEFAULT_PROVIDER_ROUTING; } @@ -392,7 +436,6 @@ export function canonicalVeryfrontCloudModelKey(modelId: string): string { */ function findServedModel(modelId: string): VeryfrontCloudCatalogModel | undefined { const index = servedIndex(); - if (!index) return undefined; return index.byKey.get(canonicalVeryfrontCloudModelKey(modelId)) ?? index.byShortId.get(normalizeVeryfrontCloudModelId(modelId)); } @@ -512,6 +555,41 @@ export function resolveVeryfrontCloudOpenAITransportPlan( return CHAT_COMPLETIONS_ADAPTIVE; } +/** @internal The catalog facts one built Veryfront Cloud model was built with. */ +export interface VeryfrontCloudModelFacts { + readonly provider: string; + readonly surface: VeryfrontCloudSurfaceId; + readonly native: boolean; + readonly transportPlan: VeryfrontCloudOpenAITransportPlan; + readonly openAITransport?: "chat-completions" | "responses"; + readonly openAIChatReasoningWithFunctionTools?: boolean; + readonly openAIChatPreserveSystemMessages?: boolean; +} + +const builtModelFacts = createPrivateWeakStore VeryfrontCloudModelFacts>(); + +/** @internal Record where a built model's current facts are read from. */ +export function registerVeryfrontCloudModelFacts( + model: ModelRuntime, + read: () => VeryfrontCloudModelFacts, +): void { + builtModelFacts.set(model, read); +} + +/** + * @internal The facts a Veryfront Cloud model built by this package currently + * calls with, so a record of a call describes the request actually sent. + * Undefined for any other object. + */ +export function readVeryfrontCloudModelFacts( + model: unknown, +): VeryfrontCloudModelFacts | undefined { + if (model === null || (typeof model !== "object" && typeof model !== "function")) { + return undefined; + } + return builtModelFacts.get(model as ModelRuntime)?.(); +} + /** Transport one call uses, given whether that call carries a hosted tool. */ export function resolveVeryfrontCloudOpenAICallTransport( provider: string, @@ -536,13 +614,11 @@ function isMistralModelId(modelId: string): boolean { } /** - * Whether a Mistral model ID is one the served catalog lists. Before the - * catalog is loaded every ID passes, and the platform refuses one it does not - * serve. + * Whether a Mistral model ID is one the catalog lists: the served catalog once + * it has loaded for the current scope, otherwise the shipped list. */ export function isSupportedMistralModelId(modelId: string): boolean { const index = servedIndex(); - if (!index) return true; return index.byKey.get(canonicalVeryfrontCloudModelKey(modelId))?.provider === "mistral"; } @@ -589,13 +665,15 @@ export function tryGetVeryfrontCloudProviderFromModelId( * Resolve a model ID or short alias to a provider-qualified model ID. * * No value resolves to the default model. A provider-qualified ID is returned - * as written. A short ID or alias resolves through the served catalog, so it - * resolves only once the catalog is loaded. + * as written. A short ID or alias resolves through the served catalog, or + * through the shipped list before the catalog has loaded; use + * `loadVeryfrontCloudModelCatalog()` first to resolve an alias the platform + * added since this release. */ export function resolveVeryfrontCloudModelId(alias?: string): string { const requestedModel = alias || resolveVeryfrontCloudDefaultModelId(); const index = servedIndex(); - const catalogModel = index?.byModelId.get(requestedModel); + const catalogModel = index.byModelId.get(requestedModel); if (catalogModel) { return catalogModel.modelId; } @@ -614,7 +692,7 @@ export function resolveVeryfrontCloudModelId(alias?: string): string { return requestedModel; } - const model = index?.byShortId.get(requestedModel); + const model = index.byShortId.get(requestedModel); if (!model) { throw INVALID_ARGUMENT.create({ detail: `Unknown model alias "${requestedModel}"`, diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index 756b8e4c44..0c4e92acdc 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -3,7 +3,11 @@ import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetc import { assertEquals, assertThrows } from "#veryfront/testing/assert.ts"; import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; import { seedServedCatalogForTests, servedCatalogPayload } from "./catalog-client.test-helpers.ts"; -import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogClockForTests, + VERYFRONT_CLOUD_CATALOG_RETRY_MS, +} from "./catalog-client.ts"; import { agent } from "#veryfront/agent"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; import { clearEmbeddingProviders, resolveEmbeddingModel } from "#veryfront/embedding/index.ts"; @@ -16,7 +20,14 @@ import { createVeryfrontCloudModel, warmVeryfrontCloudCatalog, } from "./provider.ts"; -import { resolveVeryfrontCloudModelThinking } from "./model-catalog.ts"; +import { + readVeryfrontCloudModelFacts, + resolveVeryfrontCloudModelThinking, +} from "./model-catalog.ts"; +import { loadVeryfrontCloudModelCatalog } from "./shared.ts"; +import { generateText } from "#veryfront/runtime/runtime-bridge.ts"; +import { runWithMandatoryRunEventSink } from "#veryfront/runtime/run-event-sink-context.ts"; +import type { AgentRunEvent } from "#veryfront/runtime/model-call-context.ts"; import { createVeryfrontCloudFetch, getVeryfrontCloudGatewayBaseUrl, @@ -1780,7 +1791,7 @@ describe("provider/veryfront-cloud served catalog loading", () => { }; /** Answer the catalog request from `catalog` and every other request with a finished chat stream. */ - function installGateway(catalog: () => Response): CapturedRequest[] { + function installGateway(catalog: () => Response, catalogGate?: Promise): CapturedRequest[] { const requests: CapturedRequest[] = []; const encoder = new TextEncoder(); installMockFetch( @@ -1792,7 +1803,9 @@ describe("provider/veryfront-cloud served catalog loading", () => { authorization: request.headers.get("authorization"), projectSlug: request.headers.get("x-veryfront-project-slug"), }); - if (request.url.endsWith("/ai/models")) return Promise.resolve(catalog()); + if (request.url.endsWith("/ai/models")) { + return (catalogGate ?? Promise.resolve()).then(catalog); + } return Promise.resolve( new Response( readableStreamFrom([ @@ -1808,27 +1821,50 @@ describe("provider/veryfront-cloud served catalog loading", () => { return requests; } + /** Stream once; a response the model cannot parse still leaves its request captured. */ async function streamOnce(model: ModelRuntime): Promise { - const result = await model.doStream({ - prompt: [{ role: "user", content: [{ type: "text", text: "Hi" }] }], - } as never); - await drainStream(result.stream); + try { + const result = await model.doStream({ + prompt: [{ role: "user", content: [{ type: "text", text: "Hi" }] }], + } as never); + await drainStream(result.stream); + } catch { + // expected for a Responses request answered with a chat stream + } } + const calls = (requests: CapturedRequest[]) => + requests.map(({ method, url }) => `${method} ${url.replace("https://api.veryfront.com", "")}`); + + /** A catalog serving one OpenAI-protocol model only on chat completions. */ + const chatOnlyCatalog = () => + Response.json({ + models: [{ + id: "gpt-5.9-chat", + modelId: "openai/gpt-5.9-chat", + provider: "openai", + surface: "openai", + operations: ["chat-completions"], + aliases: [], + capabilities: { thinking: true }, + }], + }); + it("builds synchronously, then loads the catalog on the first call and follows it", async () => { setCloudBootstrap(); - const requests = installGateway(() => Response.json(servedCatalogPayload())); + const requests = installGateway(chatOnlyCatalog); - // Before the catalog loads, a reasoning-style OpenAI id would use Responses. - const model = resolveModel("veryfront-cloud/openai/gpt-5.5") as ModelRuntime; + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + // Unlisted and reasoning-style: before the catalog loads it would use Responses. + assertEquals(readVeryfrontCloudModelFacts(model)?.transportPlan.transport, "responses"); assertEquals(requests.length, 0); await streamOnce(model); await streamOnce(model); - assertEquals(requests.map(({ method, url }) => `${method} ${url}`), [ - "GET https://api.veryfront.com/ai/models", - "POST https://api.veryfront.com/ai/v1/chat/completions", - "POST https://api.veryfront.com/ai/v1/chat/completions", + assertEquals(calls(requests), [ + "GET /ai/models", + "POST /ai/v1/chat/completions", + "POST /ai/v1/chat/completions", ]); assertEquals(requests[0]?.authorization, "Bearer vf_test_provider"); assertEquals(requests[0]?.projectSlug, "provider-test-project"); @@ -1844,35 +1880,133 @@ describe("provider/veryfront-cloud served catalog loading", () => { assertEquals(requests.map(({ url }) => url), ["https://api.veryfront.com/ai/models"]); }); - it("still calls the model when the catalog cannot be loaded", async () => { + it("retries the catalog on a later call after a failed first load", async () => { setCloudBootstrap(); - const requests = installGateway(() => Response.json({ error: "unavailable" }, { status: 503 })); + let now = 1_000_000; + __setVeryfrontCloudCatalogClockForTests(() => now); + let available = false; + const requests = installGateway(() => + available ? chatOnlyCatalog() : Response.json({ error: "unavailable" }, { status: 503 }) + ); - await streamOnce(resolveModel("veryfront-cloud/mistral/mistral-small-2503") as ModelRuntime); + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + await streamOnce(model); + available = true; + // Inside the retry window no new catalog request is made. + await streamOnce(model); + now += VERYFRONT_CLOUD_CATALOG_RETRY_MS; + await streamOnce(model); + await streamOnce(model); - assertEquals(requests.map(({ method, url }) => `${method} ${url}`), [ - "GET https://api.veryfront.com/ai/models", - "POST https://api.veryfront.com/ai/v1/chat/completions", + assertEquals(calls(requests), [ + "GET /ai/models", + "POST /ai/v1/responses", + "POST /ai/v1/responses", + "GET /ai/models", + "POST /ai/v1/chat/completions", + "POST /ai/v1/chat/completions", ]); }); - it("loads the catalog with the ambient credentials before thinking defaults are read", async () => { + it("does not settle a model on a caller that stopped waiting", async () => { setCloudBootstrap(); - const requests = installGateway(() => Response.json(servedCatalogPayload())); - assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), undefined); + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const requests = installGateway(chatOnlyCatalog, gate); + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + + const controller = new AbortController(); + controller.abort(); + await model.prepare?.(controller.signal); + release?.(); + await streamOnce(model); + + // The abandoned wait left the model unsettled; the call used the catalog. + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + }); + it("forwards metadata to the model rebuilt from the catalog", async () => { + setCloudBootstrap(); + installGateway(() => + Response.json({ + models: [{ + id: "acme-claude", + modelId: "acme/acme-claude", + provider: "acme", + surface: "anthropic", + operations: ["messages"], + aliases: [], + capabilities: {}, + }], + }) + ); + + const model = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + const coldCapabilities = model.runtimeCapabilities; + await model.prepare?.(); + + // A model constructed now, with the catalog loaded, is the reference. + const warm = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + assertEquals(coldCapabilities, { structuredOutput: true }); + assertEquals( + JSON.stringify(warm.runtimeCapabilities) === JSON.stringify(coldCapabilities), + false, + ); + assertEquals(model.runtimeCapabilities, warm.runtimeCapabilities); + assertEquals(model.modelProvider, "acme"); + assertEquals(readVeryfrontCloudModelFacts(model)?.surface, "anthropic"); + }); + + it("records the transport the request is sent with, from a cold start", async () => { + setCloudBootstrap(); + const bodies: Record[] = []; + const requests = installGateway(chatOnlyCatalog); + const captureBodies = globalThis.fetch; + installMockFetch( + (async (input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + if (request.method === "POST") bodies.push(await request.clone().json()); + return captureBodies(request); + }) as typeof fetch, + ); + const recorded: AgentRunEvent[] = []; + + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + assertEquals(readVeryfrontCloudModelFacts(model)?.transportPlan.transport, "responses"); + await runWithMandatoryRunEventSink( + (event) => { + recorded.push(event); + }, + () => generateText({ model, messages: [{ role: "user", content: "Hi" }], seed: 7 }), + ); + + const context = recorded.find((event) => + event.type === "AGENT_RUN_MODEL_CALL_CONTEXT_RECORDED" + ) as + | { request?: { seed?: number } } + | undefined; + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + // Chat completions carries the seed; Responses would not. Record and request agree. + assertEquals(bodies[0]?.seed, 7); + assertEquals(context?.request?.seed, 7); + }); + + it("loads the catalog with the ambient credentials for synchronous reads", async () => { + setCloudBootstrap(); + const requests = installGateway(chatOnlyCatalog); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-5.9-chat"), undefined); + + assertEquals(await loadVeryfrontCloudModelCatalog(), true); await warmVeryfrontCloudCatalog(); assertEquals(requests.map(({ url }) => url), ["https://api.veryfront.com/ai/models"]); - assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), { - enabled: true, - budgetTokens: 2048, - }); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-5.9-chat"), { enabled: true }); }); it("skips the ambient load without credentials", async () => { const requests = installGateway(() => Response.json(servedCatalogPayload())); + assertEquals(await loadVeryfrontCloudModelCatalog(), false); await warmVeryfrontCloudCatalog(); assertEquals(requests, []); diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index d17cf14109..0530219c7b 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -9,6 +9,7 @@ import type { ModelRuntime } from "../types.ts"; import { getCurrentVeryfrontCloudContext } from "./context.ts"; import { createVeryfrontCloudFetch, + loadVeryfrontCloudModelCatalog, parseVeryfrontCloudModelId, requireVeryfrontCloudBootstrap, resolveVeryfrontCloudGatewayRoute, @@ -18,13 +19,20 @@ import { createVeryfrontCloudOpenAIResponsesModel, } from "./openai.ts"; import { + registerVeryfrontCloudModelFacts, requireVeryfrontCloudWireSurface, resolveVeryfrontCloudOpenAIChatFunctionToolReasoning, resolveVeryfrontCloudOpenAIChatSystemMessages, + resolveVeryfrontCloudOpenAITransport, resolveVeryfrontCloudOpenAITransportPlan, resolveVeryfrontCloudProviderRouting, + type VeryfrontCloudModelFacts, } from "./model-catalog.ts"; -import { isVeryfrontCloudCatalogFresh, loadVeryfrontCloudCatalog } from "./catalog-client.ts"; +import { + isVeryfrontCloudCatalogFresh, + loadVeryfrontCloudCatalog, + withVeryfrontCloudCatalogScope, +} from "./catalog-client.ts"; const IntrinsicReflectApply = Reflect.apply; const HostCrypto = globalThis.crypto; @@ -38,6 +46,7 @@ const ObjectGetPrototypeOf = Object.getPrototypeOf; const ObjectHasOwn = Object.hasOwn; const ObjectPrototype = Object.prototype; const ReflectOwnKeys = Reflect.ownKeys; +const ReflectGet = Reflect.get; function bindModelMethod unknown>( method: T, @@ -83,34 +92,40 @@ function wrapVeryfrontCloudModel( return wrapped; } +/** Upper bound on how long a warm-up before a model call waits for the catalog. */ +const CATALOG_WARM_UP_MAX_WAIT_MS = 3_000; + /** - * @internal Load the served catalog with the ambient Veryfront Cloud - * credentials, before model facts are read synchronously. Never throws: without - * credentials or a catalog, the readers fall back to protocol defaults. + * @internal Load the catalog with the ambient credentials before a model call + * reads facts synchronously, waiting at most a few seconds. */ export async function warmVeryfrontCloudCatalog(abortSignal?: AbortSignal): Promise { - let bootstrap: ReturnType; - try { - bootstrap = requireVeryfrontCloudBootstrap(); - } catch { - return; - } - await loadVeryfrontCloudCatalog({ - apiBaseUrl: bootstrap.apiBaseUrl, - apiToken: bootstrap.apiToken, - ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + await loadVeryfrontCloudModelCatalog({ ...(abortSignal ? { signal: abortSignal } : {}), + maxWaitMs: CATALOG_WARM_UP_MAX_WAIT_MS, }); } +/** Metadata keys a wrapped model forwards to the model it currently calls. */ +const NON_FORWARDED_KEYS: ReadonlySet = new Set([ + "prepare", + "doGenerate", + "doStream", + "constructor", +]); + /** * Wrap a built model so its first async step loads the served catalog. When - * the catalog changes how the model is built, the calls go to a model rebuilt - * from it; the metadata stays that of the model built at construction. Once - * the catalog is settled, calls go straight to the current model. + * the catalog changes how the model is built, calls and metadata go to the + * model rebuilt from it. Once the catalog is settled, calls go straight to the + * current model. + * + * A settled model keeps the facts it settled with for its lifetime: a catalog + * refreshed later applies to models constructed after the refresh. */ function withServedCatalog( model: ModelRuntime, + current: () => ModelRuntime, settled: () => ModelRuntime | undefined, ready: (abortSignal?: AbortSignal) => Promise, ): ModelRuntime { @@ -118,28 +133,52 @@ function withServedCatalog( options !== null && typeof options === "object" ? (options as { abortSignal?: AbortSignal }).abortSignal : undefined; - return ObjectCreate(model, { + const wrapped = ObjectCreate(model, { prepare: { value: async (abortSignal?: AbortSignal): Promise => { - const current = settled() ?? await ready(abortSignal); - if (current.prepare) await current.prepare(abortSignal); + const target = settled() ?? await ready(abortSignal); + if (target.prepare) await target.prepare(abortSignal); }, }, doGenerate: { value: (options: unknown) => { - const current = settled(); - if (current) return current.doGenerate(options); + const target = settled(); + if (target) return target.doGenerate(options); return (async () => await (await ready(readSignal(options))).doGenerate(options))(); }, }, doStream: { value: (options: unknown) => { - const current = settled(); - if (current) return current.doStream(options); + const target = settled(); + if (target) return target.doStream(options); return (async () => await (await ready(readSignal(options))).doStream(options))(); }, }, }); + + // Metadata (provider attribution, capabilities, model ID) follows the model + // the calls go to, so a rebuild never leaves the construction-time values. + const forwarded = new Set(); + let source: object | null = model; + while (source && source !== ObjectPrototype) { + for (const key of ReflectOwnKeys(source)) { + if (forwarded.has(key) || NON_FORWARDED_KEYS.has(key)) continue; + forwarded.add(key); + ObjectDefineProperty(wrapped, key, { + configurable: false, + enumerable: true, + get: () => { + const target = current(); + const value: unknown = IntrinsicReflectApply(ReflectGet, undefined, [target, key]); + return typeof value === "function" + ? IntrinsicReflectApply(FunctionBind, value, [target]) + : value; + }, + }); + } + source = ObjectGetPrototypeOf(source); + } + return wrapped; } type VeryfrontCloudModelOptions = { @@ -177,66 +216,91 @@ function createVeryfrontCloudModelInternal( // cannot replace; ordinary project credentials retain extension behavior. const registry = useFirstPartyTransport ? undefined : ensureBuiltinLLMProviders(); - // The served facts a build reads. A model built before the catalog loaded is - // rebuilt at its first async step when these differ. - function buildFacts(): string { - const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language"); - const catalogModelId = `${provider}/${upstreamModelId}`; - const routing = resolveVeryfrontCloudProviderRouting(provider); - const plan = resolveVeryfrontCloudOpenAITransportPlan(provider, upstreamModelId); - return [ - provider, - routing.surface, - String(routing.native), - plan.transport, - String(plan.pinned), - String(resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogModelId)), - String(resolveVeryfrontCloudOpenAIChatSystemMessages(catalogModelId)), - ].join("\n"); + const catalogScope = { apiBaseUrl, apiToken, ...(projectSlug ? { projectSlug } : {}) }; + + // The catalog facts a build reads, from this model's own credentials and + // project. A model built before the catalog loaded is rebuilt at its first + // async step when these differ. + function readFacts(): VeryfrontCloudModelFacts { + return withVeryfrontCloudCatalogScope(catalogScope, () => { + const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId( + modelId, + "language", + ); + const catalogModelId = `${provider}/${upstreamModelId}`; + const routing = resolveVeryfrontCloudProviderRouting(provider); + const openAIChatReasoningWithFunctionTools = + resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogModelId); + const openAIChatPreserveSystemMessages = resolveVeryfrontCloudOpenAIChatSystemMessages( + catalogModelId, + ); + const openAITransport = resolveVeryfrontCloudOpenAITransport(catalogModelId); + return Object.freeze({ + provider, + surface: routing.surface, + native: routing.native === true, + transportPlan: resolveVeryfrontCloudOpenAITransportPlan(provider, upstreamModelId), + ...(openAITransport === undefined ? {} : { openAITransport }), + ...(openAIChatReasoningWithFunctionTools === undefined + ? {} + : { openAIChatReasoningWithFunctionTools }), + ...(openAIChatPreserveSystemMessages === undefined + ? {} + : { openAIChatPreserveSystemMessages }), + }); + }); } + const factsKey = (value: VeryfrontCloudModelFacts): string => + [ + value.provider, + value.surface, + String(value.native), + value.transportPlan.transport, + String(value.transportPlan.pinned), + String(value.openAITransport), + String(value.openAIChatReasoningWithFunctionTools), + String(value.openAIChatPreserveSystemMessages), + ].join("\n"); const build = (): ModelRuntime => - buildVeryfrontCloudModel({ - modelId, - inferenceCredential, - options, - apiBaseUrl, - apiToken, - projectSlug, - providerCredential, - registry, - useFirstPartyTransport, - }); - let facts = buildFacts(); + withVeryfrontCloudCatalogScope(catalogScope, () => + buildVeryfrontCloudModel({ + modelId, + inferenceCredential, + options, + apiBaseUrl, + apiToken, + projectSlug, + providerCredential, + registry, + useFirstPartyTransport, + })); + let facts = readFacts(); const built = build(); let current = built; let preparing: Promise | undefined; let isSettled = false; - const rebuildIfChanged = (): ModelRuntime => { - const next = buildFacts(); - if (next !== facts) { - facts = next; - current = build(); - } - isSettled = true; + const rebuildIfChanged = (settle: boolean): ModelRuntime => { + const next = readFacts(); + if (factsKey(next) !== factsKey(facts)) current = build(); + facts = next; + // Only a catalog actually obtained settles the model: after a failed or + // abandoned load, the next call tries again. + if (settle) isSettled = true; return current; }; // A fresh cached catalog settles the model without waiting on anything. const settled = (): ModelRuntime | undefined => { if (isSettled) return current; - if (!isVeryfrontCloudCatalogFresh({ apiBaseUrl, ...(projectSlug ? { projectSlug } : {}) })) { - return undefined; - } - return rebuildIfChanged(); + if (!isVeryfrontCloudCatalogFresh(catalogScope)) return undefined; + return rebuildIfChanged(true); }; const prepare = async (abortSignal?: AbortSignal): Promise => { - await loadVeryfrontCloudCatalog({ - apiBaseUrl, - apiToken, - ...(projectSlug ? { projectSlug } : {}), + const catalog = await loadVeryfrontCloudCatalog({ + ...catalogScope, ...(abortSignal ? { signal: abortSignal } : {}), }); - return rebuildIfChanged(); + return rebuildIfChanged(catalog !== undefined); }; const ready = async (abortSignal?: AbortSignal): Promise => { preparing ??= prepare(abortSignal); @@ -246,9 +310,13 @@ function createVeryfrontCloudModelInternal( // A failed build is retried on the next call rather than cached. preparing = undefined; throw error; + } finally { + if (!isSettled) preparing = undefined; } }; - return withServedCatalog(built, settled, ready); + const wrapped = withServedCatalog(built, () => current, settled, ready); + registerVeryfrontCloudModelFacts(wrapped, () => facts); + return wrapped; } interface VeryfrontCloudModelBuild { diff --git a/src/provider/veryfront-cloud/shared.ts b/src/provider/veryfront-cloud/shared.ts index 944f3a74b4..9983ab728a 100644 --- a/src/provider/veryfront-cloud/shared.ts +++ b/src/provider/veryfront-cloud/shared.ts @@ -24,6 +24,7 @@ import { resolveVeryfrontCloudSurface, type VeryfrontCloudProviderId, } from "./model-catalog.ts"; +import { loadVeryfrontCloudCatalog } from "./catalog-client.ts"; import { requireInferenceProviderCredential, requireProviderCredential, @@ -298,6 +299,33 @@ export function requireVeryfrontCloudBootstrap( }; } +/** + * Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud + * credentials and project in effect, so model facts read synchronously + * afterwards (thinking defaults, short aliases such as `opus`, the default + * model) come from it. Resolves to whether a catalog is available. Never + * throws: without credentials or a reachable catalog, the facts shipped with + * this package apply. + */ +export async function loadVeryfrontCloudModelCatalog( + options: { signal?: AbortSignal; maxWaitMs?: number } = {}, +): Promise { + let bootstrap: ReturnType; + try { + bootstrap = requireVeryfrontCloudBootstrap(); + } catch { + return false; + } + const catalog = await loadVeryfrontCloudCatalog({ + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + ...(options.signal ? { signal: options.signal } : {}), + ...(options.maxWaitMs === undefined ? {} : { maxWaitMs: options.maxWaitMs }), + }); + return catalog !== undefined; +} + /** * Host environment variable that restores the vendor-scoped gateway routes. * diff --git a/src/runtime/model-call-context-request.ts b/src/runtime/model-call-context-request.ts index 717268c720..900bb1f8a8 100644 --- a/src/runtime/model-call-context-request.ts +++ b/src/runtime/model-call-context-request.ts @@ -9,6 +9,7 @@ import { } from "#veryfront/provider/shared/openai-reasoning.ts"; import { readProviderOptions } from "#veryfront/provider/runtime-loader.ts"; import { + readVeryfrontCloudModelFacts, resolveVeryfrontCloudOpenAICallTransport, resolveVeryfrontCloudOpenAIChatFunctionToolReasoning, resolveVeryfrontCloudOpenAITransport, @@ -74,8 +75,10 @@ function stopControl(value: unknown): string[] | undefined { function usesOpenAIBuilder(model: ModelCallRuntimeMetadata): boolean { const provider = resolveModelCallProvider(model); if (provider === "openai") return true; - return model.provider === "veryfront-cloud" && provider !== undefined && - resolveVeryfrontCloudProviderRouting(provider).surface === "openai"; + if (model.provider !== "veryfront-cloud" || provider === undefined) return false; + // A model built by this package records the facts it was built with. + const built = readVeryfrontCloudModelFacts(model); + return (built?.surface ?? resolveVeryfrontCloudProviderRouting(provider).surface) === "openai"; } function managedOpenAITransport( @@ -90,12 +93,15 @@ function managedOpenAITransport( // against the call is the one the request is built with. A provider that is // not native to the OpenAI surface never reaches the Responses transport, // whatever its model IDs look like. - return resolveVeryfrontCloudOpenAICallTransport( - provider, - model.modelId, + const usesHostedTool = options.tools?.some((tool) => tool.type === "provider" && tool.id.startsWith("openai.")) === - true, - ); + true; + const built = readVeryfrontCloudModelFacts(model); + if (built) { + if (built.transportPlan.pinned) return built.transportPlan.transport; + return usesHostedTool ? "responses" : "chat-completions"; + } + return resolveVeryfrontCloudOpenAICallTransport(provider, model.modelId, usesHostedTool); } function openAIProviderOptions( @@ -314,10 +320,17 @@ function suppressOpenAIFunctionToolReasoning( // must not apply it either. if (resolveModelCallProvider(model) !== "openai") return false; const catalogId = `openai/${model.modelId}`; + const built = readVeryfrontCloudModelFacts(model); + const openAITransport = built + ? built.openAITransport + : resolveVeryfrontCloudOpenAITransport(catalogId); + const reasoningWithFunctionTools = built + ? built.openAIChatReasoningWithFunctionTools + : resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogId); if ( model.provider === "veryfront-cloud" && - resolveVeryfrontCloudOpenAITransport(catalogId) === "chat-completions" && - resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogId) === false + openAITransport === "chat-completions" && + reasoningWithFunctionTools === false ) { // Match the Chat builder's native bucket precedence, including an own // tools value that clears the neutral list with [] or undefined. diff --git a/src/runtime/runtime-bridge.ts b/src/runtime/runtime-bridge.ts index 190ce72b31..b3e9dddab4 100644 --- a/src/runtime/runtime-bridge.ts +++ b/src/runtime/runtime-bridge.ts @@ -1,3 +1,4 @@ +import { readVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { mapPrivateArray, pushPrivateArray } from "#veryfront/security/private-array.ts"; import { createPrivateMap } from "#veryfront/security/private-map.ts"; import { getPrivateAsyncIterator } from "#veryfront/security/private-iterator.ts"; @@ -737,6 +738,14 @@ async function emitModelCallContextEvent( ): Promise { const sinks = getActiveRunEventSinks(); if (!sinks.mandatory && !sinks.public) return; + // A Veryfront Cloud model settles how it is built on its first async step. + // It does so here, so the recorded request describes the request then sent. + if ( + readVeryfrontCloudModelFacts(options.model) !== undefined && + typeof options.model.prepare === "function" + ) { + await options.model.prepare(options.abortSignal); + } const request = buildModelCallContextRequest(options.model, directOptions); const event: AgentRunModelCallContextEvent = { diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts index 04bf2a5d49..bd1d41b5f3 100644 --- a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -126,12 +126,12 @@ describe("provider/veryfront-cloud/catalog-client", () => { describe("loadVeryfrontCloudCatalog", () => { it("is cold until the first load, then serves the loaded catalog synchronously", async () => { const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); - assertEquals(peekVeryfrontCloudCatalog(), undefined); + assertEquals(peekVeryfrontCloudCatalog(LOAD), undefined); const loaded = await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(LOAD)); assertEquals(loaded?.defaultModelId, "mistral/mistral-small-2503"); - assertEquals(peekVeryfrontCloudCatalog(), loaded); + assertEquals(peekVeryfrontCloudCatalog(LOAD), loaded); assertEquals(stub.requests.length, 1); const [request] = stub.requests; assertEquals(request?.method, "GET"); @@ -152,6 +152,65 @@ describe("provider/veryfront-cloud/catalog-client", () => { assertEquals(first, second); }); + it("keeps a separate entry per credential for the same project", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + await loadVeryfrontCloudCatalog({ ...LOAD, apiToken: "vf_other_credential" }); + }); + + assertEquals( + stub.requests.map((request) => request.headers.get("authorization")), + ["Bearer vf_catalog_test", "Bearer vf_other_credential"], + ); + assertEquals(peekVeryfrontCloudCatalog({ ...LOAD, apiToken: "vf_third" }), undefined); + }); + + it("lets one caller stop waiting without failing the load for the others", async () => { + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const stub = recordingFetch(async () => { + await gate; + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + const controller = new AbortController(); + const abandoned = loadVeryfrontCloudCatalog({ ...LOAD, signal: controller.signal }); + const waiting = loadVeryfrontCloudCatalog(LOAD); + controller.abort(); + assertEquals(await abandoned, undefined); + + release?.(); + assertEquals((await waiting)?.defaultModelId, "mistral/mistral-small-2503"); + // The abandoned wait recorded no failure: the next load is a cache hit. + assertEquals( + (await loadVeryfrontCloudCatalog(LOAD))?.defaultModelId, + "mistral/mistral-small-2503", + ); + }); + assertEquals(stub.requests.length, 1); + assertEquals(stub.requests[0]?.signal.aborted, false); + }); + + it("stops waiting after maxWaitMs while the request finishes for later callers", async () => { + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const stub = recordingFetch(async () => { + await gate; + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + assertEquals(await loadVeryfrontCloudCatalog({ ...LOAD, maxWaitMs: 1 }), undefined); + release?.(); + await settleBackgroundRefresh(); + assertEquals(peekVeryfrontCloudCatalog(LOAD)?.defaultModelId, "mistral/mistral-small-2503"); + }); + assertEquals(stub.requests.length, 1); + }); + it("keeps a separate entry per project", async () => { const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); @@ -188,7 +247,10 @@ describe("provider/veryfront-cloud/catalog-client", () => { assertEquals(stub.requests.length, 2); const refreshed = await loadVeryfrontCloudCatalog(LOAD); assertEquals(refreshed?.defaultModelId, "anthropic/claude-sonnet-4-6"); - assertEquals(peekVeryfrontCloudCatalog()?.defaultModelId, "anthropic/claude-sonnet-4-6"); + assertEquals( + peekVeryfrontCloudCatalog(LOAD)?.defaultModelId, + "anthropic/claude-sonnet-4-6", + ); assertEquals(stub.requests.length, 2); }); }); @@ -202,7 +264,7 @@ describe("provider/veryfront-cloud/catalog-client", () => { await withMockFetch(stub.fetch, async () => { assertEquals(await loadVeryfrontCloudCatalog(LOAD), undefined); - assertEquals(peekVeryfrontCloudCatalog(), undefined); + assertEquals(peekVeryfrontCloudCatalog(LOAD), undefined); status = 200; clock.advance(VERYFRONT_CLOUD_CATALOG_RETRY_MS - 1); @@ -224,7 +286,7 @@ describe("provider/veryfront-cloud/catalog-client", () => { const loaded = await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(LOAD)); assertEquals(loaded, undefined); - assertEquals(peekVeryfrontCloudCatalog(), undefined); + assertEquals(peekVeryfrontCloudCatalog(LOAD), undefined); }); it("keeps the stale catalog when a refresh fails", async () => { @@ -245,7 +307,7 @@ describe("provider/veryfront-cloud/catalog-client", () => { assertEquals(stub.requests.length, 2); assertEquals(afterFailure, loaded); - assertEquals(peekVeryfrontCloudCatalog(), loaded); + assertEquals(peekVeryfrontCloudCatalog(LOAD), loaded); }); }); }); From e3b99bd81cee76f5284e66e93fb91f4b09d77467 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 03:45:49 +0200 Subject: [PATCH 03/17] docs(agent): describe the alias wait without the runtime kind Co-Authored-By: Claude --- src/agent/hosted/default-chat-runtime.ts | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/src/agent/hosted/default-chat-runtime.ts b/src/agent/hosted/default-chat-runtime.ts index 8ad2775a65..c9374fd147 100644 --- a/src/agent/hosted/default-chat-runtime.ts +++ b/src/agent/hosted/default-chat-runtime.ts @@ -525,7 +525,7 @@ function runWithDefaultHostedRequestContext( ); } -/** Longest a hosted run waits for the served catalog before resolving a short alias. */ +/** Longest a run waits for the served catalog before resolving a short model alias. */ const CATALOG_ALIAS_MAX_WAIT_MS = 3_000; /** Create default hosted chat runtime. */ From b7bc757f4b80fe6c2bfb5b6b47c5fd316537b6a0 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 09:52:52 +0200 Subject: [PATCH 04/17] fix(provider): resolve the served default model and keep the base URL query - resolveAgentModelTransport loads the catalog before it resolves an omitted or auto model, then resolves the requested and runtime model again, so the first request uses the default the served catalog names. - The catalog request keeps the API base URL's query, as gateway URLs do. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- src/agent/runtime/model-transport.ts | 30 +++++--- .../veryfront-cloud/catalog-client.ts | 5 +- ...ryfront-cloud-served-default-model.test.ts | 73 +++++++++++++++++++ .../veryfront-cloud-catalog-client.test.ts | 18 +++++ 4 files changed, 116 insertions(+), 10 deletions(-) create mode 100644 tests/integration/agent/veryfront-cloud-served-default-model.test.ts diff --git a/src/agent/runtime/model-transport.ts b/src/agent/runtime/model-transport.ts index 9c5df2afe4..e0e974d2b7 100644 --- a/src/agent/runtime/model-transport.ts +++ b/src/agent/runtime/model-transport.ts @@ -210,18 +210,30 @@ function resolveReasoningWithDefaults( export async function resolveAgentModelTransport( input: ResolveAgentModelTransportInput, ): Promise { - const requestedModel = resolveConfiguredAgentModel(input.modelOverride || input.config.model); - const resolvedModelString = resolveRuntimeModel(input.modelOverride || input.config.model); - const usesVeryfrontCloud = IntrinsicReflectApply(StringStartsWith, resolvedModelString, [ - VERYFRONT_CLOUD_MODEL_PREFIX, - ]) as boolean; + const configuredModel = input.modelOverride || input.config.model; + const startsWithCloudPrefix = (model: string): boolean => + IntrinsicReflectApply(StringStartsWith, model, [VERYFRONT_CLOUD_MODEL_PREFIX]) as boolean; + let requestedModel = resolveConfiguredAgentModel(configuredModel); + let resolvedModelString = resolveRuntimeModel(configuredModel); + // The default model and the thinking defaults read the served catalog, so an + // ambient run loads it before either is resolved and resolves both again. A + // privately resolved model's run was prepared from the catalog as it stood + // then, and its call must keep what that preparation reserved, so a run + // with a private resolver keeps its first resolution. + let warmed = false; + if (startsWithCloudPrefix(resolvedModelString) && !input.resolveModelRuntime) { + await warmVeryfrontCloudCatalog(); + warmed = true; + requestedModel = resolveConfiguredAgentModel(configuredModel); + resolvedModelString = resolveRuntimeModel(configuredModel); + } + const usesVeryfrontCloud = startsWithCloudPrefix(resolvedModelString); const privatelyResolvedModel = input.resolveModelRuntime && usesVeryfrontCloud ? input.resolveModelRuntime(resolvedModelString) : undefined; - // Thinking defaults below read the served catalog. A privately resolved - // model's run was prepared from the catalog as it stood then, and its call - // must keep what that preparation reserved, so only ambient runs load it here. - if (usesVeryfrontCloud && !privatelyResolvedModel) await warmVeryfrontCloudCatalog(); + if (usesVeryfrontCloud && !privatelyResolvedModel && !warmed) { + await warmVeryfrontCloudCatalog(); + } const transport = privatelyResolvedModel ? undefined : await input.config.resolveModelTransport?.({ diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index 2b44c51b01..98b1db879d 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -189,11 +189,14 @@ function cacheKey(scope: VeryfrontCloudCatalogScope): string { }`; } +/** + * The catalog URL under an API base URL. Like the gateway URLs, it keeps the + * base URL's query, which some deployments use to scope or sign requests. + */ function catalogUrl(apiBaseUrl: string): string { const url = new URL(apiBaseUrl); url.pathname = `${url.pathname.replace(/\/+$/, "")}/${VERYFRONT_CLOUD_CATALOG_PATH}`; url.hash = ""; - url.search = ""; return url.toString(); } diff --git a/tests/integration/agent/veryfront-cloud-served-default-model.test.ts b/tests/integration/agent/veryfront-cloud-served-default-model.test.ts new file mode 100644 index 0000000000..0fd7e32082 --- /dev/null +++ b/tests/integration/agent/veryfront-cloud-served-default-model.test.ts @@ -0,0 +1,73 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; +import { clearModelProviders } from "#veryfront/provider"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { resolveAgentModelTransport } from "#veryfront/agent/runtime/model-transport.ts"; +import type { AgentConfig } from "#veryfront/agent/types.ts"; + +const SERVED_DEFAULT = "anthropic/claude-sonnet-4-6"; + +function servedCatalog(): Response { + return Response.json({ + models: [{ + id: "claude-sonnet-4-6", + modelId: SERVED_DEFAULT, + provider: "anthropic", + surface: "anthropic", + operations: ["messages"], + aliases: ["sonnet"], + capabilities: { thinking: true, reasoning_mode: "budget", reasoning_budget_tokens: 2048 }, + }], + defaultModelId: SERVED_DEFAULT, + }); +} + +describe("agent model transport with the served default model", () => { + let catalogRequests = 0; + + beforeEach(() => { + __resetVeryfrontCloudCatalogForTests(); + setEnv("VERYFRONT_API_TOKEN", "vf_default_model_test"); + setEnv("VERYFRONT_PROJECT_SLUG", "default-model-project"); + catalogRequests = 0; + installMockFetch( + ((input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + if (new URL(request.url).pathname === "/ai/models") { + catalogRequests++; + return Promise.resolve(servedCatalog()); + } + return Promise.resolve(new Response("unexpected", { status: 500 })); + }) as typeof fetch, + ); + }); + + afterEach(() => { + restoreMockFetch(); + __resetVeryfrontCloudCatalogForTests(); + deleteEnv("VERYFRONT_API_TOKEN"); + deleteEnv("VERYFRONT_PROJECT_SLUG"); + clearModelProviders(); + }); + + for (const model of [undefined, "auto"]) { + it(`resolves ${model ?? "an omitted model"} to the served default on the first request`, async () => { + const config: AgentConfig = { system: "You are concise.", ...(model ? { model } : {}) }; + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(catalogRequests, 1); + assertEquals(transport.resolvedModelString, `veryfront-cloud/${SERVED_DEFAULT}`); + assertEquals(transport.requestedModel, `veryfront-cloud/${SERVED_DEFAULT}`); + }); + } +}); diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts index bd1d41b5f3..815a43a4ac 100644 --- a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -140,6 +140,24 @@ describe("provider/veryfront-cloud/catalog-client", () => { assertEquals(request?.headers.get("x-veryfront-project-slug"), "catalog-test"); }); + it("keeps the API base URL's path and query on the catalog request", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + await withMockFetch( + stub.fetch, + () => + loadVeryfrontCloudCatalog({ + ...LOAD, + apiBaseUrl: "https://api.veryfront.com/tenant/?scope=signed-value", + }), + ); + + assertEquals( + stub.requests[0]?.url, + "https://api.veryfront.com/tenant/ai/models?scope=signed-value", + ); + }); + it("shares one request between concurrent loads for the same key", async () => { const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); From 43157b5547e102aa79414cff567f3da646ffb4d0 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 10:18:17 +0200 Subject: [PATCH 05/17] fix(provider): load the catalog before every model decision and read it in the same scope - The agent transport loads the served catalog before it classifies any model when Veryfront Cloud is enabled, so an explicit served-only mistral/ routes through Veryfront Cloud on a cold process. - A Veryfront Cloud id is refused as unlisted only against a served catalog. A model checks the listing against its own credentials' catalog, at construction when loaded, otherwise on its first async step. - Hosted runtime creation, hosted chat preparation and context summaries load and resolve under the run's own credentials and project; the eval judge loads before resolving its model. - The catalog cache is a bounded LRU; expired failures are forgotten and an in-flight load is never evicted. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- .../cloud-chat-execution-preparation.ts | 32 ++- src/agent/hosted/context-summary-generator.ts | 40 +++- src/agent/hosted/default-chat-runtime.ts | 23 +- src/agent/runtime/model-resolution.ts | 7 +- src/agent/runtime/model-transport.ts | 28 ++- src/eval/judges.ts | 19 +- .../veryfront-cloud/catalog-client.ts | 57 ++++- src/provider/veryfront-cloud/model-catalog.ts | 9 + src/provider/veryfront-cloud/provider.ts | 23 +- src/provider/veryfront-cloud/shared.ts | 35 ++- ...veryfront-cloud-served-only-models.test.ts | 219 ++++++++++++++++++ .../veryfront-cloud-catalog-client.test.ts | 44 ++++ 12 files changed, 471 insertions(+), 65 deletions(-) create mode 100644 tests/integration/agent/veryfront-cloud-served-only-models.test.ts diff --git a/src/agent/hosted/cloud-chat-execution-preparation.ts b/src/agent/hosted/cloud-chat-execution-preparation.ts index 08fd6c50e8..594b6860c1 100644 --- a/src/agent/hosted/cloud-chat-execution-preparation.ts +++ b/src/agent/hosted/cloud-chat-execution-preparation.ts @@ -1,5 +1,12 @@ import { resolveVeryfrontCloudModelThinking } from "#veryfront/provider"; -import { runWithVeryfrontCloudContext } from "#veryfront/provider/veryfront-cloud/context.ts"; +import { + runWithVeryfrontCloudContext, + runWithVeryfrontCloudContextAsync, +} from "#veryfront/provider/veryfront-cloud/context.ts"; +import { loadVeryfrontCloudModelCatalog } from "#veryfront/provider/veryfront-cloud/shared.ts"; + +/** Longest preparation waits for the served catalog before resolving the model. */ +const CATALOG_MAX_WAIT_MS = 3_000; import { resolveRuntimeModel } from "../runtime/model-resolution.ts"; import type { HostedChatRuntimeCreationResult } from "./chat-runtime-contract.ts"; import { @@ -88,16 +95,25 @@ export async function prepareVeryfrontCloudHostedChatExecution< HostedChatExecutionPreparationResult > { const { logger, rootRun, ...preparationInput } = input; + const cloudContext = { + apiBaseUrl: String(input.apiUrl), + apiToken: input.request.authToken, + projectSlug: input.request.projectSlug, + serviceLayer: "cloud" as const, + }; + // Model ids and thinking defaults resolve against the served catalog, loaded + // and read under the request's own credentials and project. + await runWithVeryfrontCloudContextAsync( + cloudContext, + () => loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS }), + ); const resolveModelId = (modelId: string | undefined): string | undefined => runWithVeryfrontCloudContext( - { - apiBaseUrl: String(input.apiUrl), - apiToken: input.request.authToken, - projectSlug: input.request.projectSlug, - serviceLayer: "cloud", - }, + cloudContext, () => modelId === undefined ? undefined : resolveRuntimeModel(modelId), ); + const resolveModelThinking: typeof resolveVeryfrontCloudModelThinking = (modelId) => + runWithVeryfrontCloudContext(cloudContext, () => resolveVeryfrontCloudModelThinking(modelId)); return await prepareHostedChatExecution({ ...preparationInput, @@ -106,6 +122,6 @@ export async function prepareVeryfrontCloudHostedChatExecution< logger, }), resolveModelId, - resolveModelThinking: resolveVeryfrontCloudModelThinking, + resolveModelThinking, }); } diff --git a/src/agent/hosted/context-summary-generator.ts b/src/agent/hosted/context-summary-generator.ts index f912786583..50cc709008 100644 --- a/src/agent/hosted/context-summary-generator.ts +++ b/src/agent/hosted/context-summary-generator.ts @@ -4,7 +4,12 @@ import { resolveVeryfrontCloudGatewayModelId, resolveVeryfrontCloudModelId, } from "../../provider/index.ts"; -import { runWithVeryfrontCloudContextAsync } from "#veryfront/provider/veryfront-cloud/context.ts"; +import { + runWithVeryfrontCloudContext, + runWithVeryfrontCloudContextAsync, + type VeryfrontCloudContext, +} from "#veryfront/provider/veryfront-cloud/context.ts"; +import { loadVeryfrontCloudModelCatalog } from "#veryfront/provider/veryfront-cloud/shared.ts"; import { generateText } from "../../runtime/runtime-bridge.ts"; import { redactSensitive, sanitizeUrlCredentials } from "#veryfront/utils"; import type { TextGenerationRuntimeMessage } from "../runtime/text-generation-runtime-message-types.ts"; @@ -168,12 +173,7 @@ async function summarizeSegment(input: { const generate = input.options.generateText ?? generateText; const resolve = input.options.resolveModel ?? resolveModel; const result = await runWithVeryfrontCloudContextAsync( - { - apiBaseUrl: input.options.apiUrl.toString(), - apiToken: input.options.authToken, - projectSlug: input.options.projectSlug ?? undefined, - serviceLayer: "cloud", - }, + summaryCloudContext(input.options), () => Promise.resolve(generate({ model: resolve(input.modelId), @@ -192,6 +192,20 @@ async function summarizeSegment(input: { return result.text.trim(); } +function summaryCloudContext( + options: VeryfrontCloudContextSummaryGeneratorOptions, +): VeryfrontCloudContext { + return { + apiBaseUrl: options.apiUrl.toString(), + apiToken: options.authToken, + projectSlug: options.projectSlug ?? undefined, + serviceLayer: "cloud", + }; +} + +/** Longest summary generation waits for the served catalog before resolving its model. */ +const CATALOG_MAX_WAIT_MS = 3_000; + function resolveSummaryModelId(model: string | undefined): string { const cloudModelId = resolveVeryfrontCloudModelId(model); return resolveVeryfrontCloudGatewayModelId(cloudModelId) ?? cloudModelId; @@ -202,7 +216,17 @@ export function createVeryfrontCloudContextSummaryGenerator( options: VeryfrontCloudContextSummaryGeneratorOptions, ): ContextSummaryGenerator { return async ({ messagesToSummarize, retainedMessages, customInstructions }) => { - const modelId = resolveSummaryModelId(options.model); + // The model resolves against the served catalog loaded for the same + // credentials and project the summary calls use. + const cloudContext = summaryCloudContext(options); + await runWithVeryfrontCloudContextAsync( + cloudContext, + () => loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS }), + ); + const modelId = runWithVeryfrontCloudContext( + cloudContext, + () => resolveSummaryModelId(options.model), + ); const chunks = chunkSerializedMessages(messagesToSummarize, options.maxInputTokens); let summary = ""; diff --git a/src/agent/hosted/default-chat-runtime.ts b/src/agent/hosted/default-chat-runtime.ts index c9374fd147..b12526cad7 100644 --- a/src/agent/hosted/default-chat-runtime.ts +++ b/src/agent/hosted/default-chat-runtime.ts @@ -17,7 +17,7 @@ import { resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; -import { loadVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { loadVeryfrontCloudModelCatalog } from "#veryfront/provider/veryfront-cloud/shared.ts"; import { runWithVeryfrontCloudContext, runWithVeryfrontCloudContextAsync, @@ -537,20 +537,21 @@ export async function createDefaultHostedChatRuntime( return await runWithEffectiveSourceIntegrationPolicy( input.sourceIntegrationPolicy, async () => { - // A short model alias resolves through the served catalog. - if (input.config.apiUrl && input.options.authToken) { - await loadVeryfrontCloudCatalog({ - apiBaseUrl: input.config.apiUrl, - apiToken: input.options.authToken, - ...(input.options.projectSlug ? { projectSlug: input.options.projectSlug } : {}), - maxWaitMs: CATALOG_ALIAS_MAX_WAIT_MS, - }); - } - const modelId = resolveVeryfrontCloudModelId(input.options.model); const cloudContext = createCloudContext({ config: input.config, options: input.options, }); + // A short model alias resolves through the served catalog. It is loaded + // and read under the run's own credentials and project, the same scope + // the run's later model reads use. + await runWithVeryfrontCloudContextAsync( + cloudContext, + () => loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_ALIAS_MAX_WAIT_MS }), + ); + const modelId = runWithVeryfrontCloudContext( + cloudContext, + () => resolveVeryfrontCloudModelId(input.options.model), + ); const taskContext = input.createTaskContext ? input.createTaskContext({ options: input.options, modelId }) : createDefaultTaskContext({ options: input.options, modelId }); diff --git a/src/agent/runtime/model-resolution.ts b/src/agent/runtime/model-resolution.ts index 42baa0141b..fd2d9397b6 100644 --- a/src/agent/runtime/model-resolution.ts +++ b/src/agent/runtime/model-resolution.ts @@ -8,6 +8,7 @@ import { createRetiredVeryfrontCloudModelError, isRetiredVeryfrontCloudModelId, isSupportedMistralModelId, + isVeryfrontCloudCatalogLoaded, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { DEFAULT_MODEL_CREDENTIAL_MISMATCH, NOT_SUPPORTED } from "#veryfront/errors"; import { @@ -142,7 +143,11 @@ function isSupportedHostedMistralModel(modelId: string): boolean { } function isUnsupportedVeryfrontCloudMistralModel(modelId: string): boolean { - return modelId.startsWith("veryfront-cloud/mistral/") && !isSupportedMistralModelId(modelId); + // An explicit Veryfront Cloud id is refused only against a served catalog: + // the shipped list cannot know a model the platform added since, and the + // model checks its own catalog once that has loaded. + return modelId.startsWith("veryfront-cloud/mistral/") && isVeryfrontCloudCatalogLoaded() && + !isSupportedMistralModelId(modelId); } function normalizeVeryfrontCloudRuntimeModel(modelId: string): string { diff --git a/src/agent/runtime/model-transport.ts b/src/agent/runtime/model-transport.ts index e0e974d2b7..3dc8c89eca 100644 --- a/src/agent/runtime/model-transport.ts +++ b/src/agent/runtime/model-transport.ts @@ -8,6 +8,7 @@ import { type AgentConfig, type RuntimeReasoningOption } from "../types.ts"; import { type ModelRuntime, resolveModel } from "#veryfront/provider"; import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; import { warmVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import { isVeryfrontCloudEnabled } from "#veryfront/platform/cloud/resolver.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; import { resolveConfiguredAgentModel, @@ -213,27 +214,24 @@ export async function resolveAgentModelTransport( const configuredModel = input.modelOverride || input.config.model; const startsWithCloudPrefix = (model: string): boolean => IntrinsicReflectApply(StringStartsWith, model, [VERYFRONT_CLOUD_MODEL_PREFIX]) as boolean; - let requestedModel = resolveConfiguredAgentModel(configuredModel); - let resolvedModelString = resolveRuntimeModel(configuredModel); - // The default model and the thinking defaults read the served catalog, so an - // ambient run loads it before either is resolved and resolves both again. A - // privately resolved model's run was prepared from the catalog as it stood - // then, and its call must keep what that preparation reserved, so a run - // with a private resolver keeps its first resolution. - let warmed = false; - if (startsWithCloudPrefix(resolvedModelString) && !input.resolveModelRuntime) { + // Every decision below reads the served catalog: which model an omitted or + // `auto` model means, whether an explicit provider model is served through + // Veryfront Cloud, and the thinking defaults. An ambient run that can reach + // Veryfront Cloud loads the catalog before any of them. A run with a private + // model resolver was prepared from the catalog as it stood then, and its call + // must keep what that preparation reserved, so it does not load it here. + if ( + !input.resolveModelRuntime && isVeryfrontCloudEnabled() && + !(typeof configuredModel === "string" && configuredModel.startsWith("local/")) + ) { await warmVeryfrontCloudCatalog(); - warmed = true; - requestedModel = resolveConfiguredAgentModel(configuredModel); - resolvedModelString = resolveRuntimeModel(configuredModel); } + const requestedModel = resolveConfiguredAgentModel(configuredModel); + const resolvedModelString = resolveRuntimeModel(configuredModel); const usesVeryfrontCloud = startsWithCloudPrefix(resolvedModelString); const privatelyResolvedModel = input.resolveModelRuntime && usesVeryfrontCloud ? input.resolveModelRuntime(resolvedModelString) : undefined; - if (usesVeryfrontCloud && !privatelyResolvedModel && !warmed) { - await warmVeryfrontCloudCatalog(); - } const transport = privatelyResolvedModel ? undefined : await input.config.resolveModelTransport?.({ diff --git a/src/eval/judges.ts b/src/eval/judges.ts index 8eb5dd0115..8faa9d86e1 100644 --- a/src/eval/judges.ts +++ b/src/eval/judges.ts @@ -1,5 +1,12 @@ import { resolveRuntimeModel } from "#veryfront/agent/runtime/model-resolution.ts"; -import { type ModelRuntime, resolveModel } from "#veryfront/provider"; +import { + loadVeryfrontCloudModelCatalog, + type ModelRuntime, + resolveModel, +} from "#veryfront/provider"; + +/** Longest a judge waits for the served catalog before resolving its model. */ +const CATALOG_MAX_WAIT_MS = 3_000; import { generateText } from "#veryfront/runtime/runtime-bridge.ts"; import { classifyEvalModelAccessDenial, isEvalModelAccessDeniedError } from "./model-access.ts"; @@ -83,8 +90,12 @@ function clampScore(score: number): number { return Math.max(0, Math.min(1, score)); } -function resolveJudgeModel(model: string | ModelRuntime | undefined): ModelRuntime { +async function resolveJudgeModel( + model: string | ModelRuntime | undefined, +): Promise { if (model && typeof model === "object") return model; + // Whether the judge model routes through Veryfront Cloud reads the served catalog. + await loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS }); return resolveModel(resolveRuntimeModel(model ?? DEFAULT_JUDGE_MODEL)); } @@ -369,7 +380,7 @@ function createLlmRubricJudge( return async (input) => { try { - const model = resolveJudgeModel(options.model); + const model = await resolveJudgeModel(options.model); const response = await generateText({ model, messages: [ @@ -434,7 +445,7 @@ function createLlmGroundednessJudge( return async (input) => { try { - const model = resolveJudgeModel(validatedOptions.model); + const model = await resolveJudgeModel(validatedOptions.model); const response = await generateText({ model, messages: [{ diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index 98b1db879d..6aae5d2f06 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -26,6 +26,12 @@ import { logger } from "#veryfront/utils/logger/logger.ts"; export const VERYFRONT_CLOUD_CATALOG_TTL_MS = 5 * 60_000; /** How long a failed load waits before the next attempt for the same key. */ export const VERYFRONT_CLOUD_CATALOG_RETRY_MS = 30_000; +/** + * Most catalogs kept at once. Run-scoped credentials rotate, so each run can + * add a key; the least recently used entry goes first, and a load in flight is + * never evicted. + */ +export const VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES = 256; /** Upper bound on one catalog request. */ const VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS = 10_000; /** Header naming the project a catalog request is scoped to. */ @@ -183,6 +189,38 @@ function credentialFingerprint(token: string): string { return `${a.toString(16)}${b.toString(16)}`; } +/** Store an entry as the most recently used, evicting the least recently used over the cap. */ +function rememberEntry(key: string, entry: CatalogEntry): void { + entries.delete(key); + entries.set(key, entry); + evictOldest(entries); +} + +/** Mark an entry as just used, so eviction keeps it longest. */ +function touchEntry(key: string, entry: CatalogEntry): void { + entries.delete(key); + entries.set(key, entry); +} + +/** Drop the oldest keys over the cap, skipping any key with a load in flight. */ +function evictOldest(map: Map): void { + if (map.size <= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) return; + for (const key of [...map.keys()]) { + if (map.size <= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) return; + if (!inflight.has(key)) map.delete(key); + } +} + +/** Record a failed load, forgetting failures whose retry window has passed. */ +function rememberFailure(key: string, at: number): void { + for (const [failedKey, failedTime] of failedAt) { + if (at - failedTime >= VERYFRONT_CLOUD_CATALOG_RETRY_MS) failedAt.delete(failedKey); + } + failedAt.delete(key); + failedAt.set(key, at); + evictOldest(failedAt); +} + function cacheKey(scope: VeryfrontCloudCatalogScope): string { return `${scope.apiBaseUrl}\n${scope.projectSlug ?? ""}\n${ credentialFingerprint(scope.apiToken) @@ -242,14 +280,14 @@ function refresh( const request = fetchCatalog(options).then( (catalog) => { if (started !== generation) return catalog; - entries.set(key, { catalog, fetchedAt: now() }); + rememberEntry(key, { catalog, fetchedAt: now() }); failedAt.delete(key); failureLogged = false; return catalog; }, (error: unknown) => { if (started !== generation) return undefined; - failedAt.set(key, now()); + rememberFailure(key, now()); if (!failureLogged) { failureLogged = true; logger.warn( @@ -302,12 +340,14 @@ export function loadVeryfrontCloudCatalog( const entry = entries.get(key); const current = now(); if (entry && current - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS) { + touchEntry(key, entry); return Promise.resolve(entry.catalog); } const lastFailure = failedAt.get(key); if (lastFailure !== undefined && current - lastFailure < VERYFRONT_CLOUD_CATALOG_RETRY_MS) { return Promise.resolve(entry?.catalog); } + if (lastFailure !== undefined) failedAt.delete(key); const request = refresh(key, { apiBaseUrl: options.apiBaseUrl, apiToken: options.apiToken, @@ -358,7 +398,11 @@ export function peekVeryfrontCloudCatalog( ): VeryfrontCloudCatalog | undefined { if (seeded) return seeded; const key = scope ? cacheKey(scope) : activeKey; - return key === undefined ? undefined : entries.get(key)?.catalog; + if (key === undefined) return undefined; + const entry = entries.get(key); + if (!entry) return undefined; + touchEntry(key, entry); + return entry.catalog; } /** @internal Serve a fixed catalog for every key, as if freshly loaded. `undefined` clears it. */ @@ -373,7 +417,12 @@ export function __setVeryfrontCloudCatalogForScopeForTests( ): void { const catalog = parseVeryfrontCloudCatalog(payload); if (!catalog) throw new TypeError("Test catalog payload has no model list"); - entries.set(cacheKey(scope), { catalog, fetchedAt: now() }); + rememberEntry(cacheKey(scope), { catalog, fetchedAt: now() }); +} + +/** @internal How many catalogs and recorded failures the cache holds. */ +export function __veryfrontCloudCatalogSizesForTests(): { entries: number; failures: number } { + return { entries: entries.size, failures: failedAt.size }; } /** @internal Forget every loaded catalog, pending load and failure. */ diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index c91c57f6f6..cd3a0db0fd 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -264,6 +264,15 @@ function loadedCatalog(): VeryfrontCloudCatalog | undefined { return scope ? peekVeryfrontCloudCatalog(scope) : peekVeryfrontCloudCatalog(); } +/** + * Whether a served catalog has loaded for the scope reads use right now. While + * it has not, reads fall back to the shipped list, which cannot know models + * the platform added since, so a caller should not refuse a model on it alone. + */ +export function isVeryfrontCloudCatalogLoaded(): boolean { + return loadedCatalog() !== undefined; +} + /** * The index reads use: the served catalog loaded for the current scope, or the * shipped list while none has loaded for it. diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index 0530219c7b..18229202b6 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -8,6 +8,7 @@ import { getHostSecret } from "#veryfront/platform/compat/process/env.ts"; import type { ModelRuntime } from "../types.ts"; import { getCurrentVeryfrontCloudContext } from "./context.ts"; import { + assertVeryfrontCloudModelListed, createVeryfrontCloudFetch, loadVeryfrontCloudModelCatalog, parseVeryfrontCloudModelId, @@ -31,6 +32,7 @@ import { import { isVeryfrontCloudCatalogFresh, loadVeryfrontCloudCatalog, + peekVeryfrontCloudCatalog, withVeryfrontCloudCatalogScope, } from "./catalog-client.ts"; @@ -194,9 +196,10 @@ function createVeryfrontCloudModelInternal( inferenceCredential?: string, options: VeryfrontCloudModelOptions = {}, ): ModelRuntime { - // Parsed here so a malformed ID fails at construction; the provider is - // resolved again at build time, once the served catalog may name its alias. - parseVeryfrontCloudModelId(modelId, "language"); + // Parsed here so a malformed or retired ID fails at construction. Whether the + // catalog lists the model is checked against this model's own catalog: now + // when it has loaded, otherwise once the first async step has loaded it. + parseVeryfrontCloudModelId(modelId, "language", { catalogChecks: false }); const { apiBaseUrl, apiToken, projectSlug } = options.credentialSource === "application" ? requireApplicationBootstrap() : requireVeryfrontCloudBootstrap(inferenceCredential, options.apiBaseUrl); @@ -226,6 +229,7 @@ function createVeryfrontCloudModelInternal( const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId( modelId, "language", + { catalogChecks: false }, ); const catalogModelId = `${provider}/${upstreamModelId}`; const routing = resolveVeryfrontCloudProviderRouting(provider); @@ -275,6 +279,14 @@ function createVeryfrontCloudModelInternal( registry, useFirstPartyTransport, })); + // The listing check runs in this model's own scope, against a catalog + // actually loaded for it; before that, the platform answers for the model. + const assertListed = (): void => + withVeryfrontCloudCatalogScope(catalogScope, () => { + const parsed = parseVeryfrontCloudModelId(modelId, "language", { catalogChecks: false }); + assertVeryfrontCloudModelListed(parsed.provider, parsed.modelId); + }); + if (peekVeryfrontCloudCatalog(catalogScope) !== undefined) assertListed(); let facts = readFacts(); const built = build(); let current = built; @@ -282,6 +294,7 @@ function createVeryfrontCloudModelInternal( let isSettled = false; const rebuildIfChanged = (settle: boolean): ModelRuntime => { const next = readFacts(); + if (settle) assertListed(); if (factsKey(next) !== factsKey(facts)) current = build(); facts = next; // Only a catalog actually obtained settles the model: after a failed or @@ -344,7 +357,9 @@ function buildVeryfrontCloudModel(build: VeryfrontCloudModelBuild): ModelRuntime registry, useFirstPartyTransport, } = build; - const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language"); + const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language", { + catalogChecks: false, + }); // Builders keep the upstream model id; on a vendor-neutral route the fetch // wrapper sends it as `/`. const { baseURL, wireModelProvider } = resolveVeryfrontCloudGatewayRoute(apiBaseUrl, provider); diff --git a/src/provider/veryfront-cloud/shared.ts b/src/provider/veryfront-cloud/shared.ts index 8d26654976..c222856adf 100644 --- a/src/provider/veryfront-cloud/shared.ts +++ b/src/provider/veryfront-cloud/shared.ts @@ -213,6 +213,14 @@ function createInvalidModelIdError(modelId: string): Error { export function parseVeryfrontCloudModelId( modelId: string, kind: "language" | "embedding", + options: { + /** + * Also refuse a model the catalog in effect does not list (the Mistral + * check). A model built before its catalog loaded passes `false` and runs + * the check once the catalog for its own credentials is known. + */ + catalogChecks?: boolean; + } = {}, ): ParsedVeryfrontCloudModelId { const slashIndex = modelId.indexOf("/"); if (slashIndex === -1) { @@ -243,16 +251,8 @@ export function parseVeryfrontCloudModelId( ); } - if ( - kind === "language" && normalizedProvider === "mistral" && - !isSupportedMistralModelId(`mistral/${upstreamModelId}`) - ) { - throw toError( - createError({ - type: "config", - message: `Unsupported Mistral model "mistral/${upstreamModelId}"`, - }), - ); + if (kind === "language" && options.catalogChecks !== false) { + assertVeryfrontCloudModelListed(normalizedProvider, upstreamModelId); } if (kind === "language" && isRetiredVeryfrontCloudModelId(modelId)) { @@ -265,6 +265,21 @@ export function parseVeryfrontCloudModelId( }; } +/** + * Refuse a Mistral model the catalog in effect does not list, so a caller gets + * a clear error rather than a gateway-side failure. + */ +export function assertVeryfrontCloudModelListed(provider: string, upstreamModelId: string): void { + if (provider === "mistral" && !isSupportedMistralModelId(`mistral/${upstreamModelId}`)) { + throw toError( + createError({ + type: "config", + message: `Unsupported Mistral model "mistral/${upstreamModelId}"`, + }), + ); + } +} + export function requireVeryfrontCloudBootstrap( apiTokenOverride?: string, inferenceApiBaseUrlOverride?: string, diff --git a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts new file mode 100644 index 0000000000..c9990ac782 --- /dev/null +++ b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts @@ -0,0 +1,219 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals, assertRejects } from "#veryfront/testing/assert.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; +import { clearModelProviders, loadVeryfrontCloudModelCatalog } from "#veryfront/provider"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { resolveVeryfrontCloudModelId } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; +import { createVeryfrontCloudInferenceModel } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import type { ModelRuntime } from "#veryfront/provider/types.ts"; +import { resolveAgentModelTransport } from "#veryfront/agent/runtime/model-transport.ts"; +import { createDefaultHostedChatRuntime } from "#veryfront/agent/hosted/default-chat-runtime.ts"; +import type { + RemoteMCPToolSourceConfig, + RemoteToolSource, + ToolExecutionContext, +} from "#veryfront/tool"; +import { defineSchema } from "#veryfront/schemas/define.ts"; + +/** + * A process that has loaded no catalog meets a model and an alias only the + * served catalog knows. Each case drives one entry point cold, with ambient or + * explicit run credentials, and checks it routes through Veryfront Cloud with + * the catalog loaded for those same credentials. + */ + +const SERVED_ONLY_MODEL = "mistral/mistral-medium-2609"; +const SERVED_ONLY_ALIAS = "medium"; +const AMBIENT_TOKEN = "vf_ambient_token"; +const RUN_TOKEN = "vf_run_token"; + +type Captured = { method: string; path: string; authorization: string | null; model?: unknown }; + +/** A catalog that knows the served-only model and alias, for one credential. */ +function catalogFor(authorization: string | null): Response { + const knowsModel = authorization === `Bearer ${RUN_TOKEN}` || + authorization === `Bearer ${AMBIENT_TOKEN}`; + return Response.json({ + models: knowsModel + ? [{ + id: "mistral-medium-2609", + modelId: SERVED_ONLY_MODEL, + provider: "mistral", + surface: "openai", + operations: ["chat-completions"], + aliases: [SERVED_ONLY_ALIAS], + capabilities: {}, + }] + : [], + }); +} + +function installGateway(onlyRunCredentialKnowsModel = false): Captured[] { + const captured: Captured[] = []; + const encoder = new TextEncoder(); + installMockFetch( + (async (input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + const authorization = request.headers.get("authorization"); + const entry: Captured = { + method: request.method, + path: new URL(request.url).pathname, + authorization, + }; + captured.push(entry); + if (entry.path === "/ai/models") { + return catalogFor( + onlyRunCredentialKnowsModel && authorization !== `Bearer ${RUN_TOKEN}` + ? null + : authorization, + ); + } + if (request.method === "POST") entry.model = (await request.json()).model; + return new Response( + new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode('data: {"choices":[{"finish_reason":"stop"}]}\n\n')); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + }, + }), + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); + }) as typeof fetch, + ); + return captured; +} + +async function streamOnce(model: ModelRuntime): Promise { + const result = await model.doStream({ prompt: [] } as never); + const reader = result.stream.getReader(); + while (!(await reader.read()).done) { + // drain + } +} + +describe("served-only models from a cold process", () => { + beforeEach(() => { + __resetVeryfrontCloudCatalogForTests(); + }); + + afterEach(() => { + restoreMockFetch(); + __resetVeryfrontCloudCatalogForTests(); + deleteEnv("VERYFRONT_API_TOKEN"); + deleteEnv("VERYFRONT_PROJECT_SLUG"); + clearModelProviders(); + }); + + describe("decided before the catalog loads", () => { + it("routes an explicit served-only model through Veryfront Cloud with ambient credentials", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + const captured = installGateway(); + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config: { model: SERVED_ONLY_MODEL, system: "You are concise." }, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(transport.resolvedModelString, `veryfront-cloud/${SERVED_ONLY_MODEL}`); + assertEquals(captured.map(({ path }) => path), ["/ai/models"]); + }); + + it("resolves a served-only alias once the ambient catalog is loaded", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + installGateway(); + + assertEquals(await loadVeryfrontCloudModelCatalog(), true); + assertEquals(resolveVeryfrontCloudModelId(SERVED_ONLY_ALIAS), SERVED_ONLY_MODEL); + }); + + it("builds a served-only model with run credentials and serves it after its own load", async () => { + const captured = installGateway(); + + const model = createVeryfrontCloudInferenceModel(SERVED_ONLY_MODEL, RUN_TOKEN); + await streamOnce(model); + + assertEquals( + captured.map(({ method, path, authorization }) => `${method} ${path} ${authorization}`), + [ + `GET /ai/models Bearer ${RUN_TOKEN}`, + `POST /ai/v1/chat/completions Bearer ${RUN_TOKEN}`, + ], + ); + assertEquals(captured[1]?.model, SERVED_ONLY_MODEL); + }); + + it("refuses a Mistral model the run's own catalog does not list, on the first call", async () => { + installGateway(); + + const model = createVeryfrontCloudInferenceModel("mistral/not-served", RUN_TOKEN); + await assertRejects( + () => streamOnce(model), + Error, + 'Unsupported Mistral model "mistral/not-served"', + ); + }); + }); + + describe("read in the scope the catalog was loaded for", () => { + it("resolves a hosted alias from the run's catalog, not the ambient one", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "ambient-project"); + const captured = installGateway(true); + + const runtime = await createDefaultHostedChatRuntime({ + sourceIntegrationPolicy: { schemaVersion: 1, mode: "unrestricted" }, + options: { + projectId: "project-1", + projectSlug: "run-project", + branchId: "branch-1", + authToken: RUN_TOKEN, + instructions: "Base instructions", + model: SERVED_ONLY_ALIAS, + allowedTools: [], + conversationId: "conversation-1", + userId: "user-1", + }, + config: { + apiUrl: "https://api.veryfront.com", + apiMcpUrl: "https://api.veryfront.com/mcp", + studioMcpUrl: "https://studio.example.com/mcp", + }, + buildLocalTools: () => ({ noop: localTool("No-op") }), + createRemoteToolSource: emptyRemoteSource, + preloadLatestConversationUserText: false, + }); + + assertEquals(runtime.modelId, SERVED_ONLY_MODEL); + const catalogRequests = captured.filter(({ path }) => path === "/ai/models"); + assertEquals(catalogRequests.map(({ authorization }) => authorization), [ + `Bearer ${RUN_TOKEN}`, + ]); + await runtime.cleanup?.(); + }); + }); +}); + +function localTool(description: string) { + return { + description, + inputSchema: defineSchema((v) => v.object({}))(), + execute: () => ({ ok: true }), + }; +} + +function emptyRemoteSource(config: RemoteMCPToolSourceConfig): RemoteToolSource { + return { + id: config.id ?? "source", + listTools: () => Promise.resolve([]), + executeTool: (_toolName: string, _args: unknown, _context?: ToolExecutionContext) => + Promise.resolve({ ok: true }), + }; +} diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts index 815a43a4ac..d9ca9eeaa9 100644 --- a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -5,9 +5,12 @@ import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { __resetVeryfrontCloudCatalogForTests, __setVeryfrontCloudCatalogClockForTests, + __setVeryfrontCloudCatalogForScopeForTests, + __veryfrontCloudCatalogSizesForTests, loadVeryfrontCloudCatalog, parseVeryfrontCloudCatalog, peekVeryfrontCloudCatalog, + VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES, VERYFRONT_CLOUD_CATALOG_RETRY_MS, VERYFRONT_CLOUD_CATALOG_TTL_MS, } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; @@ -170,6 +173,47 @@ describe("provider/veryfront-cloud/catalog-client", () => { assertEquals(first, second); }); + it("evicts the least recently used catalog over the cap and keeps the newest", () => { + const scopeFor = (index: number) => ({ ...LOAD, apiToken: `vf_rotating_${index}` }); + const payload = { models: [], defaultModelId: "mistral/mistral-small-2503" }; + for (let index = 0; index < VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES; index++) { + __setVeryfrontCloudCatalogForScopeForTests(scopeFor(index), payload); + } + // Reading the second entry makes it recent, so the first is evicted instead. + assertEquals(peekVeryfrontCloudCatalog(scopeFor(1)) !== undefined, true); + + __setVeryfrontCloudCatalogForScopeForTests( + scopeFor(VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES), + payload, + ); + + assertEquals( + __veryfrontCloudCatalogSizesForTests().entries, + VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES, + ); + assertEquals(peekVeryfrontCloudCatalog(scopeFor(0)), undefined); + assertEquals(peekVeryfrontCloudCatalog(scopeFor(1)) !== undefined, true); + assertEquals( + peekVeryfrontCloudCatalog(scopeFor(VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES)) !== undefined, + true, + ); + }); + + it("forgets a failure once its retry window has passed", async () => { + const clock = useClock(); + const stub = recordingFetch(() => jsonResponse({}, 503)); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + assertEquals(__veryfrontCloudCatalogSizesForTests().failures, 1); + clock.advance(VERYFRONT_CLOUD_CATALOG_RETRY_MS); + await loadVeryfrontCloudCatalog({ ...LOAD, apiToken: "vf_other_credential" }); + }); + + // The first failure's window passed, so only the new one is kept. + assertEquals(__veryfrontCloudCatalogSizesForTests().failures, 1); + }); + it("keeps a separate entry per credential for the same project", async () => { const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); From 77a262d6237530865e1aa7e02782c7e9b5e2c76b Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 10:28:07 +0200 Subject: [PATCH 06/17] fix(provider): resolve served-only aliases and keep each caller's wait its own - Agent model resolution resolves a short alias only the loaded served catalog knows, after the known aliases so a bare vendor name keeps its direct-key meaning; retired models are never returned. - A model no longer shares one prepare promise between callers: each caller waits on its own signal while the catalog request stays shared, so one caller giving up never decides for a concurrent one. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- src/agent/runtime/model-resolution.ts | 7 +++- src/provider/veryfront-cloud/model-catalog.ts | 13 ++++++++ src/provider/veryfront-cloud/provider.test.ts | 19 +++++++++++ src/provider/veryfront-cloud/provider.ts | 18 +++-------- ...veryfront-cloud-served-only-models.test.ts | 32 +++++++++++++++++++ 5 files changed, 74 insertions(+), 15 deletions(-) diff --git a/src/agent/runtime/model-resolution.ts b/src/agent/runtime/model-resolution.ts index fd2d9397b6..86f0aba7ab 100644 --- a/src/agent/runtime/model-resolution.ts +++ b/src/agent/runtime/model-resolution.ts @@ -9,6 +9,7 @@ import { isRetiredVeryfrontCloudModelId, isSupportedMistralModelId, isVeryfrontCloudCatalogLoaded, + resolveServedVeryfrontCloudAlias, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { DEFAULT_MODEL_CREDENTIAL_MISMATCH, NOT_SUPPORTED } from "#veryfront/errors"; import { @@ -96,7 +97,11 @@ export function resolveConfiguredAgentModel(model?: string): string { return normalized; } - return LEGACY_MODEL_ALIASES.get(normalized) ?? normalized; + // Known aliases first, so a bare vendor name keeps its direct-key meaning; + // then an alias only the loaded served catalog knows. + return LEGACY_MODEL_ALIASES.get(normalized) ?? + resolveServedVeryfrontCloudAlias(normalized) ?? + normalized; } /** Resolve the provider-options key used by the effective model runtime. */ diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index cd3a0db0fd..48b252d01f 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -696,6 +696,19 @@ export function tryGetVeryfrontCloudProviderFromModelId( } } +/** + * Provider-qualified ID of a short alias only the served catalog knows, for + * example one the platform added after this release. Reads a catalog loaded + * for the current scope, never the shipped list, and never a retired model. + * Undefined when no served catalog has loaded or it does not name the alias. + */ +export function resolveServedVeryfrontCloudAlias(alias: string): string | undefined { + if (alias.includes("/") || loadedCatalog() === undefined) return undefined; + const model = servedIndex().byShortId.get(alias); + if (!model || isRetiredVeryfrontCloudModelId(model.modelId)) return undefined; + return model.modelId; +} + /** * Resolve a model ID or short alias to a provider-qualified model ID. * diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index a50e3bea14..610a3a63e9 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -1941,6 +1941,25 @@ describe("provider/veryfront-cloud served catalog loading", () => { assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); }); + it("lets one caller give up without deciding for a concurrent caller", async () => { + setCloudBootstrap(); + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const requests = installGateway(chatOnlyCatalog, gate); + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + + const controller = new AbortController(); + const abandoned = model.prepare?.(controller.signal); + const waiting = streamOnce(model); + controller.abort(); + await abandoned; + release?.(); + await waiting; + + // The waiting caller used the served catalog, fetched once for both. + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + }); + it("forwards metadata to the model rebuilt from the catalog", async () => { setCloudBootstrap(); installGateway(() => diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index 18229202b6..143d698ddb 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -290,7 +290,6 @@ function createVeryfrontCloudModelInternal( let facts = readFacts(); const built = build(); let current = built; - let preparing: Promise | undefined; let isSettled = false; const rebuildIfChanged = (settle: boolean): ModelRuntime => { const next = readFacts(); @@ -308,25 +307,16 @@ function createVeryfrontCloudModelInternal( if (!isVeryfrontCloudCatalogFresh(catalogScope)) return undefined; return rebuildIfChanged(true); }; - const prepare = async (abortSignal?: AbortSignal): Promise => { + // Each caller waits on its own signal. Concurrent callers share only the + // underlying catalog request (one per credentials and project, in the + // catalog client), so one caller giving up never decides for another. + const ready = async (abortSignal?: AbortSignal): Promise => { const catalog = await loadVeryfrontCloudCatalog({ ...catalogScope, ...(abortSignal ? { signal: abortSignal } : {}), }); return rebuildIfChanged(catalog !== undefined); }; - const ready = async (abortSignal?: AbortSignal): Promise => { - preparing ??= prepare(abortSignal); - try { - return await preparing; - } catch (error) { - // A failed build is retried on the next call rather than cached. - preparing = undefined; - throw error; - } finally { - if (!isSettled) preparing = undefined; - } - }; const wrapped = withServedCatalog(built, () => current, settled, ready); registerVeryfrontCloudModelFacts(wrapped, () => facts); return wrapped; diff --git a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts index c9990ac782..7e7d543673 100644 --- a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts +++ b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts @@ -9,6 +9,7 @@ import { resolveVeryfrontCloudModelId } from "#veryfront/provider/veryfront-clou import { createVeryfrontCloudInferenceModel } from "#veryfront/provider/veryfront-cloud/provider.ts"; import type { ModelRuntime } from "#veryfront/provider/types.ts"; import { resolveAgentModelTransport } from "#veryfront/agent/runtime/model-transport.ts"; +import { resolveRuntimeModel } from "#veryfront/agent/runtime/model-resolution.ts"; import { createDefaultHostedChatRuntime } from "#veryfront/agent/hosted/default-chat-runtime.ts"; import type { RemoteMCPToolSourceConfig, @@ -125,6 +126,37 @@ describe("served-only models from a cold process", () => { assertEquals(captured.map(({ path }) => path), ["/ai/models"]); }); + it("routes a served-only short alias through Veryfront Cloud from the agent transport", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + const captured = installGateway(); + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config: { model: SERVED_ONLY_ALIAS, system: "You are concise." }, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(transport.resolvedModelString, `veryfront-cloud/${SERVED_ONLY_MODEL}`); + assertEquals(transport.requestedModel, SERVED_ONLY_MODEL); + assertEquals(captured.map(({ path }) => path), ["/ai/models"]); + }); + + it("resolves a served-only alias in runtime model resolution only once a catalog loaded", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + installGateway(); + + // Cold: nothing names the alias, so it is left as written. + assertEquals(resolveRuntimeModel(SERVED_ONLY_ALIAS), SERVED_ONLY_ALIAS); + await loadVeryfrontCloudModelCatalog(); + assertEquals(resolveRuntimeModel(SERVED_ONLY_ALIAS), `veryfront-cloud/${SERVED_ONLY_MODEL}`); + // A known alias keeps its meaning. + assertEquals(resolveRuntimeModel("sonnet"), "veryfront-cloud/anthropic/claude-sonnet-4-6"); + }); + it("resolves a served-only alias once the ambient catalog is loaded", async () => { setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); From 7f5f2fffa9640c3d3252a91df54e84f041be6a2b Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 10:42:13 +0200 Subject: [PATCH 07/17] fix(provider): refuse models only against a fresh catalog; route served new providers - A model the catalog does not list is refused only against a fresh served catalog, or the shipped list while none has loaded. A stale catalog never refuses: construction defers the check, and the first async step waits for the refresh before settling. - Explicit loads for decisions wait for the refresh of a stale entry. - Runtime model resolution treats any model the loaded served catalog lists as a Veryfront Cloud candidate, including a provider this package does not name. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- src/agent/runtime/model-resolution.ts | 15 ++++- .../veryfront-cloud/catalog-client.ts | 25 ++++++-- src/provider/veryfront-cloud/model-catalog.ts | 31 +++++++++- src/provider/veryfront-cloud/provider.test.ts | 62 +++++++++++++++++++ src/provider/veryfront-cloud/provider.ts | 10 ++- src/provider/veryfront-cloud/shared.ts | 7 ++- ...veryfront-cloud-served-only-models.test.ts | 28 +++++++++ 7 files changed, 165 insertions(+), 13 deletions(-) diff --git a/src/agent/runtime/model-resolution.ts b/src/agent/runtime/model-resolution.ts index 86f0aba7ab..771607d34f 100644 --- a/src/agent/runtime/model-resolution.ts +++ b/src/agent/runtime/model-resolution.ts @@ -5,7 +5,9 @@ import { getOpenAIEnvConfig, } from "#veryfront/config/env.ts"; import { + canVeryfrontCloudCatalogRefuse, createRetiredVeryfrontCloudModelError, + isListedInServedVeryfrontCloudCatalog, isRetiredVeryfrontCloudModelId, isSupportedMistralModelId, isVeryfrontCloudCatalogLoaded, @@ -144,7 +146,8 @@ function listAvailableDirectProviders(): string[] { } function isSupportedHostedMistralModel(modelId: string): boolean { - return isSupportedMistralModelId(`mistral/${modelId}`); + // A stale served catalog cannot refuse: the platform answers for the model. + return !canVeryfrontCloudCatalogRefuse() || isSupportedMistralModelId(`mistral/${modelId}`); } function isUnsupportedVeryfrontCloudMistralModel(modelId: string): boolean { @@ -152,7 +155,7 @@ function isUnsupportedVeryfrontCloudMistralModel(modelId: string): boolean { // the shipped list cannot know a model the platform added since, and the // model checks its own catalog once that has loaded. return modelId.startsWith("veryfront-cloud/mistral/") && isVeryfrontCloudCatalogLoaded() && - !isSupportedMistralModelId(modelId); + canVeryfrontCloudCatalogRefuse() && !isSupportedMistralModelId(modelId); } function normalizeVeryfrontCloudRuntimeModel(modelId: string): string { @@ -249,7 +252,13 @@ export function resolveRuntimeModel(model?: string): string { const provider = configuredModel.slice(0, slashIndex); const modelId = configuredModel.slice(slashIndex + 1); - if (!HOSTED_PROVIDER_NAMES.has(provider) || !modelId) { + // A provider this package names, or any model the loaded served catalog + // lists (a provider the platform added since), is a Veryfront Cloud candidate. + if ( + !modelId || + (!HOSTED_PROVIDER_NAMES.has(provider) && + !isListedInServedVeryfrontCloudCatalog(configuredModel)) + ) { return configuredModel; } diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index 6aae5d2f06..78360edd54 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -87,6 +87,12 @@ export interface VeryfrontCloudCatalogLoadOptions extends VeryfrontCloudCatalogS readonly signal?: AbortSignal; /** Longest this caller waits for a request in flight before it goes on without it. */ readonly maxWaitMs?: number; + /** + * Wait for the refresh of a stale entry instead of answering with it at once, + * for a caller about to make a decision the catalog must be current for. The + * stale entry still answers when the refresh fails or the wait ends. + */ + readonly fresh?: boolean; } interface CatalogEntry { @@ -354,14 +360,23 @@ export function loadVeryfrontCloudCatalog( ...(options.projectSlug ? { projectSlug: options.projectSlug } : {}), }); // Stale while revalidate: the stale entry answers now, the refresh replaces it. - if (entry) return Promise.resolve(entry.catalog); - return waitFor(request, undefined, options.signal, options.maxWaitMs); + if (entry && !options.fresh) return Promise.resolve(entry.catalog); + return waitFor(request, entry?.catalog, options.signal, options.maxWaitMs).then((catalog) => + catalog ?? entry?.catalog + ); } -/** Whether a load for this scope would answer from a fresh cache entry, without a request. */ -export function isVeryfrontCloudCatalogFresh(scope: VeryfrontCloudCatalogScope): boolean { +/** + * Whether the catalog for a scope is fresh: loaded within the TTL. Without a + * scope, reads the one {@link withVeryfrontCloudCatalogScope} names. A stale + * catalog may miss models the platform has enabled since, so it must not be + * used to refuse one. + */ +export function isVeryfrontCloudCatalogFresh(scope?: VeryfrontCloudCatalogScope): boolean { if (seeded) return true; - const entry = entries.get(cacheKey(scope)); + const key = scope ? cacheKey(scope) : activeKey; + if (key === undefined) return false; + const entry = entries.get(key); return entry !== undefined && now() - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS; } diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index 48b252d01f..7e0928ef0b 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -5,6 +5,7 @@ import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.t import type { ModelRuntime } from "../types.ts"; import { hasActiveVeryfrontCloudCatalogScope, + isVeryfrontCloudCatalogFresh, peekVeryfrontCloudCatalog, type VeryfrontCloudCatalog, type VeryfrontCloudCatalogModel, @@ -264,6 +265,30 @@ function loadedCatalog(): VeryfrontCloudCatalog | undefined { return scope ? peekVeryfrontCloudCatalog(scope) : peekVeryfrontCloudCatalog(); } +/** + * Whether the catalog reads use may refuse a model it does not list: a fresh + * served catalog, or the shipped list while none has loaded. A stale served + * catalog may miss a model the platform has enabled since, so a refusal waits + * until it is refreshed and the platform answers for the model meanwhile. + */ +export function canVeryfrontCloudCatalogRefuse(): boolean { + if (loadedCatalog() === undefined) return true; + if (hasActiveVeryfrontCloudCatalogScope()) return isVeryfrontCloudCatalogFresh(); + const scope = ambientScope(); + return scope ? isVeryfrontCloudCatalogFresh(scope) : isVeryfrontCloudCatalogFresh(); +} + +/** + * Whether the served catalog loaded for the current scope lists this model, + * under any accepted spelling. A listed model is a Veryfront Cloud candidate + * even when its provider is one this package does not name. + */ +export function isListedInServedVeryfrontCloudCatalog(modelId: string): boolean { + if (loadedCatalog() === undefined) return false; + const model = servedIndex().byKey.get(canonicalVeryfrontCloudModelKey(modelId)); + return model !== undefined && !isRetiredVeryfrontCloudModelId(model.modelId); +} + /** * Whether a served catalog has loaded for the scope reads use right now. While * it has not, reads fall back to the shipped list, which cannot know models @@ -735,6 +760,7 @@ export function resolveVeryfrontCloudModelId(alias?: string): string { // list so callers get a clear error rather than a gateway-side failure. if ( isMistralModelId(requestedModel) && + canVeryfrontCloudCatalogRefuse() && !isSupportedMistralModelId(requestedModel) ) { throw NOT_SUPPORTED.create({ @@ -790,7 +816,10 @@ export function resolveVeryfrontCloudGatewayModelId( // Unsupported Mistral ids are passed through unprefixed (not routed through // the Veryfront Cloud gateway prefix). - if (isMistralModelId(modelId) && !isSupportedMistralModelId(modelId)) { + if ( + isMistralModelId(modelId) && canVeryfrontCloudCatalogRefuse() && + !isSupportedMistralModelId(modelId) + ) { return modelId; } diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index 610a3a63e9..9d4fe51e7c 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -6,7 +6,9 @@ import { seedServedCatalogForTests, servedCatalogPayload } from "./catalog-clien import { __resetVeryfrontCloudCatalogForTests, __setVeryfrontCloudCatalogClockForTests, + __setVeryfrontCloudCatalogForScopeForTests, VERYFRONT_CLOUD_CATALOG_RETRY_MS, + VERYFRONT_CLOUD_CATALOG_TTL_MS, } from "./catalog-client.ts"; import { agent } from "#veryfront/agent"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; @@ -1960,6 +1962,66 @@ describe("provider/veryfront-cloud served catalog loading", () => { assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); }); + it("does not refuse a model against a stale catalog, and serves it after the refresh", async () => { + setCloudBootstrap(); + let now = 1_000_000; + __setVeryfrontCloudCatalogClockForTests(() => now); + const scope = { + apiBaseUrl: "https://api.veryfront.com", + apiToken: "vf_test_provider", + projectSlug: "provider-test-project", + }; + // A catalog loaded before the platform enabled mistral/mistral-new. + __setVeryfrontCloudCatalogForScopeForTests(scope, { + models: [{ + id: "mistral-small-2503", + modelId: "mistral/mistral-small-2503", + provider: "mistral", + surface: "openai", + operations: ["chat-completions"], + }], + }); + now += VERYFRONT_CLOUD_CATALOG_TTL_MS; + const requests = installGateway(() => + Response.json({ + models: [{ + id: "mistral-new", + modelId: "mistral/mistral-new", + provider: "mistral", + surface: "openai", + operations: ["chat-completions"], + }], + }) + ); + + const model = resolveModel("veryfront-cloud/mistral/mistral-new") as ModelRuntime; + await streamOnce(model); + + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + }); + + it("still refuses a model a fresh catalog does not list", () => { + setCloudBootstrap(); + __setVeryfrontCloudCatalogForScopeForTests({ + apiBaseUrl: "https://api.veryfront.com", + apiToken: "vf_test_provider", + projectSlug: "provider-test-project", + }, { + models: [{ + id: "mistral-small-2503", + modelId: "mistral/mistral-small-2503", + provider: "mistral", + surface: "openai", + }], + }); + + assertThrows( + () => resolveModel("veryfront-cloud/mistral/mistral-new"), + Error, + 'Unsupported Mistral model "mistral/mistral-new"', + ); + }); + it("forwards metadata to the model rebuilt from the catalog", async () => { setCloudBootstrap(); installGateway(() => diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index 143d698ddb..f35b7ac7ae 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -32,7 +32,6 @@ import { import { isVeryfrontCloudCatalogFresh, loadVeryfrontCloudCatalog, - peekVeryfrontCloudCatalog, withVeryfrontCloudCatalogScope, } from "./catalog-client.ts"; @@ -286,7 +285,9 @@ function createVeryfrontCloudModelInternal( const parsed = parseVeryfrontCloudModelId(modelId, "language", { catalogChecks: false }); assertVeryfrontCloudModelListed(parsed.provider, parsed.modelId); }); - if (peekVeryfrontCloudCatalog(catalogScope) !== undefined) assertListed(); + // Refuse at construction only against a fresh catalog: a stale one may miss a + // model enabled since, so the check waits for the refresh on the first call. + if (isVeryfrontCloudCatalogFresh(catalogScope)) assertListed(); let facts = readFacts(); const built = build(); let current = built; @@ -313,9 +314,12 @@ function createVeryfrontCloudModelInternal( const ready = async (abortSignal?: AbortSignal): Promise => { const catalog = await loadVeryfrontCloudCatalog({ ...catalogScope, + fresh: true, ...(abortSignal ? { signal: abortSignal } : {}), }); - return rebuildIfChanged(catalog !== undefined); + // Settle (and run the listing check) only on a fresh catalog. A stale one + // answers this call, and a later call tries the refresh again. + return rebuildIfChanged(catalog !== undefined && isVeryfrontCloudCatalogFresh(catalogScope)); }; const wrapped = withServedCatalog(built, () => current, settled, ready); registerVeryfrontCloudModelFacts(wrapped, () => facts); diff --git a/src/provider/veryfront-cloud/shared.ts b/src/provider/veryfront-cloud/shared.ts index c222856adf..ca11a8b383 100644 --- a/src/provider/veryfront-cloud/shared.ts +++ b/src/provider/veryfront-cloud/shared.ts @@ -18,6 +18,7 @@ import { markCurrentVeryfrontCloudBillingGroupUsed, } from "./context.ts"; import { + canVeryfrontCloudCatalogRefuse, createRetiredVeryfrontCloudModelError, isRetiredVeryfrontCloudModelId, isSupportedMistralModelId, @@ -270,7 +271,10 @@ export function parseVeryfrontCloudModelId( * a clear error rather than a gateway-side failure. */ export function assertVeryfrontCloudModelListed(provider: string, upstreamModelId: string): void { - if (provider === "mistral" && !isSupportedMistralModelId(`mistral/${upstreamModelId}`)) { + if ( + provider === "mistral" && canVeryfrontCloudCatalogRefuse() && + !isSupportedMistralModelId(`mistral/${upstreamModelId}`) + ) { throw toError( createError({ type: "config", @@ -338,6 +342,7 @@ export async function loadVeryfrontCloudModelCatalog( return false; } const catalog = await loadVeryfrontCloudCatalog({ + fresh: true, apiBaseUrl: bootstrap.apiBaseUrl, apiToken: bootstrap.apiToken, ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), diff --git a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts index 7e7d543673..53a0e15ba8 100644 --- a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts +++ b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts @@ -27,6 +27,8 @@ import { defineSchema } from "#veryfront/schemas/define.ts"; const SERVED_ONLY_MODEL = "mistral/mistral-medium-2609"; const SERVED_ONLY_ALIAS = "medium"; +const NEW_PROVIDER_MODEL = "acme-labs/m1"; +const NEW_PROVIDER_ALIAS = "acme-m1"; const AMBIENT_TOKEN = "vf_ambient_token"; const RUN_TOKEN = "vf_run_token"; @@ -46,6 +48,14 @@ function catalogFor(authorization: string | null): Response { operations: ["chat-completions"], aliases: [SERVED_ONLY_ALIAS], capabilities: {}, + }, { + id: "m1", + modelId: NEW_PROVIDER_MODEL, + provider: "acme-labs", + surface: "openai", + operations: ["chat-completions"], + aliases: [NEW_PROVIDER_ALIAS], + capabilities: {}, }] : [], }); @@ -157,6 +167,24 @@ describe("served-only models from a cold process", () => { assertEquals(resolveRuntimeModel("sonnet"), "veryfront-cloud/anthropic/claude-sonnet-4-6"); }); + for (const model of [NEW_PROVIDER_MODEL, NEW_PROVIDER_ALIAS]) { + it(`routes ${model}, served for a provider this package does not name, through Veryfront Cloud`, async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + installGateway(); + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config: { model, system: "You are concise." }, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(transport.resolvedModelString, `veryfront-cloud/${NEW_PROVIDER_MODEL}`); + }); + } + it("resolves a served-only alias once the ambient catalog is loaded", async () => { setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); From 6a39b6c300c31ab12e7c0cf874115ed0cdfa2c72 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 10:51:21 +0200 Subject: [PATCH 08/17] docs(changelog): describe served-only aliases and stale-catalog refusals Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- CHANGELOG.md | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2790c242cf..287c45cdde 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -40,6 +40,14 @@ unchanged. call. - A model keeps the facts it settled with for its lifetime. A catalog refreshed later applies to models constructed after the refresh. +- Agents resolve a short alias or a provider the platform added after this + release through Veryfront Cloud once the catalog has loaded, after the + built-in aliases, so a bare vendor model name keeps its meaning for your own + provider key. +- A model the catalog does not list is refused only against a catalog loaded + within the last five minutes, or the shipped list before any has loaded. A + model enabled since the catalog was cached is not refused; the catalog is + refreshed first. - `loadVeryfrontCloudModelCatalog()` loads the catalog for the Veryfront Cloud credentials in effect, so synchronous helpers such as `resolveVeryfrontCloudModelId("opus")` and From 10c0991b26ede315237983f1e24791b5dd8b8de4 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 11:10:21 +0200 Subject: [PATCH 09/17] fix(agent): let credential-free contexts read the run's catalog by key Hosted project tools, including invoke_agent child tools, run in a context without the run's credential, so their synchronous model reads fell back to the shipped aliases. The context now carries the run's non-secret catalog scope key (API base URL, project and a per-process salted credential fingerprint) and synchronous reads use it; the credential never re-enters the context. A context whose run loaded nothing keeps the shipped facts. Part of veryfront/veryfront-issue-inbox#1573. Co-Authored-By: Claude --- src/agent/hosted/default-chat-runtime.ts | 8 ++ src/platform/cloud/resolver.ts | 9 +- .../veryfront-cloud/catalog-client.ts | 43 ++++++- src/provider/veryfront-cloud/context.ts | 6 + src/provider/veryfront-cloud/model-catalog.ts | 23 +++- ...veryfront-cloud-served-only-models.test.ts | 117 +++++++++++++++++- 6 files changed, 193 insertions(+), 13 deletions(-) diff --git a/src/agent/hosted/default-chat-runtime.ts b/src/agent/hosted/default-chat-runtime.ts index b12526cad7..3b9f305608 100644 --- a/src/agent/hosted/default-chat-runtime.ts +++ b/src/agent/hosted/default-chat-runtime.ts @@ -12,6 +12,7 @@ import { runWithRequestContext as runWithProjectRequestContext, } from "#veryfront/platform/adapters/fs/veryfront/request-context.ts"; import { + currentVeryfrontCloudCatalogScopeKey, resolveVeryfrontCloudModelId, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudReasoningOption, @@ -408,12 +409,19 @@ function withoutHostedCredentials(input: { cloudContext: VeryfrontCloudContext; operation: () => Promise; }): Promise { + // The run's catalog is named by its non-secret scope key, so model reads in + // project code use the catalog the run loaded without holding its credential. + const catalogScopeKey = runWithVeryfrontCloudContext( + input.cloudContext, + currentVeryfrontCloudCatalogScopeKey, + ); const publicCloudContext: VeryfrontCloudContext = { apiBaseUrl: input.cloudContext.apiBaseUrl, projectSlug: input.cloudContext.projectSlug, serviceLayer: input.cloudContext.serviceLayer, billingGroupId: input.cloudContext.billingGroupId, billingGroupUsed: input.cloudContext.billingGroupUsed, + ...(catalogScopeKey ? { catalogScopeKey } : {}), }; const runWithPublicCloudContext = () => runWithVeryfrontCloudContextAsync(publicCloudContext, input.operation); diff --git a/src/platform/cloud/resolver.ts b/src/platform/cloud/resolver.ts index 9fdaef2951..c0acbc0a55 100644 --- a/src/platform/cloud/resolver.ts +++ b/src/platform/cloud/resolver.ts @@ -1,5 +1,8 @@ import { getRuntimeRequestContext } from "#veryfront/platform/runtime-request-context.ts"; -import { peekVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { + peekVeryfrontCloudCatalog, + type VeryfrontCloudCatalogScopeKey, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { getHostEnv, getHostEnvExcludingEnvFile, @@ -238,12 +241,16 @@ export function isVeryfrontCloudEnabled(): boolean { */ export function getDefaultVeryfrontCloudModel(): string { const bootstrap = getVeryfrontCloudBootstrap(); + // A context without credentials may carry the key of the run's catalog. + const carriedKey = getCurrentVeryfrontCloudContext()?.catalogScopeKey; const served = (bootstrap.apiToken ? peekVeryfrontCloudCatalog({ apiBaseUrl: bootstrap.apiBaseUrl, apiToken: bootstrap.apiToken, ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), }) + : carriedKey + ? peekVeryfrontCloudCatalog(carriedKey as VeryfrontCloudCatalogScopeKey) : peekVeryfrontCloudCatalog())?.defaultModelId; return normalizeCloudModelString( getHostEnv("VERYFRONT_DEFAULT_MODEL"), diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index d740e19578..d1f5fd37c1 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -184,11 +184,21 @@ export function parseVeryfrontCloudCatalog(payload: unknown): VeryfrontCloudCata * Non-reversible fingerprint of a credential, so entries for different * credentials never share a key and the key never holds the credential. */ +/** + * Per-process salt for credential fingerprints, so a fingerprint means nothing + * outside this process and cannot be matched against a known credential. + */ +const FINGERPRINT_SALT = Array.from( + crypto.getRandomValues(new Uint8Array(16)), + (byte) => byte.toString(16).padStart(2, "0"), +).join(""); + function credentialFingerprint(token: string): string { let a = 0x811c9dc5; let b = 0x01000193; - for (let index = 0; index < token.length; index++) { - const code = token.charCodeAt(index); + const salted = `${FINGERPRINT_SALT}:${token}`; + for (let index = 0; index < salted.length; index++) { + const code = salted.charCodeAt(index); a = Math.imul(a ^ code, 0x01000193) >>> 0; b = Math.imul(b ^ code, 0x85ebca6b) >>> 0; } @@ -227,6 +237,25 @@ function rememberFailure(key: string, at: number): void { evictOldest(failedAt); } +/** + * A catalog scope key: the cache key for a scope. It carries the API base URL, + * the project and a salted credential fingerprint, never the credential, so a + * context that must not hold the credential can still name the catalog loaded + * for it. + */ +export type VeryfrontCloudCatalogScopeKey = string & { readonly __catalogScopeKey: true }; + +/** The non-secret scope key for a scope. */ +export function veryfrontCloudCatalogScopeKey( + scope: VeryfrontCloudCatalogScope, +): VeryfrontCloudCatalogScopeKey { + return cacheKey(scope) as VeryfrontCloudCatalogScopeKey; +} + +function keyOf(scope: VeryfrontCloudCatalogScope | VeryfrontCloudCatalogScopeKey): string { + return typeof scope === "string" ? scope : cacheKey(scope); +} + function cacheKey(scope: VeryfrontCloudCatalogScope): string { return `${scope.apiBaseUrl}\n${scope.projectSlug ?? ""}\n${ credentialFingerprint(scope.apiToken) @@ -372,9 +401,11 @@ export function loadVeryfrontCloudCatalog( * catalog may miss models the platform has enabled since, so it must not be * used to refuse one. */ -export function isVeryfrontCloudCatalogFresh(scope?: VeryfrontCloudCatalogScope): boolean { +export function isVeryfrontCloudCatalogFresh( + scope?: VeryfrontCloudCatalogScope | VeryfrontCloudCatalogScopeKey, +): boolean { if (seeded) return true; - const key = scope ? cacheKey(scope) : activeKey; + const key = scope ? keyOf(scope) : activeKey; if (key === undefined) return false; const entry = entries.get(key); return entry !== undefined && now() - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS; @@ -409,10 +440,10 @@ export function hasActiveVeryfrontCloudCatalogScope(): boolean { * names, and undefined outside it. */ export function peekVeryfrontCloudCatalog( - scope?: VeryfrontCloudCatalogScope, + scope?: VeryfrontCloudCatalogScope | VeryfrontCloudCatalogScopeKey, ): VeryfrontCloudCatalog | undefined { if (seeded) return seeded; - const key = scope ? cacheKey(scope) : activeKey; + const key = scope ? keyOf(scope) : activeKey; if (key === undefined) return undefined; const entry = entries.get(key); if (!entry) return undefined; diff --git a/src/provider/veryfront-cloud/context.ts b/src/provider/veryfront-cloud/context.ts index 5f8a95ff0e..2a32842de6 100644 --- a/src/provider/veryfront-cloud/context.ts +++ b/src/provider/veryfront-cloud/context.ts @@ -14,6 +14,12 @@ export interface VeryfrontCloudContext { billingGroupRequestAdmitted?: boolean; projectSlug?: string; serviceLayer?: string; + /** + * Names the model catalog loaded for credentials this context does not hold, + * so synchronous model reads in a credential-free context use the same + * catalog as the run. It carries no credential. + */ + catalogScopeKey?: string; } const veryfrontCloudContextStorage = new AsyncLocalStorage(); diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index 7e0928ef0b..e62077affc 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -3,12 +3,15 @@ import { isOpenAIReasoningModel } from "../shared/openai-reasoning.ts"; import { getVeryfrontCloudBootstrap } from "#veryfront/platform/cloud/resolver.ts"; import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; import type { ModelRuntime } from "../types.ts"; +import { getCurrentVeryfrontCloudContext } from "./context.ts"; import { hasActiveVeryfrontCloudCatalogScope, isVeryfrontCloudCatalogFresh, peekVeryfrontCloudCatalog, type VeryfrontCloudCatalog, type VeryfrontCloudCatalogModel, + type VeryfrontCloudCatalogScopeKey, + veryfrontCloudCatalogScopeKey, } from "./catalog-client.ts"; import { SHIPPED_VERYFRONT_CLOUD_CATALOG } from "./model-catalog.deprecated.ts"; @@ -257,12 +260,24 @@ function ambientScope(): }; } +/** + * @internal The non-secret key of the catalog synchronous reads use outside a + * model build: the ambient credentials' scope, or, in a context that does not + * hold credentials, the scope key it carries. Undefined when neither applies. + */ +export function currentVeryfrontCloudCatalogScopeKey(): VeryfrontCloudCatalogScopeKey | undefined { + const scope = ambientScope(); + if (scope) return veryfrontCloudCatalogScopeKey(scope); + const carried = getCurrentVeryfrontCloudContext()?.catalogScopeKey; + return carried ? carried as VeryfrontCloudCatalogScopeKey : undefined; +} + /** The served catalog loaded for the scope reads use, or undefined before it loads. */ function loadedCatalog(): VeryfrontCloudCatalog | undefined { if (hasActiveVeryfrontCloudCatalogScope()) return peekVeryfrontCloudCatalog(); - const scope = ambientScope(); + const key = currentVeryfrontCloudCatalogScopeKey(); // A scope-less read still sees a catalog fixed by a test hook. - return scope ? peekVeryfrontCloudCatalog(scope) : peekVeryfrontCloudCatalog(); + return key ? peekVeryfrontCloudCatalog(key) : peekVeryfrontCloudCatalog(); } /** @@ -274,8 +289,8 @@ function loadedCatalog(): VeryfrontCloudCatalog | undefined { export function canVeryfrontCloudCatalogRefuse(): boolean { if (loadedCatalog() === undefined) return true; if (hasActiveVeryfrontCloudCatalogScope()) return isVeryfrontCloudCatalogFresh(); - const scope = ambientScope(); - return scope ? isVeryfrontCloudCatalogFresh(scope) : isVeryfrontCloudCatalogFresh(); + const key = currentVeryfrontCloudCatalogScopeKey(); + return key ? isVeryfrontCloudCatalogFresh(key) : isVeryfrontCloudCatalogFresh(); } /** diff --git a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts index 53a0e15ba8..77a362ab34 100644 --- a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts +++ b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts @@ -1,9 +1,18 @@ import "#veryfront/schemas/_test-setup.ts"; -import { assertEquals, assertRejects } from "#veryfront/testing/assert.ts"; +import { assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; -import { clearModelProviders, loadVeryfrontCloudModelCatalog } from "#veryfront/provider"; +import { + clearModelProviders, + loadVeryfrontCloudModelCatalog, + registerModelProvider, +} from "#veryfront/provider"; +import { + getCurrentVeryfrontCloudContext, + runWithVeryfrontCloudContext, + type VeryfrontCloudContext, +} from "#veryfront/provider/veryfront-cloud/context.ts"; import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { resolveVeryfrontCloudModelId } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { createVeryfrontCloudInferenceModel } from "#veryfront/provider/veryfront-cloud/provider.ts"; @@ -223,6 +232,110 @@ describe("served-only models from a cold process", () => { }); describe("read in the scope the catalog was loaded for", () => { + it("resolves a served-only alias in a credential-free hosted tool context, without the credential", async () => { + const captured = installGateway(true); + let resolvedInTool: string | undefined; + let toolContext: VeryfrontCloudContext | undefined; + let modelCalls = 0; + registerModelProvider("test", () => ({ + provider: "test", + modelId: "test/stripped-context", + doGenerate: () => Promise.reject(new Error("unused")), + doStream() { + modelCalls++; + return Promise.resolve({ + stream: new ReadableStream({ + start(controller) { + if (modelCalls === 1) { + controller.enqueue({ + type: "tool-call", + toolCallId: "child-1", + toolName: "invoke_child", + input: {}, + }); + controller.enqueue({ type: "finish", finishReason: "tool-calls", usage: {} }); + } else { + controller.enqueue({ type: "text-delta", text: "done" }); + controller.enqueue({ type: "finish", finishReason: "stop", usage: {} }); + } + controller.close(); + }, + }), + }); + }, + })); + + const runtime = await createDefaultHostedChatRuntime({ + sourceIntegrationPolicy: { schemaVersion: 1, mode: "unrestricted" }, + options: { + projectId: "project-1", + projectSlug: "run-project", + authToken: RUN_TOKEN, + instructions: "Invoke the child.", + model: "test/stripped-context", + allowedTools: ["invoke_child"], + }, + config: { + apiUrl: "https://api.veryfront.com", + apiMcpUrl: "https://api.veryfront.com/mcp", + }, + buildLocalTools: () => ({ + invoke_child: { + description: "Resolve a child model the way invoke_agent does", + inputSchema: defineSchema((v) => v.object({}))(), + execute: () => { + toolContext = getCurrentVeryfrontCloudContext(); + // invoke_agent resolves the child's model with this function. + resolvedInTool = resolveVeryfrontCloudModelId(SERVED_ONLY_ALIAS); + return { ok: true }; + }, + }, + }), + createRemoteToolSource: emptyRemoteSource, + preloadLatestConversationUserText: false, + }); + try { + const result = await runtime.agent.stream({ + messages: [], + abortSignal: new AbortController().signal, + }); + for await (const _chunk of result.toUIMessageStream()) { + // Consume the tool round trip. + } + } finally { + await runtime.cleanup?.(); + } + + assertEquals(resolvedInTool, SERVED_ONLY_MODEL); + assertEquals(toolContext?.apiToken, undefined); + assertEquals(typeof toolContext?.catalogScopeKey, "string"); + assertEquals(toolContext?.catalogScopeKey?.includes(RUN_TOKEN), false); + assertEquals(JSON.stringify(toolContext).includes(RUN_TOKEN), false); + assertEquals( + captured.filter(({ path }) => path === "/ai/models").map(({ authorization }) => + authorization + ), + [`Bearer ${RUN_TOKEN}`], + ); + }); + + it("falls back to the shipped aliases in a credential-free context whose run loaded nothing", () => { + const stripped: VeryfrontCloudContext = { + apiBaseUrl: "https://api.veryfront.com", + projectSlug: "run-project", + serviceLayer: "cloud", + }; + + runWithVeryfrontCloudContext(stripped, () => { + assertEquals(resolveVeryfrontCloudModelId("opus"), "anthropic/claude-opus-4-8"); + assertThrows( + () => resolveVeryfrontCloudModelId(SERVED_ONLY_ALIAS), + Error, + "Unknown model alias", + ); + }); + }); + it("resolves a hosted alias from the run's catalog, not the ambient one", async () => { setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); setEnv("VERYFRONT_PROJECT_SLUG", "ambient-project"); From a75ffb4dc65306a73b763ccf8d304c50f410fb2c Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 11:50:25 +0200 Subject: [PATCH 10/17] fix(provider): settle a served model before building or describing it - generateText and streamText settle a Veryfront Cloud model before building and validating its call options, so response formats, tools and provider option keys are checked against the protocol the model is sent with. - The agent transport settles the model before choosing its provider option key. - Hosted application models read metadata from the model they call, and the executor broker settles models (bounded wait) before describing them. --- .../hosted/application-model-resolver.ts | 36 +++++++++++++++---- src/agent/hosted/executor-model-bridge.ts | 32 ++++++++++++++++- src/agent/runtime/model-transport.ts | 11 ++++++ src/provider/veryfront-cloud/provider.test.ts | 34 ++++++++++++++++++ src/runtime/runtime-bridge.ts | 25 ++++++++----- 5 files changed, 122 insertions(+), 16 deletions(-) diff --git a/src/agent/hosted/application-model-resolver.ts b/src/agent/hosted/application-model-resolver.ts index 1865eef432..98ad689170 100644 --- a/src/agent/hosted/application-model-resolver.ts +++ b/src/agent/hosted/application-model-resolver.ts @@ -6,6 +6,10 @@ import { type VeryfrontCloudContext, } from "#veryfront/provider/veryfront-cloud/context.ts"; import { createVeryfrontCloudModel } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import { + readVeryfrontCloudModelFacts, + registerVeryfrontCloudModelFacts, +} from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { requireSecureInferenceApiBaseUrl } from "#veryfront/provider/veryfront-cloud/shared.ts"; import { type AgentModelRuntimeResolver, @@ -136,14 +140,30 @@ export function createHostedApplicationModelResolver(input: { }) ); const reconcile = model._reconcileProviderMetadata; + // Metadata is read from the model on each access: a Veryfront Cloud model + // may settle a different protocol and capabilities on its first async step. const proxy: ModelRuntime = Object.freeze({ - specificationVersion: model.specificationVersion, - provider: model.provider, - modelProvider: model.modelProvider, - modelId: model.modelId, - executionMode: model.executionMode, - runtimeCapabilities: model.runtimeCapabilities, - _generateViaStream: model._generateViaStream, + get specificationVersion() { + return model.specificationVersion; + }, + get provider() { + return model.provider; + }, + get modelProvider() { + return model.modelProvider; + }, + get modelId() { + return model.modelId; + }, + get executionMode() { + return model.executionMode; + }, + get runtimeCapabilities() { + return model.runtimeCapabilities; + }, + get _generateViaStream() { + return model._generateViaStream; + }, async prepare(abortSignal?: AbortSignal) { await run(callScope(abortSignal), (signal) => model.prepare?.(signal)); }, @@ -188,6 +208,8 @@ export function createHostedApplicationModelResolver(input: { } : {}), }); + // The proxy reports the facts of the model it calls. + registerVeryfrontCloudModelFacts(proxy, () => readVeryfrontCloudModelFacts(model)!); models.set(id, proxy); return proxy; }; diff --git a/src/agent/hosted/executor-model-bridge.ts b/src/agent/hosted/executor-model-bridge.ts index 53d29a0051..ff96cea699 100644 --- a/src/agent/hosted/executor-model-bridge.ts +++ b/src/agent/hosted/executor-model-bridge.ts @@ -1,4 +1,5 @@ const hasOwn = Object.hasOwn; +import { readVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import type { ModelRuntime, ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { createPrivateReadableStream } from "#veryfront/security/private-stream.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; @@ -132,9 +133,15 @@ export function createExecutorModelBroker(options: { return new Map([ ["model.metadata", { mode: "unary", - handle(input, context) { + async handle(input, context) { parseExecutorModelData(getExecutorModelEmptySchema(), input); context.signal.throwIfAborted(); + // A Veryfront Cloud model settles its protocol and capabilities on its + // first async step; settle each before describing it to the executor. + await Promise.all( + [...allowed].map((id) => settleForMetadata(getModel(id), context.signal)), + ); + context.signal.throwIfAborted(); const metadata = [...allowed].map((id) => modelMetadata(id, getModel(id))); return executorModelJson( parseExecutorModelData(getExecutorModelMetadataSchema(), executorModelJson(metadata)), @@ -269,6 +276,29 @@ export function createExecutorModelBroker(options: { ]); } +/** Longest describing a model waits for it to settle before describing it as built. */ +const METADATA_SETTLE_MAX_WAIT_MS = 3_000; + +/** + * Settle a Veryfront Cloud model before its metadata is read, waiting a bounded + * time. A failure or a slow catalog leaves the model as built; its call + * surfaces any failure. + */ +async function settleForMetadata(model: ModelRuntime, signal: AbortSignal): Promise { + if (readVeryfrontCloudModelFacts(model) === undefined || typeof model.prepare !== "function") { + return; + } + let timer: ReturnType | undefined; + try { + await Promise.race([ + Promise.resolve(model.prepare(signal)).catch(() => {}), + new Promise((resolve) => timer = setTimeout(resolve, METADATA_SETTLE_MAX_WAIT_MS)), + ]); + } finally { + if (timer !== undefined) clearTimeout(timer); + } +} + function modelMetadata(id: string, model: ModelRuntime) { return { id, diff --git a/src/agent/runtime/model-transport.ts b/src/agent/runtime/model-transport.ts index 3dc8c89eca..49e4702ddc 100644 --- a/src/agent/runtime/model-transport.ts +++ b/src/agent/runtime/model-transport.ts @@ -8,6 +8,7 @@ import { type AgentConfig, type RuntimeReasoningOption } from "../types.ts"; import { type ModelRuntime, resolveModel } from "#veryfront/provider"; import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; import { warmVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import { readVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { isVeryfrontCloudEnabled } from "#veryfront/platform/cloud/resolver.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; import { @@ -259,6 +260,16 @@ export async function resolveAgentModelTransport( : resolveProviderOptionsWithDefaults(resolvedModelString, transport?.providerOptions); const languageModel = privatelyResolvedModel ?? transport?.model ?? resolveModel(resolvedModelString); + // A Veryfront Cloud model settles its protocol and capabilities on its first + // async step. Settle it here, before anything reads them: the provider option + // key below and the runtime's tool-calling, structured-output and replay + // checks. A failure surfaces again when the model is called. + if ( + readVeryfrontCloudModelFacts(languageModel) !== undefined && + typeof languageModel.prepare === "function" + ) { + await Promise.resolve(languageModel.prepare()).catch(() => {}); + } const providerOptionKey = resolveModelProviderOptionKey(resolvedModelString, languageModel); return { diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index 9d4fe51e7c..9ff5f04457 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -2022,6 +2022,40 @@ describe("provider/veryfront-cloud served catalog loading", () => { ); }); + it("validates a response format against the protocol the served catalog settles", async () => { + setCloudBootstrap(); + const requests = installGateway(() => + Response.json({ + models: [{ + id: "acme-claude", + modelId: "acme/acme-claude", + provider: "acme", + surface: "anthropic", + operations: ["messages"], + aliases: [], + capabilities: {}, + }], + }) + ); + const call = (model: ModelRuntime) => + generateText({ + model, + messages: [{ role: "user", content: "Hi" }], + responseFormat: { type: "json" }, + }); + + // Cold, the unlisted provider would look like an OpenAI-protocol model. + const cold = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + const coldError = await call(cold).then(() => undefined, (error: unknown) => error); + const warm = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + const warmError = await call(warm).then(() => undefined, (error: unknown) => error); + + assertEquals(warmError instanceof Error, true); + assertEquals((coldError as Error | undefined)?.message, (warmError as Error).message); + // Refused before any inference request. + assertEquals(calls(requests), ["GET /ai/models"]); + }); + it("forwards metadata to the model rebuilt from the catalog", async () => { setCloudBootstrap(); installGateway(() => diff --git a/src/runtime/runtime-bridge.ts b/src/runtime/runtime-bridge.ts index b3e9dddab4..72597f9819 100644 --- a/src/runtime/runtime-bridge.ts +++ b/src/runtime/runtime-bridge.ts @@ -732,20 +732,27 @@ function buildDirectModelOptions( }; } -async function emitModelCallContextEvent( - options: DirectTextOptions, - directOptions: DirectModelOptions, -): Promise { - const sinks = getActiveRunEventSinks(); - if (!sinks.mandatory && !sinks.public) return; - // A Veryfront Cloud model settles how it is built on its first async step. - // It does so here, so the recorded request describes the request then sent. +/** + * Settle a Veryfront Cloud model before anything reads its protocol or + * capabilities: it settles how it is built on its first async step, so the + * options validated and built for it, and the request recorded for it, match + * the request then sent. + */ +async function settleVeryfrontCloudModel(options: DirectTextOptions): Promise { if ( readVeryfrontCloudModelFacts(options.model) !== undefined && typeof options.model.prepare === "function" ) { await options.model.prepare(options.abortSignal); } +} + +async function emitModelCallContextEvent( + options: DirectTextOptions, + directOptions: DirectModelOptions, +): Promise { + const sinks = getActiveRunEventSinks(); + if (!sinks.mandatory && !sinks.public) return; const request = buildModelCallContextRequest(options.model, directOptions); const event: AgentRunModelCallContextEvent = { @@ -1257,6 +1264,7 @@ async function* textDeltasFromStream(stream: ReadableStream): AsyncIter export function generateText(options: GenerateTextOptions): PromiseLike { return resolveDirectTools(options.tools).then(async (tools) => { + await settleVeryfrontCloudModel(options); const directOptions = buildDirectModelOptions(options, tools); await emitModelCallContextEvent(options, directOptions); if (shouldGenerateViaStream(options.model)) { @@ -1271,6 +1279,7 @@ export function generateText(options: GenerateTextOptions): PromiseLike { + await settleVeryfrontCloudModel(options); const directOptions = buildDirectModelOptions(options, tools); await emitModelCallContextEvent(options, directOptions); return options.model.doStream(directOptions); From 47afa90aaba7b92c5c85a13854f591bf06f5da90 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 11:53:18 +0200 Subject: [PATCH 11/17] fix(provider): log catalog failures per scope and state the stale-catalog fallback - Each catalog scope logs its own failure once per outage, naming the API base URL and project but never the credential; a failed refresh that keeps the last catalog says so instead of claiming the built-in list applies. - loadVeryfrontCloudModelCatalog documents that a failed refresh keeps the last loaded catalog; the shipped facts apply only when none has loaded. --- docs/api-reference/veryfront/provider.md | 42 +++++++++---------- .../veryfront-cloud/catalog-client.ts | 26 ++++++++---- src/provider/veryfront-cloud/shared.ts | 7 ++-- .../veryfront-cloud-catalog-client.test.ts | 23 ++++++++++ 4 files changed, 66 insertions(+), 32 deletions(-) diff --git a/docs/api-reference/veryfront/provider.md b/docs/api-reference/veryfront/provider.md index 790550c6b4..f90ab5997e 100644 --- a/docs/api-reference/veryfront/provider.md +++ b/docs/api-reference/veryfront/provider.md @@ -71,27 +71,27 @@ Clear all registered model providers and reset lazy built-ins (for testing). ### Functions -| Name | Description | Source | -| ---------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | -| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `findVeryfrontCloudModel` | Find a shipped chat model by its short id. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | -| `findVeryfrontCloudModelByModelId` | Find a shipped chat model by its provider-qualified id, in any provider spelling. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | -| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `groupVeryfrontCloudModelsByProvider` | Group the shipped chat models by provider, in display order. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | -| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `loadVeryfrontCloudModelCatalog` | Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud credentials and project in effect, so model facts read synchronously afterwards (thinking defaults, short aliases such as `opus`, the default model) come from it. Resolves to whether a catalog is available. Never throws: without credentials or a reachable catalog, the facts shipped with this package apply. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/shared.ts) | -| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveVeryfrontCloudDefaultModelId` | Provider-qualified ID of the default model: the one the served catalog names once it is loaded, otherwise the built-in default. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelId` | Resolve a model ID or short alias to a provider-qualified model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ---------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ | +| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `findVeryfrontCloudModel` | Find a shipped chat model by its short id. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `findVeryfrontCloudModelByModelId` | Find a shipped chat model by its provider-qualified id, in any provider spelling. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `groupVeryfrontCloudModelsByProvider` | Group the shipped chat models by provider, in display order. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `loadVeryfrontCloudModelCatalog` | Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud credentials and project in effect, so model facts read synchronously afterward (thinking defaults, short aliases such as `opus`, the default model) come from it. Resolves to whether a catalog is available. Never throws. When a refresh fails, the last catalog loaded for these credentials stays in use; only when none has loaded (no credentials, or no load has succeeded yet) do the facts shipped with this package apply. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/shared.ts) | +| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveVeryfrontCloudDefaultModelId` | Provider-qualified ID of the default model: the one the served catalog names once it is loaded, otherwise the built-in default. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelId` | Resolve a model ID or short alias to a provider-qualified model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Types diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index d1f5fd37c1..2af991304b 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -106,7 +106,8 @@ const failedAt = new Map(); /** Key of the scope a synchronous read uses, while {@link withVeryfrontCloudCatalogScope} runs. */ let activeKey: string | undefined; let seeded: VeryfrontCloudCatalog | undefined; -let failureLogged = false; +/** Scopes whose current failure has been logged, so each scope logs once per outage. */ +const loggedFailures = new Set(); let now: () => number = Date.now; /** Bumped by a test reset, so a load that settles afterwards changes nothing. */ let generation = 0; @@ -317,20 +318,29 @@ function refresh( if (started !== generation) return catalog; rememberEntry(key, { catalog, fetchedAt: now() }); failedAt.delete(key); - failureLogged = false; + loggedFailures.delete(key); return catalog; }, (error: unknown) => { if (started !== generation) return undefined; rememberFailure(key, now()); - if (!failureLogged) { - failureLogged = true; + const stale = entries.get(key)?.catalog; + if (!loggedFailures.has(key)) { + if (loggedFailures.size >= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) loggedFailures.clear(); + loggedFailures.add(key); + // Names the scope, never the credential. logger.warn( - "Veryfront Cloud model catalog is unavailable; model facts fall back to the built-in list", - { error: error instanceof Error ? error.message : String(error) }, + stale + ? "Veryfront Cloud model catalog refresh failed; the last loaded catalog stays in use" + : "Veryfront Cloud model catalog is unavailable; model facts fall back to the built-in list", + { + apiBaseUrl: options.apiBaseUrl, + projectSlug: options.projectSlug, + error: error instanceof Error ? error.message : String(error), + }, ); } - return entries.get(key)?.catalog; + return stale; }, ).finally(() => { if (started === generation) inflight.delete(key); @@ -479,7 +489,7 @@ export function __resetVeryfrontCloudCatalogForTests(): void { failedAt.clear(); activeKey = undefined; seeded = undefined; - failureLogged = false; + loggedFailures.clear(); now = Date.now; } diff --git a/src/provider/veryfront-cloud/shared.ts b/src/provider/veryfront-cloud/shared.ts index ca11a8b383..15a53b8366 100644 --- a/src/provider/veryfront-cloud/shared.ts +++ b/src/provider/veryfront-cloud/shared.ts @@ -327,10 +327,11 @@ export function requireVeryfrontCloudBootstrap( /** * Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud * credentials and project in effect, so model facts read synchronously - * afterwards (thinking defaults, short aliases such as `opus`, the default + * afterward (thinking defaults, short aliases such as `opus`, the default * model) come from it. Resolves to whether a catalog is available. Never - * throws: without credentials or a reachable catalog, the facts shipped with - * this package apply. + * throws. When a refresh fails, the last catalog loaded for these credentials + * stays in use; only when none has loaded (no credentials, or no load has + * succeeded yet) do the facts shipped with this package apply. */ export async function loadVeryfrontCloudModelCatalog( options: { signal?: AbortSignal; maxWaitMs?: number } = {}, diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts index d9ca9eeaa9..c421e63c46 100644 --- a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -15,6 +15,7 @@ import { VERYFRONT_CLOUD_CATALOG_TTL_MS, } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { servedCatalogPayload } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { logger } from "#veryfront/utils/logger/logger.ts"; const API_BASE_URL = "https://api.veryfront.com"; const LOAD = { apiBaseUrl: API_BASE_URL, apiToken: "vf_catalog_test", projectSlug: "catalog-test" }; @@ -372,5 +373,27 @@ describe("provider/veryfront-cloud/catalog-client", () => { assertEquals(peekVeryfrontCloudCatalog(LOAD), loaded); }); }); + + it("logs one warning per failing scope, naming the scope but never the credential", async () => { + const warnings: { message: string; fields: unknown }[] = []; + const originalWarn = logger.warn; + logger.warn = (message: string, fields?: unknown) => warnings.push({ message, fields }); + const other = { ...LOAD, apiToken: "vf_catalog_other", projectSlug: "catalog-other" }; + const stub = recordingFetch(() => Promise.reject(new TypeError("network down"))); + try { + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + await loadVeryfrontCloudCatalog(other); + }); + } finally { + logger.warn = originalWarn; + } + + assertEquals( + warnings.map((warning) => (warning.fields as { projectSlug?: string }).projectSlug), + ["catalog-test", "catalog-other"], + ); + assertEquals(JSON.stringify(warnings).includes("vf_catalog_"), false); + }); }); }); From d0ee50cac06a13c3fd443e8917639fdf33649369 Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 12:09:42 +0200 Subject: [PATCH 12/17] fix(provider): forward members a rebuilt model gains; cancel judge catalog waits - A served-catalog wrapper forwards optional runtime members (such as the provider-metadata reconciliation hook) even when the model it was first built as lacks them, so a model rebuilt onto another protocol exposes them. - Hosted application models read the reconciliation hook when it is described and when it is invoked. - LLM judges pass the evaluation's signal to the catalog load, so a cancelled evaluation stops waiting at once. --- .../hosted/application-model-resolver.test.ts | 33 +++++++++++++ .../hosted/application-model-resolver.ts | 28 +++++------ src/eval/judges.test.ts | 35 ++++++++++++++ src/eval/judges.ts | 9 ++-- src/provider/veryfront-cloud/provider.ts | 46 +++++++++++++------ 5 files changed, 120 insertions(+), 31 deletions(-) diff --git a/src/agent/hosted/application-model-resolver.test.ts b/src/agent/hosted/application-model-resolver.test.ts index 103066df2e..41aed0084f 100644 --- a/src/agent/hosted/application-model-resolver.test.ts +++ b/src/agent/hosted/application-model-resolver.test.ts @@ -3,6 +3,10 @@ import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/te import { describe, it } from "#veryfront/testing/bdd.ts"; import { revokeModelRuntimeResolver } from "../runtime/model-transport.ts"; import { createHostedApplicationModelResolver } from "./application-model-resolver.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForScopeForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; const modelId = "veryfront-cloud/openai/gpt-4o"; @@ -150,4 +154,33 @@ describe("hosted application model authority", () => { "authority is revoked", ); }); + it("exposes the reconciliation hook of the protocol a model settles on", async () => { + const servedId = "veryfront-cloud/acme/acme-gemini"; + const options = { ...resolverOptions(), allowedModelIds: new Set([servedId]) }; + try { + const model = createHostedApplicationModelResolver(options)(servedId)!; + // Cold, an unlisted provider is built as an OpenAI-protocol model. + assertEquals(model._reconcileProviderMetadata, undefined); + + __setVeryfrontCloudCatalogForScopeForTests( + { apiBaseUrl: options.apiBaseUrl, apiToken: options.authToken }, + { + models: [{ + id: "acme-gemini", + modelId: "acme/acme-gemini", + provider: "acme", + surface: "google", + operations: ["chat-completions"], + aliases: [], + capabilities: {}, + }], + }, + ); + await model.prepare!(); + + assert(typeof model._reconcileProviderMetadata === "function"); + } finally { + __resetVeryfrontCloudCatalogForTests(); + } + }); }); diff --git a/src/agent/hosted/application-model-resolver.ts b/src/agent/hosted/application-model-resolver.ts index 98ad689170..5283a6fa5d 100644 --- a/src/agent/hosted/application-model-resolver.ts +++ b/src/agent/hosted/application-model-resolver.ts @@ -139,7 +139,6 @@ export function createHostedApplicationModelResolver(input: { assertCredentialActive: assertActive, }) ); - const reconcile = model._reconcileProviderMetadata; // Metadata is read from the model on each access: a Veryfront Cloud model // may settle a different protocol and capabilities on its first async step. const proxy: ModelRuntime = Object.freeze({ @@ -195,18 +194,21 @@ export function createHostedApplicationModelResolver(input: { throw error; } }, - ...(typeof reconcile === "function" - ? { - async _reconcileProviderMetadata(options: { - providerMetadata: Record; - suppressedToolCalls: readonly { id: string; name: string }[]; - abortSignal?: AbortSignal; - }) { - return await run(callScope(options.abortSignal), (signal) => - reconcile.call(model, { ...options, abortSignal: signal })); - }, - } - : {}), + // Read when described and when invoked: a model rebuilt onto another + // protocol may gain or lose its reconciliation hook. + get _reconcileProviderMetadata() { + const reconcile = model._reconcileProviderMetadata; + if (typeof reconcile !== "function") return undefined; + return async (options: { + providerMetadata: Record; + suppressedToolCalls: readonly { id: string; name: string }[]; + abortSignal?: AbortSignal; + }) => + await run( + callScope(options.abortSignal), + (signal) => reconcile.call(model, { ...options, abortSignal: signal }), + ); + }, }); // The proxy reports the facts of the model it calls. registerVeryfrontCloudModelFacts(proxy, () => readVeryfrontCloudModelFacts(model)!); diff --git a/src/eval/judges.test.ts b/src/eval/judges.test.ts index b289f786e1..8d1bcefabe 100644 --- a/src/eval/judges.test.ts +++ b/src/eval/judges.test.ts @@ -9,6 +9,9 @@ import { buildProviderError } from "#veryfront/provider/runtime-loader/provider- import { describe, it } from "#veryfront/testing/bdd.ts"; import type { ModelRuntime } from "veryfront/provider"; import { judges } from "veryfront/eval"; +import { withEnv } from "#veryfront/testing"; +import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; function createJudgeModel(text: string, calls: unknown[]): ModelRuntime { return { @@ -84,6 +87,38 @@ describe("eval/judges", () => { assertStringIncludes(dataPrompt, "END EVALUATION DATA"); }); + it("stops waiting for the model catalog when the evaluation is cancelled", async () => { + const controller = new AbortController(); + controller.abort(); + // The catalog request does not answer until the test ends; a cancelled judge + // must not wait for it. + let release: (response: Response) => void = () => {}; + const pending = new Promise((resolve) => release = resolve); + const hanging = () => pending; + try { + const started = Date.now(); + const result = await withEnv( + { VERYFRONT_API_TOKEN: "vf_judge_test", VERYFRONT_PROJECT_SLUG: "judge-test" }, + () => + withMockFetch(hanging, () => + judges.llm.rubric({ model: "openai/gpt-5-nano" })({ + input: "Question", + output: { text: "Answer" }, + metadata: {}, + rubric: "Correct.", + signal: controller.signal, + })), + ); + assertEquals(result.pass, false); + assertEquals(Date.now() - started < 1_000, true); + } finally { + // Let the shared request settle so its timeout is cleared. + release(new Response(null, { status: 503 })); + await new Promise((resolve) => setTimeout(resolve, 0)); + __resetVeryfrontCloudCatalogForTests(); + } + }); + it("propagates gateway credit denials from both built-in LLM judges", async () => { const deniedModel: ModelRuntime = { provider: "test", diff --git a/src/eval/judges.ts b/src/eval/judges.ts index 8faa9d86e1..2754f81e44 100644 --- a/src/eval/judges.ts +++ b/src/eval/judges.ts @@ -92,10 +92,13 @@ function clampScore(score: number): number { async function resolveJudgeModel( model: string | ModelRuntime | undefined, + signal?: AbortSignal, ): Promise { if (model && typeof model === "object") return model; // Whether the judge model routes through Veryfront Cloud reads the served catalog. - await loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS }); + // A cancelled evaluation stops waiting for it at once. + await loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS, signal }); + signal?.throwIfAborted(); return resolveModel(resolveRuntimeModel(model ?? DEFAULT_JUDGE_MODEL)); } @@ -380,7 +383,7 @@ function createLlmRubricJudge( return async (input) => { try { - const model = await resolveJudgeModel(options.model); + const model = await resolveJudgeModel(options.model, input.signal); const response = await generateText({ model, messages: [ @@ -445,7 +448,7 @@ function createLlmGroundednessJudge( return async (input) => { try { - const model = await resolveJudgeModel(validatedOptions.model); + const model = await resolveJudgeModel(validatedOptions.model, input.signal); const response = await generateText({ model, messages: [{ diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index f35b7ac7ae..e37b39ff0f 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -115,6 +115,20 @@ const NON_FORWARDED_KEYS: ReadonlySet = new Set([ "constructor", ]); +/** + * Optional members a model rebuilt onto another protocol can gain even when + * the model it was first built as lacks them (the Google protocol adds + * `_reconcileProviderMetadata`), so the wrapper forwards them regardless. + */ +const OPTIONAL_FORWARDED_KEYS: readonly PropertyKey[] = [ + "_reconcileProviderMetadata", + "_generateViaStream", + "runtimeCapabilities", + "executionMode", + "modelProvider", + "specificationVersion", +]; + /** * Wrap a built model so its first async step loads the served catalog. When * the catalog changes how the model is built, calls and metadata go to the @@ -160,25 +174,27 @@ function withServedCatalog( // Metadata (provider attribution, capabilities, model ID) follows the model // the calls go to, so a rebuild never leaves the construction-time values. const forwarded = new Set(); + const forward = (key: PropertyKey): void => { + if (forwarded.has(key) || NON_FORWARDED_KEYS.has(key)) return; + forwarded.add(key); + ObjectDefineProperty(wrapped, key, { + configurable: false, + enumerable: true, + get: () => { + const target = current(); + const value: unknown = IntrinsicReflectApply(ReflectGet, undefined, [target, key]); + return typeof value === "function" + ? IntrinsicReflectApply(FunctionBind, value, [target]) + : value; + }, + }); + }; let source: object | null = model; while (source && source !== ObjectPrototype) { - for (const key of ReflectOwnKeys(source)) { - if (forwarded.has(key) || NON_FORWARDED_KEYS.has(key)) continue; - forwarded.add(key); - ObjectDefineProperty(wrapped, key, { - configurable: false, - enumerable: true, - get: () => { - const target = current(); - const value: unknown = IntrinsicReflectApply(ReflectGet, undefined, [target, key]); - return typeof value === "function" - ? IntrinsicReflectApply(FunctionBind, value, [target]) - : value; - }, - }); - } + for (const key of ReflectOwnKeys(source)) forward(key); source = ObjectGetPrototypeOf(source); } + for (const key of OPTIONAL_FORWARDED_KEYS) forward(key); return wrapped; } From 9f3ce105749401c1d8bf88f06f97adcd6f4ebe5e Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 13:05:55 +0200 Subject: [PATCH 13/17] fix(provider): keep the catalog cache within its cap after concurrent loads Eviction skips scopes with a load in flight, so a burst of concurrent scopes could leave the cache over its cap. Eviction now runs again once each load settles. The judge cancellation test moves to integration, since it drives the catalog request. --- src/eval/judges.test.ts | 35 ---------------- .../veryfront-cloud/catalog-client.ts | 7 +++- .../eval/judge-catalog-cancellation.test.ts | 41 +++++++++++++++++++ .../veryfront-cloud-catalog-client.test.ts | 26 ++++++++++++ 4 files changed, 73 insertions(+), 36 deletions(-) create mode 100644 tests/integration/eval/judge-catalog-cancellation.test.ts diff --git a/src/eval/judges.test.ts b/src/eval/judges.test.ts index 8d1bcefabe..b289f786e1 100644 --- a/src/eval/judges.test.ts +++ b/src/eval/judges.test.ts @@ -9,9 +9,6 @@ import { buildProviderError } from "#veryfront/provider/runtime-loader/provider- import { describe, it } from "#veryfront/testing/bdd.ts"; import type { ModelRuntime } from "veryfront/provider"; import { judges } from "veryfront/eval"; -import { withEnv } from "#veryfront/testing"; -import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; -import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; function createJudgeModel(text: string, calls: unknown[]): ModelRuntime { return { @@ -87,38 +84,6 @@ describe("eval/judges", () => { assertStringIncludes(dataPrompt, "END EVALUATION DATA"); }); - it("stops waiting for the model catalog when the evaluation is cancelled", async () => { - const controller = new AbortController(); - controller.abort(); - // The catalog request does not answer until the test ends; a cancelled judge - // must not wait for it. - let release: (response: Response) => void = () => {}; - const pending = new Promise((resolve) => release = resolve); - const hanging = () => pending; - try { - const started = Date.now(); - const result = await withEnv( - { VERYFRONT_API_TOKEN: "vf_judge_test", VERYFRONT_PROJECT_SLUG: "judge-test" }, - () => - withMockFetch(hanging, () => - judges.llm.rubric({ model: "openai/gpt-5-nano" })({ - input: "Question", - output: { text: "Answer" }, - metadata: {}, - rubric: "Correct.", - signal: controller.signal, - })), - ); - assertEquals(result.pass, false); - assertEquals(Date.now() - started < 1_000, true); - } finally { - // Let the shared request settle so its timeout is cleared. - release(new Response(null, { status: 503 })); - await new Promise((resolve) => setTimeout(resolve, 0)); - __resetVeryfrontCloudCatalogForTests(); - } - }); - it("propagates gateway credit denials from both built-in LLM judges", async () => { const deniedModel: ModelRuntime = { provider: "test", diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index 2af991304b..68e24e9f5f 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -343,7 +343,12 @@ function refresh( return stale; }, ).finally(() => { - if (started === generation) inflight.delete(key); + if (started !== generation) return; + inflight.delete(key); + // Eviction skips keys with a load in flight; retry now that this one has + // settled, so a burst of concurrent scopes cannot leave the maps over the cap. + evictOldest(entries); + evictOldest(failedAt); }); inflight.set(key, request); return request; diff --git a/tests/integration/eval/judge-catalog-cancellation.test.ts b/tests/integration/eval/judge-catalog-cancellation.test.ts new file mode 100644 index 0000000000..60770059d1 --- /dev/null +++ b/tests/integration/eval/judge-catalog-cancellation.test.ts @@ -0,0 +1,41 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { describe, it } from "#veryfront/testing/bdd.ts"; +import { withEnv } from "#veryfront/testing"; +import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { judges } from "veryfront/eval"; + +describe("eval judges with Veryfront Cloud", () => { + it("stops waiting for the model catalog when the evaluation is cancelled", async () => { + const controller = new AbortController(); + controller.abort(); + // The catalog request does not answer until the test ends; a cancelled judge + // must not wait for it. + let release: (response: Response) => void = () => {}; + const pending = new Promise((resolve) => release = resolve); + const hanging = () => pending; + try { + const started = Date.now(); + const result = await withEnv( + { VERYFRONT_API_TOKEN: "vf_judge_test", VERYFRONT_PROJECT_SLUG: "judge-test" }, + () => + withMockFetch(hanging, () => + judges.llm.rubric({ model: "openai/gpt-5-nano" })({ + input: "Question", + output: { text: "Answer" }, + metadata: {}, + rubric: "Correct.", + signal: controller.signal, + })), + ); + assertEquals(result.pass, false); + assertEquals(Date.now() - started < 1_000, true); + } finally { + // Let the shared request settle so its timeout is cleared. + release(new Response(null, { status: 503 })); + await new Promise((resolve) => setTimeout(resolve, 0)); + __resetVeryfrontCloudCatalogForTests(); + } + }); +}); diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts index c421e63c46..29be66d5fb 100644 --- a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -200,6 +200,32 @@ describe("provider/veryfront-cloud/catalog-client", () => { ); }); + it("stays within the cap after more concurrent scopes than it holds all settle", async () => { + const count = VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES + 8; + let release: () => void = () => {}; + const gate = new Promise((resolve) => release = resolve); + // Every load is in flight at once, so none can be evicted while it lands. + const stub = recordingFetch(async () => { + await gate; + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + const loads = Array.from( + { length: count }, + (_, index) => loadVeryfrontCloudCatalog({ ...LOAD, apiToken: `vf_burst_${index}` }), + ); + release(); + await Promise.all(loads); + }); + + assertEquals(stub.requests.length, count); + assertEquals( + __veryfrontCloudCatalogSizesForTests().entries, + VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES, + ); + }); + it("forgets a failure once its retry window has passed", async () => { const clock = useClock(); const stub = recordingFetch(() => jsonResponse({}, 503)); From 8b1444559d61aeb276b7f3290c84470ad59441ab Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 13:19:48 +0200 Subject: [PATCH 14/17] fix(agent): describe a served model to the executor only once it has settled The broker no longer races model preparation against a fixed wait before describing a model, so it never describes a construction that a pending rebuild can still replace. The wait is bounded by the catalog request's own timeout and the caller's signal. --- .../hosted/executor-model-bridge.test.ts | 30 +++++++++++++++++++ src/agent/hosted/executor-model-bridge.ts | 20 ++++--------- 2 files changed, 35 insertions(+), 15 deletions(-) diff --git a/src/agent/hosted/executor-model-bridge.test.ts b/src/agent/hosted/executor-model-bridge.test.ts index fba4bcea19..b35a16522a 100644 --- a/src/agent/hosted/executor-model-bridge.test.ts +++ b/src/agent/hosted/executor-model-bridge.test.ts @@ -14,6 +14,7 @@ import { createExecutorModelBroker, createExecutorModelRuntimeResolver, } from "./executor-model-bridge.ts"; +import { registerVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; const modelId = "veryfront-cloud/openai/synthetic-model"; const allowedModelIds = new Set([modelId]); @@ -977,4 +978,33 @@ describe("executor managed model bridge", () => { } } }); + + it("describes a served model only after its preparation settles, however long it takes", async () => { + // Longer than any fixed wait the broker might apply before describing. + const settleMs = 3_200; + let settledProvider = "openai"; + const model = stubModel({ + async prepare() { + await new Promise((resolve) => setTimeout(resolve, settleMs)); + settledProvider = "anthropic"; + }, + }); + Object.defineProperty(model, "modelProvider", { get: () => settledProvider }); + registerVeryfrontCloudModelFacts(model, () => + ({ + surface: settledProvider, + nativeProtocol: false, + }) as never); + const broker = createExecutorModelBroker({ + allowedModelIds, + resolveModelRuntime: () => model, + }); + const metadata = await broker.get("model.metadata")!.handle!({}, { + binding: { allocationId: "allocation-test", generation: 1, invocationId: "invocation-test" }, + signal: new AbortController().signal, + deadline: Date.now() + 60_000, + }) as { modelProvider?: string }[]; + + assertEquals(metadata.map((entry) => entry.modelProvider), ["anthropic"]); + }); }); diff --git a/src/agent/hosted/executor-model-bridge.ts b/src/agent/hosted/executor-model-bridge.ts index ff96cea699..ee37f948cc 100644 --- a/src/agent/hosted/executor-model-bridge.ts +++ b/src/agent/hosted/executor-model-bridge.ts @@ -276,27 +276,17 @@ export function createExecutorModelBroker(options: { ]); } -/** Longest describing a model waits for it to settle before describing it as built. */ -const METADATA_SETTLE_MAX_WAIT_MS = 3_000; - /** - * Settle a Veryfront Cloud model before its metadata is read, waiting a bounded - * time. A failure or a slow catalog leaves the model as built; its call - * surfaces any failure. + * Settle a Veryfront Cloud model before its metadata is read, so the executor + * is never told about a construction a pending rebuild can still replace. The + * wait is bounded by the catalog request's own timeout and the caller's signal. + * A failed load leaves the model as built; its call surfaces any failure. */ async function settleForMetadata(model: ModelRuntime, signal: AbortSignal): Promise { if (readVeryfrontCloudModelFacts(model) === undefined || typeof model.prepare !== "function") { return; } - let timer: ReturnType | undefined; - try { - await Promise.race([ - Promise.resolve(model.prepare(signal)).catch(() => {}), - new Promise((resolve) => timer = setTimeout(resolve, METADATA_SETTLE_MAX_WAIT_MS)), - ]); - } finally { - if (timer !== undefined) clearTimeout(timer); - } + await Promise.resolve(model.prepare(signal)).catch(() => {}); } function modelMetadata(id: string, model: ModelRuntime) { From a8c92550146ffd453bd711b1831f435dcdead25f Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 13:34:00 +0200 Subject: [PATCH 15/17] fix(agent): apply Anthropic defaults by served surface, not the provider name A newly served provider on the Anthropic surface now gets the Anthropic thinking defaults, the same reasoning-option handling and the same reasoning token reservation as anthropic/* models. One helper, isVeryfrontCloudAnthropicSurfaceModel, decides this from the served catalog. --- src/agent/hosted/runtime-preparation-core.ts | 9 ++++--- .../runtime/default-provider-options.test.ts | 26 ++++++++++++++++++- src/agent/runtime/default-provider-options.ts | 14 +++++++--- src/agent/runtime/model-transport.ts | 4 +-- src/provider/veryfront-cloud/model-catalog.ts | 10 +++++++ 5 files changed, 52 insertions(+), 11 deletions(-) diff --git a/src/agent/hosted/runtime-preparation-core.ts b/src/agent/hosted/runtime-preparation-core.ts index 75cbca12c7..90427519ff 100644 --- a/src/agent/hosted/runtime-preparation-core.ts +++ b/src/agent/hosted/runtime-preparation-core.ts @@ -10,10 +10,10 @@ import { } from "#veryfront/security/private-promise.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; import { + isVeryfrontCloudAnthropicSurfaceModel, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, - tryGetVeryfrontCloudProviderFromModelId, VERYFRONT_CLOUD_MODEL_PREFIX, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { getExecutorModelAdditiveReasoningTokens } from "#veryfront/agent/hosted/executor-model-grant.ts"; @@ -497,11 +497,12 @@ export function createRuntimePreparationCore(input: RuntimePreparationCoreOption const thinking = request.thinking ?? definition.thinking ?? resolveVeryfrontCloudModelThinking(modelId); let availableOutputTokens = modelGrant.maxOutputTokens; - const modelProvider = tryGetVeryfrontCloudProviderFromModelId(modelId); - if (modelProvider === "anthropic") { + // The served surface decides the protocol, so a newly served provider on + // the Anthropic surface reserves its reasoning tokens like `anthropic/*`. + if (isVeryfrontCloudAnthropicSurfaceModel(modelId)) { try { const effectiveThinking = thinking ?? resolveVeryfrontCloudModelThinking(modelId); - const model = { id: modelId, modelId, provider: modelProvider }; + const model = { id: modelId, modelId, provider: "anthropic" }; const options = { reasoning: resolveVeryfrontCloudReasoningOption(modelId, effectiveThinking), providerOptions: resolveVeryfrontCloudThinkingProviderOptions( diff --git a/src/agent/runtime/default-provider-options.test.ts b/src/agent/runtime/default-provider-options.test.ts index 4c1ddeaf51..c39ddef4b6 100644 --- a/src/agent/runtime/default-provider-options.test.ts +++ b/src/agent/runtime/default-provider-options.test.ts @@ -2,7 +2,10 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals } from "#veryfront/testing/assert.ts"; import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; -import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; describe("resolveProviderOptionsWithDefaults", () => { @@ -36,6 +39,27 @@ describe("resolveProviderOptionsWithDefaults", () => { }); }); + it("applies Anthropic defaults to a newly served provider on the Anthropic surface", () => { + __setVeryfrontCloudCatalogForTests({ + models: [{ + id: "m1", + modelId: "acme-labs/m1", + provider: "acme-labs", + surface: "anthropic", + operations: ["messages"], + aliases: [], + capabilities: { thinking: true, reasoning: true, reasoning_mode: "adaptive" }, + }], + }); + + const result = resolveProviderOptionsWithDefaults("veryfront-cloud/acme-labs/m1", undefined); + + assertEquals( + (result?.anthropic as { thinking?: unknown } | undefined)?.thinking, + { type: "adaptive", display: "summarized" }, + ); + }); + it("does not enable thinking for non-Anthropic models", () => { assertEquals( resolveProviderOptionsWithDefaults("openai/gpt-5.5", undefined), diff --git a/src/agent/runtime/default-provider-options.ts b/src/agent/runtime/default-provider-options.ts index 996d0d047f..a0d7f9dd3b 100644 --- a/src/agent/runtime/default-provider-options.ts +++ b/src/agent/runtime/default-provider-options.ts @@ -8,6 +8,7 @@ */ import { + isVeryfrontCloudAnthropicSurfaceModel, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudThinkingProviderOptions, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; @@ -15,11 +16,16 @@ import { const VERYFRONT_CLOUD_PREFIX = "veryfront-cloud/"; const ANTHROPIC_PREFIX = "anthropic/"; +/** + * Whether a model speaks the Anthropic protocol. A Veryfront Cloud model + * speaks the surface the served catalog gives its provider, so a newly served + * provider on the Anthropic surface gets the same defaults as `anthropic/*`. + */ function isAnthropicModel(modelString: string): boolean { - const normalized = modelString.startsWith(VERYFRONT_CLOUD_PREFIX) - ? modelString.slice(VERYFRONT_CLOUD_PREFIX.length) - : modelString; - return normalized.startsWith(ANTHROPIC_PREFIX); + if (modelString.startsWith(VERYFRONT_CLOUD_PREFIX)) { + return isVeryfrontCloudAnthropicSurfaceModel(modelString); + } + return modelString.startsWith(ANTHROPIC_PREFIX); } function hasAnthropicThinkingConfig(existing: Record | undefined): boolean { diff --git a/src/agent/runtime/model-transport.ts b/src/agent/runtime/model-transport.ts index 49e4702ddc..72f1b33403 100644 --- a/src/agent/runtime/model-transport.ts +++ b/src/agent/runtime/model-transport.ts @@ -17,10 +17,10 @@ import { resolveRuntimeModel, } from "./model-resolution.ts"; import { + isVeryfrontCloudAnthropicSurfaceModel, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, - tryGetVeryfrontCloudProviderFromModelId, VERYFRONT_CLOUD_MODEL_PREFIX, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { hasDisabledThinking } from "./model-capabilities.ts"; @@ -201,7 +201,7 @@ function resolveReasoningWithDefaults( return { enabled: false }; } - if (tryGetVeryfrontCloudProviderFromModelId(modelString) === "anthropic") { + if (isVeryfrontCloudAnthropicSurfaceModel(modelString)) { return undefined; } diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index e62077affc..b693450352 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -726,6 +726,16 @@ export function getVeryfrontCloudProviderFromModelId( } /** Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. */ +/** + * Whether a Veryfront Cloud model ID speaks the Anthropic protocol: its + * provider is served on the Anthropic surface. A newly served provider on that + * surface counts, not only `anthropic/*`. + */ +export function isVeryfrontCloudAnthropicSurfaceModel(modelId: string): boolean { + const provider = tryGetVeryfrontCloudProviderFromModelId(modelId); + return provider !== undefined && resolveVeryfrontCloudSurface(provider) === "anthropic"; +} + export function tryGetVeryfrontCloudProviderFromModelId( modelId: string, ): VeryfrontCloudProviderId | undefined { From 3c9ed06537d10be0a83d231c9efd1a714ee4af6d Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 13:39:54 +0200 Subject: [PATCH 16/17] fix(provider): never log a signed query value from a catalog failure The catalog failure warning logs the API base URL without its query or fragment, and strips them from every URL the error text quotes. --- .../veryfront-cloud/catalog-client.ts | 22 +++++++++++++++--- .../veryfront-cloud-catalog-client.test.ts | 23 +++++++++++++++++++ 2 files changed, 42 insertions(+), 3 deletions(-) diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts index 68e24e9f5f..c942eb0cc7 100644 --- a/src/provider/veryfront-cloud/catalog-client.ts +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -274,6 +274,22 @@ function catalogUrl(apiBaseUrl: string): string { return url.toString(); } +/** An API base URL without its query or fragment, which can carry signed values. */ +function loggableBaseUrl(apiBaseUrl: string): string { + try { + const url = new URL(apiBaseUrl); + return `${url.origin}${url.pathname}`; + } catch { + return "[invalid URL]"; + } +} + +/** Strip the query and fragment from every URL an error message quotes. */ +function loggableErrorMessage(error: unknown): string { + const message = error instanceof Error ? error.message : String(error); + return message.replace(/(https?:\/\/[^\s?#"'<>)]*)[?#][^\s"'<>)]*/g, "$1"); +} + async function fetchCatalog( options: VeryfrontCloudCatalogScope, ): Promise { @@ -328,15 +344,15 @@ function refresh( if (!loggedFailures.has(key)) { if (loggedFailures.size >= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) loggedFailures.clear(); loggedFailures.add(key); - // Names the scope, never the credential. + // Names the scope, never the credential or a signed query value. logger.warn( stale ? "Veryfront Cloud model catalog refresh failed; the last loaded catalog stays in use" : "Veryfront Cloud model catalog is unavailable; model facts fall back to the built-in list", { - apiBaseUrl: options.apiBaseUrl, + apiBaseUrl: loggableBaseUrl(options.apiBaseUrl), projectSlug: options.projectSlug, - error: error instanceof Error ? error.message : String(error), + error: loggableErrorMessage(error), }, ); } diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts index 29be66d5fb..a081f95ae5 100644 --- a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -400,6 +400,29 @@ describe("provider/veryfront-cloud/catalog-client", () => { }); }); + it("never logs a signed query value from the API base URL or the error text", async () => { + const warnings: unknown[] = []; + const originalWarn = logger.warn; + logger.warn = (_message: string, fields?: unknown) => warnings.push(fields); + const signed = { ...LOAD, apiBaseUrl: `${API_BASE_URL}/tenant/?scope=signed-secret-value` }; + const stub = recordingFetch(() => + Promise.reject( + new TypeError( + `error sending request for url (${API_BASE_URL}/tenant/ai/models?scope=signed-secret-value)`, + ), + ) + ); + try { + await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(signed)); + } finally { + logger.warn = originalWarn; + } + + assertEquals(warnings.length, 1); + assertEquals(JSON.stringify(warnings).includes("signed-secret-value"), false); + assertEquals((warnings[0] as { apiBaseUrl?: string }).apiBaseUrl, `${API_BASE_URL}/tenant/`); + }); + it("logs one warning per failing scope, naming the scope but never the credential", async () => { const warnings: { message: string; fields: unknown }[] = []; const originalWarn = logger.warn; From 7aa40a2fa8c5537656878da562a9969090c9185d Mon Sep 17 00:00:00 2001 From: Koji Wakayama Date: Sun, 27 Sep 2026 14:01:03 +0200 Subject: [PATCH 17/17] fix(runtime): record native controls by served surface The durable model-call record picks Anthropic and Google controls and reasoning from the surface a Veryfront Cloud model settled on, so a newly served provider on those surfaces records what its native builder sent. --- .../model-call-context-request.test.ts | 25 +++++++++++++++++ src/runtime/model-call-context-request.ts | 27 +++++++++++++------ 2 files changed, 44 insertions(+), 8 deletions(-) diff --git a/src/runtime/model-call-context-request.test.ts b/src/runtime/model-call-context-request.test.ts index 16c92501ea..c6c975787b 100644 --- a/src/runtime/model-call-context-request.test.ts +++ b/src/runtime/model-call-context-request.test.ts @@ -9,6 +9,7 @@ import { resolveVeryfrontCloudOpenAITransport, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { buildModelCallContextRequest } from "#veryfront/runtime/model-call-context-request.ts"; +import { registerVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { buildOpenAIChatRequest } from "../../extensions/ext-llm-openai/src/openai-chat-request-builder.ts"; import { buildOpenAIResponsesRequest } from "../../extensions/ext-llm-openai/src/openai-responses-request-builder.ts"; import { buildAnthropicMessagesRequest } from "../../extensions/ext-llm-anthropic/src/anthropic-request-builder.ts"; @@ -328,6 +329,30 @@ describe("model call request projection", () => { assertEquals((body.output_config as Record).effort, undefined); }); + it("records native controls by served surface for a newly served provider", () => { + const options: ModelRuntimeCallOptions = { + prompt, + providerOptions: { + anthropic: { thinking: { type: "adaptive" }, output_config: { effort: "high" } }, + }, + }; + const project = (modelProvider: string) => { + const model = { provider: "veryfront-cloud", modelProvider, modelId: "m1" }; + registerVeryfrontCloudModelFacts(model as never, () => + ({ + provider: modelProvider, + surface: "anthropic", + native: false, + transportPlan: "chat-completions", + }) as never); + return buildModelCallContextRequest(model, options); + }; + + // A provider served on the Anthropic surface records what anthropic/* records. + assertEquals(project("acme"), project("anthropic")); + assertEquals(project("acme")?.reasoning?.enabled, true); + }); + for (const modelId of ["gpt-5.4", "gpt-5.5"]) { it(`matches ${modelId} Cloud Chat reasoning with and without function tools`, () => { const catalogId = `openai/${modelId}`; diff --git a/src/runtime/model-call-context-request.ts b/src/runtime/model-call-context-request.ts index 900bb1f8a8..9c417fa7e3 100644 --- a/src/runtime/model-call-context-request.ts +++ b/src/runtime/model-call-context-request.ts @@ -53,7 +53,8 @@ function readProviderControl( ): PropertyDescriptor | undefined { const provider = resolveModelCallProvider(model); let selected: PropertyDescriptor | undefined; - for (const name of [provider, model.provider ?? provider]) { + // The protocol's bucket first, so a provider-named bucket still takes precedence. + for (const name of [resolveModelCallProtocol(model), provider, model.provider ?? provider]) { if (!name) continue; const bucket = readOwnEnumerableDataDescriptor(options.providerOptions, name)?.value; if (Array.isArray(bucket)) continue; @@ -130,9 +131,9 @@ function resolvePersistedControls( model: ModelCallRuntimeMetadata, options: ModelCallRequestSource, ): ModelCallRequestSource { - const provider = resolveModelCallProvider(model); - if (provider === "anthropic") return resolveAnthropicControls(model, options); - if (provider === "google") return resolveGoogleControls(model, options); + const protocol = resolveModelCallProtocol(model); + if (protocol === "anthropic") return resolveAnthropicControls(model, options); + if (protocol === "google") return resolveGoogleControls(model, options); if (!usesOpenAIBuilder(model)) { return options; } @@ -278,6 +279,18 @@ function buildModelCallRequest( } /** Resolve the canonical provider recorded by the existing durable contract. */ +/** + * The wire protocol a model's request is built for. A Veryfront Cloud model + * speaks the surface it settled on, so a newly served provider on the Anthropic + * or Google surface records the same native controls as `anthropic/*` or + * `google/*`. Other models are identified by their provider name. + */ +function resolveModelCallProtocol(model: ModelCallRuntimeMetadata): string | undefined { + const surface = readVeryfrontCloudModelFacts(model)?.surface; + if (surface === "anthropic" || surface === "google") return surface; + return resolveModelCallProvider(model); +} + export function resolveModelCallProvider(model: ModelCallRuntimeMetadata): string | undefined { if (typeof model.modelProvider === "string" && model.modelProvider !== "") { return model.modelProvider; @@ -289,8 +302,7 @@ function resolvePersistedReasoning( model: ModelCallRuntimeMetadata, options: ModelCallRequestSource, ): RuntimeReasoningOption | undefined { - const modelProvider = resolveModelCallProvider(model); - if (modelProvider === "google") return resolveGoogleReasoning(model, options); + if (resolveModelCallProtocol(model) === "google") return resolveGoogleReasoning(model, options); if (usesOpenAIBuilder(model) && typeof model.modelId === "string") { const neutral = resolveOpenAINeutralReasoning(model, options); const transport = managedOpenAITransport(model, options); @@ -366,10 +378,9 @@ function resolveNonOpenAIReasoning( model: ModelCallRuntimeMetadata, options: ModelCallRequestSource, ): RuntimeReasoningOption | undefined { - const modelProvider = resolveModelCallProvider(model); // The Anthropic request builder only gives neutral reasoning precedence when // it enables thinking; otherwise a raw provider thinking config remains effective. - if (modelProvider !== "anthropic" || options.reasoning?.enabled === true) { + if (resolveModelCallProtocol(model) !== "anthropic" || options.reasoning?.enabled === true) { return options.reasoning; }