diff --git a/CHANGELOG.md b/CHANGELOG.md index 574ec82643..287c45cdde 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -20,6 +20,47 @@ For finish-required streams in the active lifecycle, a completed turn containing only unavailable tool calls now permits the same recovery turn as the legacy lifecycle. Rejected tools are not executed, and malformed or empty handoff requests still fail. +### Changed: Veryfront Cloud model facts come from the served model catalog + +`veryfront-cloud/*` models now read their facts from the model catalog your +Veryfront API serves at `/ai/models`, instead of from a table shipped in +this package. The facts are the wire protocol, whether the Responses operation +is served, the default thinking budget, the two chat completions capability +flags, short model aliases such as `opus`, and the default model. For the +models this package listed, the served facts are the same, so requests are +unchanged. + +- Model construction stays synchronous and makes no network call. The catalog + loads on the first async step of a model (`prepare`, `doGenerate` or + `doStream`), with the same credentials and project as inference, and is + cached for five minutes per API, project and credential. +- Until the catalog has loaded for the credentials in use, and whenever it + cannot be loaded, the facts shipped with this package apply, as in the + previous release. A model whose first load failed tries again on a later + call. +- A model keeps the facts it settled with for its lifetime. A catalog refreshed + later applies to models constructed after the refresh. +- Agents resolve a short alias or a provider the platform added after this + release through Veryfront Cloud once the catalog has loaded, after the + built-in aliases, so a bare vendor model name keeps its meaning for your own + provider key. +- A model the catalog does not list is refused only against a catalog loaded + within the last five minutes, or the shipped list before any has loaded. A + model enabled since the catalog was cached is not refused; the catalog is + refreshed first. +- `loadVeryfrontCloudModelCatalog()` loads the catalog for the Veryfront Cloud + credentials in effect, so synchronous helpers such as + `resolveVeryfrontCloudModelId("opus")` and + `resolveVeryfrontCloudModelThinking()` read served facts, including models + the platform added after this release. +- `resolveVeryfrontCloudDefaultModelId()` returns the default model the + catalog names, or the built-in default before it loads. + `VeryfrontCloudModelId` types a model ID as `/`. +- `VERYFRONT_CLOUD_CHAT_MODELS`, `findVeryfrontCloudModel`, + `findVeryfrontCloudModelByModelId` and `groupVeryfrontCloudModelsByProvider` + are deprecated. They still return the list shipped with this package, and a + later release removes them. + ### Changed: Veryfront Cloud models call the vendor-neutral endpoints `veryfront-cloud/*` models that speak the OpenAI protocol now send requests to diff --git a/docs/api-reference/veryfront/provider.md b/docs/api-reference/veryfront/provider.md index ab1ee0f5e6..f635d54e5f 100644 --- a/docs/api-reference/veryfront/provider.md +++ b/docs/api-reference/veryfront/provider.md @@ -63,44 +63,48 @@ Clear all registered model providers and reset lazy built-ins (for testing). ### Components -| Name | Description | Source | -| ---------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `DEFAULT_VERYFRONT_CLOUD_MODEL_ID` | Default Veryfront Cloud model ID used when no model is configured. Update this when the current default is deprecated - otherwise the default path silently breaks for users who have not set an explicit model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VERYFRONT_CLOUD_CHAT_MODELS` | Shared Veryfront Cloud chat models value. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VERYFRONT_CLOUD_MODEL_PREFIX` | Shared Veryfront Cloud model prefix value. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ---------------------------------- | -------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------ | +| `DEFAULT_VERYFRONT_CLOUD_MODEL_ID` | Short ID of the built-in default model, used when no model is configured and the served catalog has not been loaded. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VERYFRONT_CLOUD_CHAT_MODELS` | Chat models shipped with this package. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `VERYFRONT_CLOUD_MODEL_PREFIX` | Shared Veryfront Cloud model prefix value. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Functions -| Name | Description | Source | -| ---------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `findVeryfrontCloudModel` | Find Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `findVeryfrontCloudModelByModelId` | Find Veryfront Cloud model by model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `groupVeryfrontCloudModelsByProvider` | Group Veryfront Cloud models by provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelId` | Resolves Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `tryGetVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ---------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------ | +| `clearModelProviders` | Clear all registered model providers and reset lazy built-ins (for testing). | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ensureModelReady` | Eagerly verify that the resolved model's runtime is available. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `findVeryfrontCloudModel` | Find a shipped chat model by its short id. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `findVeryfrontCloudModelByModelId` | Find a shipped chat model by its provider-qualified id, in any provider spelling. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `getRegisteredModelProviders` | Get provider names available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `getVeryfrontCloudProviderFromModelId` | Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `groupVeryfrontCloudModelsByProvider` | Group the shipped chat models by provider, in display order. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.deprecated.ts) | +| `hasModelProvider` | Check whether a model provider is available in the current scope. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `loadVeryfrontCloudModelCatalog` | Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud credentials and project in effect, so model facts read synchronously afterward (thinking defaults, short aliases such as `opus`, the default model) come from it. Resolves to whether a catalog is available. Never throws. When a refresh fails, the last catalog loaded for these credentials stays in use; only when none has loaded (no credentials, or no load has succeeded yet) do the facts shipped with this package apply. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/shared.ts) | +| `normalizeVeryfrontCloudModelId` | Normalizes Veryfront Cloud model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `registerModelProvider` | Register a custom model provider factory for the active project scope or application bootstrap. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveModel` | Resolve a "provider/model" string to a framework-compatible model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `resolveVeryfrontCloudDefaultModelId` | Provider-qualified ID of the default model: the one the served catalog names once it is loaded, otherwise the built-in default. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudGatewayModelId` | Prefix a model ID so it resolves through the Veryfront Cloud gateway, including a provider this package does not list. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelId` | Resolve a model ID or short alias to a provider-qualified model ID. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudModelThinking` | Resolves Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudReasoningOption` | Resolves provider-neutral runtime reasoning for a Veryfront Cloud model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `resolveVeryfrontCloudThinkingProviderOptions` | Options accepted by resolve Veryfront Cloud thinking provider. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `tryGetVeryfrontCloudProviderFromModelId` | | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Types -| Name | Description | Source | -| ----------------------------------- | ------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | -| `ModelProviderFactory` | Public API contract for model provider factory. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ModelProviderRegistrationDisposer` | | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | -| `ModelRuntime` | Public API contract for model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/types.ts) | -| `VeryfrontCloudChatModel` | Public API contract for Veryfront Cloud chat model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VeryfrontCloudModelThinkingConfig` | Configuration used by Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | -| `VeryfrontCloudProviderId` | A Veryfront Cloud provider ID: a listed provider, or any other well-formed provider string. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| Name | Description | Source | +| ----------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------- | +| `ModelProviderFactory` | Public API contract for model provider factory. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ModelProviderRegistrationDisposer` | | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/model-registry.ts) | +| `ModelRuntime` | Public API contract for model runtime. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/types.ts) | +| `VeryfrontCloudChatModel` | Public API contract for Veryfront Cloud chat model. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudModelId` | A provider-qualified Veryfront Cloud model ID, for example `anthropic/claude-sonnet-4-6`. Any provider and model the platform serves fits, so a new model needs no release of this package. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudModelThinkingConfig` | Configuration used by Veryfront Cloud model thinking. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudProviderId` | A Veryfront Cloud provider ID: a listed provider, or any other well-formed provider string. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | +| `VeryfrontCloudRuntimeModelId` | A model ID routed through Veryfront Cloud: `veryfront-cloud//`. | [source](https://github.com/veryfront/veryfront-code/blob/main/src/provider/veryfront-cloud/model-catalog.ts) | ### Constants diff --git a/src/agent/hosted/application-model-resolver.test.ts b/src/agent/hosted/application-model-resolver.test.ts index 103066df2e..41aed0084f 100644 --- a/src/agent/hosted/application-model-resolver.test.ts +++ b/src/agent/hosted/application-model-resolver.test.ts @@ -3,6 +3,10 @@ import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/te import { describe, it } from "#veryfront/testing/bdd.ts"; import { revokeModelRuntimeResolver } from "../runtime/model-transport.ts"; import { createHostedApplicationModelResolver } from "./application-model-resolver.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForScopeForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; const modelId = "veryfront-cloud/openai/gpt-4o"; @@ -150,4 +154,33 @@ describe("hosted application model authority", () => { "authority is revoked", ); }); + it("exposes the reconciliation hook of the protocol a model settles on", async () => { + const servedId = "veryfront-cloud/acme/acme-gemini"; + const options = { ...resolverOptions(), allowedModelIds: new Set([servedId]) }; + try { + const model = createHostedApplicationModelResolver(options)(servedId)!; + // Cold, an unlisted provider is built as an OpenAI-protocol model. + assertEquals(model._reconcileProviderMetadata, undefined); + + __setVeryfrontCloudCatalogForScopeForTests( + { apiBaseUrl: options.apiBaseUrl, apiToken: options.authToken }, + { + models: [{ + id: "acme-gemini", + modelId: "acme/acme-gemini", + provider: "acme", + surface: "google", + operations: ["chat-completions"], + aliases: [], + capabilities: {}, + }], + }, + ); + await model.prepare!(); + + assert(typeof model._reconcileProviderMetadata === "function"); + } finally { + __resetVeryfrontCloudCatalogForTests(); + } + }); }); diff --git a/src/agent/hosted/application-model-resolver.ts b/src/agent/hosted/application-model-resolver.ts index 1865eef432..5283a6fa5d 100644 --- a/src/agent/hosted/application-model-resolver.ts +++ b/src/agent/hosted/application-model-resolver.ts @@ -6,6 +6,10 @@ import { type VeryfrontCloudContext, } from "#veryfront/provider/veryfront-cloud/context.ts"; import { createVeryfrontCloudModel } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import { + readVeryfrontCloudModelFacts, + registerVeryfrontCloudModelFacts, +} from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { requireSecureInferenceApiBaseUrl } from "#veryfront/provider/veryfront-cloud/shared.ts"; import { type AgentModelRuntimeResolver, @@ -135,15 +139,30 @@ export function createHostedApplicationModelResolver(input: { assertCredentialActive: assertActive, }) ); - const reconcile = model._reconcileProviderMetadata; + // Metadata is read from the model on each access: a Veryfront Cloud model + // may settle a different protocol and capabilities on its first async step. const proxy: ModelRuntime = Object.freeze({ - specificationVersion: model.specificationVersion, - provider: model.provider, - modelProvider: model.modelProvider, - modelId: model.modelId, - executionMode: model.executionMode, - runtimeCapabilities: model.runtimeCapabilities, - _generateViaStream: model._generateViaStream, + get specificationVersion() { + return model.specificationVersion; + }, + get provider() { + return model.provider; + }, + get modelProvider() { + return model.modelProvider; + }, + get modelId() { + return model.modelId; + }, + get executionMode() { + return model.executionMode; + }, + get runtimeCapabilities() { + return model.runtimeCapabilities; + }, + get _generateViaStream() { + return model._generateViaStream; + }, async prepare(abortSignal?: AbortSignal) { await run(callScope(abortSignal), (signal) => model.prepare?.(signal)); }, @@ -175,19 +194,24 @@ export function createHostedApplicationModelResolver(input: { throw error; } }, - ...(typeof reconcile === "function" - ? { - async _reconcileProviderMetadata(options: { - providerMetadata: Record; - suppressedToolCalls: readonly { id: string; name: string }[]; - abortSignal?: AbortSignal; - }) { - return await run(callScope(options.abortSignal), (signal) => - reconcile.call(model, { ...options, abortSignal: signal })); - }, - } - : {}), + // Read when described and when invoked: a model rebuilt onto another + // protocol may gain or lose its reconciliation hook. + get _reconcileProviderMetadata() { + const reconcile = model._reconcileProviderMetadata; + if (typeof reconcile !== "function") return undefined; + return async (options: { + providerMetadata: Record; + suppressedToolCalls: readonly { id: string; name: string }[]; + abortSignal?: AbortSignal; + }) => + await run( + callScope(options.abortSignal), + (signal) => reconcile.call(model, { ...options, abortSignal: signal }), + ); + }, }); + // The proxy reports the facts of the model it calls. + registerVeryfrontCloudModelFacts(proxy, () => readVeryfrontCloudModelFacts(model)!); models.set(id, proxy); return proxy; }; diff --git a/src/agent/hosted/cloud-chat-execution-preparation.ts b/src/agent/hosted/cloud-chat-execution-preparation.ts index 08fd6c50e8..594b6860c1 100644 --- a/src/agent/hosted/cloud-chat-execution-preparation.ts +++ b/src/agent/hosted/cloud-chat-execution-preparation.ts @@ -1,5 +1,12 @@ import { resolveVeryfrontCloudModelThinking } from "#veryfront/provider"; -import { runWithVeryfrontCloudContext } from "#veryfront/provider/veryfront-cloud/context.ts"; +import { + runWithVeryfrontCloudContext, + runWithVeryfrontCloudContextAsync, +} from "#veryfront/provider/veryfront-cloud/context.ts"; +import { loadVeryfrontCloudModelCatalog } from "#veryfront/provider/veryfront-cloud/shared.ts"; + +/** Longest preparation waits for the served catalog before resolving the model. */ +const CATALOG_MAX_WAIT_MS = 3_000; import { resolveRuntimeModel } from "../runtime/model-resolution.ts"; import type { HostedChatRuntimeCreationResult } from "./chat-runtime-contract.ts"; import { @@ -88,16 +95,25 @@ export async function prepareVeryfrontCloudHostedChatExecution< HostedChatExecutionPreparationResult > { const { logger, rootRun, ...preparationInput } = input; + const cloudContext = { + apiBaseUrl: String(input.apiUrl), + apiToken: input.request.authToken, + projectSlug: input.request.projectSlug, + serviceLayer: "cloud" as const, + }; + // Model ids and thinking defaults resolve against the served catalog, loaded + // and read under the request's own credentials and project. + await runWithVeryfrontCloudContextAsync( + cloudContext, + () => loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS }), + ); const resolveModelId = (modelId: string | undefined): string | undefined => runWithVeryfrontCloudContext( - { - apiBaseUrl: String(input.apiUrl), - apiToken: input.request.authToken, - projectSlug: input.request.projectSlug, - serviceLayer: "cloud", - }, + cloudContext, () => modelId === undefined ? undefined : resolveRuntimeModel(modelId), ); + const resolveModelThinking: typeof resolveVeryfrontCloudModelThinking = (modelId) => + runWithVeryfrontCloudContext(cloudContext, () => resolveVeryfrontCloudModelThinking(modelId)); return await prepareHostedChatExecution({ ...preparationInput, @@ -106,6 +122,6 @@ export async function prepareVeryfrontCloudHostedChatExecution< logger, }), resolveModelId, - resolveModelThinking: resolveVeryfrontCloudModelThinking, + resolveModelThinking, }); } diff --git a/src/agent/hosted/context-summary-generator.ts b/src/agent/hosted/context-summary-generator.ts index f912786583..50cc709008 100644 --- a/src/agent/hosted/context-summary-generator.ts +++ b/src/agent/hosted/context-summary-generator.ts @@ -4,7 +4,12 @@ import { resolveVeryfrontCloudGatewayModelId, resolveVeryfrontCloudModelId, } from "../../provider/index.ts"; -import { runWithVeryfrontCloudContextAsync } from "#veryfront/provider/veryfront-cloud/context.ts"; +import { + runWithVeryfrontCloudContext, + runWithVeryfrontCloudContextAsync, + type VeryfrontCloudContext, +} from "#veryfront/provider/veryfront-cloud/context.ts"; +import { loadVeryfrontCloudModelCatalog } from "#veryfront/provider/veryfront-cloud/shared.ts"; import { generateText } from "../../runtime/runtime-bridge.ts"; import { redactSensitive, sanitizeUrlCredentials } from "#veryfront/utils"; import type { TextGenerationRuntimeMessage } from "../runtime/text-generation-runtime-message-types.ts"; @@ -168,12 +173,7 @@ async function summarizeSegment(input: { const generate = input.options.generateText ?? generateText; const resolve = input.options.resolveModel ?? resolveModel; const result = await runWithVeryfrontCloudContextAsync( - { - apiBaseUrl: input.options.apiUrl.toString(), - apiToken: input.options.authToken, - projectSlug: input.options.projectSlug ?? undefined, - serviceLayer: "cloud", - }, + summaryCloudContext(input.options), () => Promise.resolve(generate({ model: resolve(input.modelId), @@ -192,6 +192,20 @@ async function summarizeSegment(input: { return result.text.trim(); } +function summaryCloudContext( + options: VeryfrontCloudContextSummaryGeneratorOptions, +): VeryfrontCloudContext { + return { + apiBaseUrl: options.apiUrl.toString(), + apiToken: options.authToken, + projectSlug: options.projectSlug ?? undefined, + serviceLayer: "cloud", + }; +} + +/** Longest summary generation waits for the served catalog before resolving its model. */ +const CATALOG_MAX_WAIT_MS = 3_000; + function resolveSummaryModelId(model: string | undefined): string { const cloudModelId = resolveVeryfrontCloudModelId(model); return resolveVeryfrontCloudGatewayModelId(cloudModelId) ?? cloudModelId; @@ -202,7 +216,17 @@ export function createVeryfrontCloudContextSummaryGenerator( options: VeryfrontCloudContextSummaryGeneratorOptions, ): ContextSummaryGenerator { return async ({ messagesToSummarize, retainedMessages, customInstructions }) => { - const modelId = resolveSummaryModelId(options.model); + // The model resolves against the served catalog loaded for the same + // credentials and project the summary calls use. + const cloudContext = summaryCloudContext(options); + await runWithVeryfrontCloudContextAsync( + cloudContext, + () => loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS }), + ); + const modelId = runWithVeryfrontCloudContext( + cloudContext, + () => resolveSummaryModelId(options.model), + ); const chunks = chunkSerializedMessages(messagesToSummarize, options.maxInputTokens); let summary = ""; diff --git a/src/agent/hosted/default-chat-runtime.test.ts b/src/agent/hosted/default-chat-runtime.test.ts index 4922c0e703..2e3d11db59 100644 --- a/src/agent/hosted/default-chat-runtime.test.ts +++ b/src/agent/hosted/default-chat-runtime.test.ts @@ -8,6 +8,7 @@ import { assertStringIncludes, } from "#veryfront/testing/assert.ts"; import { it } from "#veryfront/testing/bdd.ts"; +import { useServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; import { deleteEnv, getEnv, setEnv } from "#veryfront/compat/process.ts"; import { refreshEnvironmentConfig } from "#veryfront/config/environment-config.ts"; import { clearModelProviders, type ModelRuntime, registerModelProvider } from "#veryfront/provider"; @@ -465,6 +466,7 @@ it("applies refreshed structured system messages in hosted chat", async () => { }); Deno.test("createDefaultHostedChatRuntime builds a cloud-backed hosted runtime", async () => { + using _catalog = useServedCatalogForTests(); let capturedContext: DefaultHostedChatRuntimeTaskContext | undefined; let capturedCapability: unknown; const runEventWriterCapability = createHostedRunEventWriterCapability({ @@ -1027,6 +1029,7 @@ Deno.test("hosted first provider call filters skill tools for every tool selecto }); Deno.test("createDefaultHostedChatRuntime forwards hosted project slug to integration discovery", async () => { + using _catalog = useServedCatalogForTests(); const previousApiBaseUrl = getEnv("VERYFRONT_API_BASE_URL"); const previousApiToken = getEnv("VERYFRONT_API_TOKEN"); const previousProjectSlug = getEnv("VERYFRONT_PROJECT_SLUG"); @@ -1106,6 +1109,7 @@ Deno.test("createDefaultHostedChatRuntime forwards hosted project slug to integr }); Deno.test("createDefaultHostedChatRuntime keeps per-run host tools out of the global registry", async () => { + using _catalog = useServedCatalogForTests(); try { const createRuntime = (description: string) => createDefaultHostedChatRuntime({ diff --git a/src/agent/hosted/default-chat-runtime.ts b/src/agent/hosted/default-chat-runtime.ts index 7476324247..3b9f305608 100644 --- a/src/agent/hosted/default-chat-runtime.ts +++ b/src/agent/hosted/default-chat-runtime.ts @@ -12,11 +12,13 @@ import { runWithRequestContext as runWithProjectRequestContext, } from "#veryfront/platform/adapters/fs/veryfront/request-context.ts"; import { + currentVeryfrontCloudCatalogScopeKey, resolveVeryfrontCloudModelId, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; +import { loadVeryfrontCloudModelCatalog } from "#veryfront/provider/veryfront-cloud/shared.ts"; import { runWithVeryfrontCloudContext, runWithVeryfrontCloudContextAsync, @@ -407,12 +409,19 @@ function withoutHostedCredentials(input: { cloudContext: VeryfrontCloudContext; operation: () => Promise; }): Promise { + // The run's catalog is named by its non-secret scope key, so model reads in + // project code use the catalog the run loaded without holding its credential. + const catalogScopeKey = runWithVeryfrontCloudContext( + input.cloudContext, + currentVeryfrontCloudCatalogScopeKey, + ); const publicCloudContext: VeryfrontCloudContext = { apiBaseUrl: input.cloudContext.apiBaseUrl, projectSlug: input.cloudContext.projectSlug, serviceLayer: input.cloudContext.serviceLayer, billingGroupId: input.cloudContext.billingGroupId, billingGroupUsed: input.cloudContext.billingGroupUsed, + ...(catalogScopeKey ? { catalogScopeKey } : {}), }; const runWithPublicCloudContext = () => runWithVeryfrontCloudContextAsync(publicCloudContext, input.operation); @@ -524,6 +533,9 @@ function runWithDefaultHostedRequestContext( ); } +/** Longest a run waits for the served catalog before resolving a short model alias. */ +const CATALOG_ALIAS_MAX_WAIT_MS = 3_000; + /** Create default hosted chat runtime. */ export async function createDefaultHostedChatRuntime( input: CreateDefaultHostedChatRuntimeOptions, @@ -533,11 +545,21 @@ export async function createDefaultHostedChatRuntime( return await runWithEffectiveSourceIntegrationPolicy( input.sourceIntegrationPolicy, async () => { - const modelId = resolveVeryfrontCloudModelId(input.options.model); const cloudContext = createCloudContext({ config: input.config, options: input.options, }); + // A short model alias resolves through the served catalog. It is loaded + // and read under the run's own credentials and project, the same scope + // the run's later model reads use. + await runWithVeryfrontCloudContextAsync( + cloudContext, + () => loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_ALIAS_MAX_WAIT_MS }), + ); + const modelId = runWithVeryfrontCloudContext( + cloudContext, + () => resolveVeryfrontCloudModelId(input.options.model), + ); const taskContext = input.createTaskContext ? input.createTaskContext({ options: input.options, modelId }) : createDefaultTaskContext({ options: input.options, modelId }); diff --git a/src/agent/hosted/executor-model-bridge.test.ts b/src/agent/hosted/executor-model-bridge.test.ts index fba4bcea19..b35a16522a 100644 --- a/src/agent/hosted/executor-model-bridge.test.ts +++ b/src/agent/hosted/executor-model-bridge.test.ts @@ -14,6 +14,7 @@ import { createExecutorModelBroker, createExecutorModelRuntimeResolver, } from "./executor-model-bridge.ts"; +import { registerVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; const modelId = "veryfront-cloud/openai/synthetic-model"; const allowedModelIds = new Set([modelId]); @@ -977,4 +978,33 @@ describe("executor managed model bridge", () => { } } }); + + it("describes a served model only after its preparation settles, however long it takes", async () => { + // Longer than any fixed wait the broker might apply before describing. + const settleMs = 3_200; + let settledProvider = "openai"; + const model = stubModel({ + async prepare() { + await new Promise((resolve) => setTimeout(resolve, settleMs)); + settledProvider = "anthropic"; + }, + }); + Object.defineProperty(model, "modelProvider", { get: () => settledProvider }); + registerVeryfrontCloudModelFacts(model, () => + ({ + surface: settledProvider, + nativeProtocol: false, + }) as never); + const broker = createExecutorModelBroker({ + allowedModelIds, + resolveModelRuntime: () => model, + }); + const metadata = await broker.get("model.metadata")!.handle!({}, { + binding: { allocationId: "allocation-test", generation: 1, invocationId: "invocation-test" }, + signal: new AbortController().signal, + deadline: Date.now() + 60_000, + }) as { modelProvider?: string }[]; + + assertEquals(metadata.map((entry) => entry.modelProvider), ["anthropic"]); + }); }); diff --git a/src/agent/hosted/executor-model-bridge.ts b/src/agent/hosted/executor-model-bridge.ts index 53d29a0051..ee37f948cc 100644 --- a/src/agent/hosted/executor-model-bridge.ts +++ b/src/agent/hosted/executor-model-bridge.ts @@ -1,4 +1,5 @@ const hasOwn = Object.hasOwn; +import { readVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import type { ModelRuntime, ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { createPrivateReadableStream } from "#veryfront/security/private-stream.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; @@ -132,9 +133,15 @@ export function createExecutorModelBroker(options: { return new Map([ ["model.metadata", { mode: "unary", - handle(input, context) { + async handle(input, context) { parseExecutorModelData(getExecutorModelEmptySchema(), input); context.signal.throwIfAborted(); + // A Veryfront Cloud model settles its protocol and capabilities on its + // first async step; settle each before describing it to the executor. + await Promise.all( + [...allowed].map((id) => settleForMetadata(getModel(id), context.signal)), + ); + context.signal.throwIfAborted(); const metadata = [...allowed].map((id) => modelMetadata(id, getModel(id))); return executorModelJson( parseExecutorModelData(getExecutorModelMetadataSchema(), executorModelJson(metadata)), @@ -269,6 +276,19 @@ export function createExecutorModelBroker(options: { ]); } +/** + * Settle a Veryfront Cloud model before its metadata is read, so the executor + * is never told about a construction a pending rebuild can still replace. The + * wait is bounded by the catalog request's own timeout and the caller's signal. + * A failed load leaves the model as built; its call surfaces any failure. + */ +async function settleForMetadata(model: ModelRuntime, signal: AbortSignal): Promise { + if (readVeryfrontCloudModelFacts(model) === undefined || typeof model.prepare !== "function") { + return; + } + await Promise.resolve(model.prepare(signal)).catch(() => {}); +} + function modelMetadata(id: string, model: ModelRuntime) { return { id, diff --git a/src/agent/hosted/executor-runtime-prepare.test.ts b/src/agent/hosted/executor-runtime-prepare.test.ts index 4d9b3e7946..13db959a58 100644 --- a/src/agent/hosted/executor-runtime-prepare.test.ts +++ b/src/agent/hosted/executor-runtime-prepare.test.ts @@ -1,7 +1,12 @@ import "#veryfront/schemas/_test-setup.ts"; import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; import { PERMISSION_DENIED } from "#veryfront/errors"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; import type { ModelRuntime, ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { defineSchema } from "#veryfront/schemas/index.ts"; @@ -166,6 +171,8 @@ async function prepare( } describe("executor runtime preparation", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("extracts inline tools under the source policy in the full-runtime profile", async () => { const policy = { schemaVersion: 1 as const, mode: "allowlist" as const, integrations: {} }; const observed: unknown[] = []; @@ -847,6 +854,8 @@ function syntheticRemoteTool(name: string): ToolDefinition { } describe("executor runtime preparation review regressions", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); const outputCases: { name: string; request: Record; @@ -2230,6 +2239,110 @@ Synthetic source instructions.`, }); } + it("reads thinking defaults from the catalog the loadModelCatalog facade loads", async () => { + __setVeryfrontCloudCatalogForTests(undefined); + const selectedModel = "veryfront-cloud/anthropic/claude-sonnet-4-6"; + let captured: ModelRuntimeCallOptions | undefined; + let catalogLoads = 0; + const f = fixture({ + grant: { + ...grant, + defaultModelId: selectedModel, + models: new Map([[selectedModel, { maxOutputTokens: 8192, providerToolNames: [] }]]), + }, + facades: { + loadModelCatalog: () => { + catalogLoads++; + __setVeryfrontCloudCatalogForTests({ + models: [{ + id: "claude-sonnet-4-6", + modelId: "anthropic/claude-sonnet-4-6", + provider: "anthropic", + surface: "anthropic", + operations: ["messages"], + capabilities: { + thinking: true, + reasoning_mode: "budget", + reasoning_budget_tokens: 1024, + }, + }], + }); + return Promise.resolve(); + }, + resolveModelRuntime: () => ({ + ...model, + doStream: (options) => { + captured = options as ModelRuntimeCallOptions; + return finishStream(); + }, + }), + }, + }); + try { + await Array.fromAsync(await preparedStream(f)); + assert(captured); + assertEquals(catalogLoads, 1); + assertEquals(captured.reasoning, { enabled: true, budgetTokens: 1024 }); + assertEquals(captured.maxOutputTokens, 7168); + } finally { + await f.owner.close(); + } + }); + + it("calls loadModelCatalog only after every grant check passes", async () => { + let catalogLoads = 0; + const f = fixture({ + facades: { + loadModelCatalog: () => { + catalogLoads++; + return Promise.resolve(); + }, + }, + }); + try { + assertEquals( + await prepare( + f.owner, + { agentId: "coder", modelId: "veryfront-cloud/not-granted/x" } as JsonValue, + ), + { ok: false, code: "EXECUTOR_RUNTIME_NOT_GRANTED" }, + ); + assertEquals(catalogLoads, 0); + } finally { + await f.owner.close(); + } + }); + + it("prepares with the shipped facts when loadModelCatalog fails", async () => { + __setVeryfrontCloudCatalogForTests(undefined); + const selectedModel = "veryfront-cloud/anthropic/claude-sonnet-4-6"; + let captured: ModelRuntimeCallOptions | undefined; + const f = fixture({ + grant: { + ...grant, + defaultModelId: selectedModel, + models: new Map([[selectedModel, { maxOutputTokens: 8192, providerToolNames: [] }]]), + }, + facades: { + loadModelCatalog: () => Promise.reject(new Error("catalog unavailable")), + resolveModelRuntime: () => ({ + ...model, + doStream: (options) => { + captured = options as ModelRuntimeCallOptions; + return finishStream(); + }, + }), + }, + }); + try { + await Array.fromAsync(await preparedStream(f)); + assert(captured); + assertEquals(captured.reasoning, { enabled: true, budgetTokens: 2048 }); + } finally { + await f.owner.close(); + } + }); + it("reserves the catalog thinking budget when the request omits thinking and output limits", async () => { const selectedModel = "veryfront-cloud/anthropic/claude-sonnet-4-6"; let captured: ModelRuntimeCallOptions | undefined; @@ -2289,14 +2402,14 @@ Synthetic source instructions.`, const selectedModel of [ modelId, "veryfront-cloud/anthropic/claude-sonnet-4-6", - "veryfront-cloud/anthropic/claude-opus-4-7", + "veryfront-cloud/anthropic/claude-opus-4-8", ] ) { for (const thinking of [{ enabled: false }, { enabled: true, budgetTokens: 4096 }]) { it(`carries explicit thinking ${thinking.enabled} into ${selectedModel} call data`, async () => { let captured: ModelRuntimeCallOptions | undefined; let sourceTransportCalls = 0; - const adaptive = selectedModel.endsWith("claude-opus-4-7") && thinking.enabled; + const adaptive = selectedModel.endsWith("claude-opus-4-8") && thinking.enabled; const f = fixture({ config: { thinking: { enabled: !thinking.enabled }, diff --git a/src/agent/hosted/runtime-preparation-core.ts b/src/agent/hosted/runtime-preparation-core.ts index a695c6f9dc..90427519ff 100644 --- a/src/agent/hosted/runtime-preparation-core.ts +++ b/src/agent/hosted/runtime-preparation-core.ts @@ -10,10 +10,10 @@ import { } from "#veryfront/security/private-promise.ts"; import type { JsonValue } from "#veryfront/schemas/index.ts"; import { + isVeryfrontCloudAnthropicSurfaceModel, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, - tryGetVeryfrontCloudProviderFromModelId, VERYFRONT_CLOUD_MODEL_PREFIX, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { getExecutorModelAdditiveReasoningTokens } from "#veryfront/agent/hosted/executor-model-grant.ts"; @@ -167,6 +167,15 @@ export type ExecutorRuntimePreparationGrant = Omit Promise; hostTools: ReadonlyMap; remoteToolSources: ReadonlyMap; projectSteering?: { @@ -325,6 +334,7 @@ export function createRuntimePreparationCore(input: RuntimePreparationCoreOption const installedModelResolver = input.facades.resolveModelRuntime; const facades: ExecutorRuntimeFacades = { resolveModelRuntime: snapshotFacadeMethod(input.facades, "resolveModelRuntime"), + loadModelCatalog: snapshotFacadeMethod(input.facades, "loadModelCatalog"), cleanup: snapshotFacadeMethod(input.facades, "cleanup"), projectSteering: snapshotSteeringFacade(input.facades.projectSteering), latestConversationUserText: snapshotFacadeMethod(input.facades, "latestConversationUserText"), @@ -476,14 +486,23 @@ export function createRuntimePreparationCore(input: RuntimePreparationCoreOption (request.maxOutputTokens !== undefined && request.maxOutputTokens > modelGrant.maxOutputTokens) ) refuse("EXECUTOR_RUNTIME_NOT_GRANTED"); + if (facades.loadModelCatalog) { + try { + await observePrivatePromise(facades.loadModelCatalog(context.signal)); + } catch { + // The reads below fall back to the shipped facts. + } + assertActive(); + } const thinking = request.thinking ?? definition.thinking ?? resolveVeryfrontCloudModelThinking(modelId); let availableOutputTokens = modelGrant.maxOutputTokens; - const modelProvider = tryGetVeryfrontCloudProviderFromModelId(modelId); - if (modelProvider === "anthropic") { + // The served surface decides the protocol, so a newly served provider on + // the Anthropic surface reserves its reasoning tokens like `anthropic/*`. + if (isVeryfrontCloudAnthropicSurfaceModel(modelId)) { try { const effectiveThinking = thinking ?? resolveVeryfrontCloudModelThinking(modelId); - const model = { id: modelId, modelId, provider: modelProvider }; + const model = { id: modelId, modelId, provider: "anthropic" }; const options = { reasoning: resolveVeryfrontCloudReasoningOption(modelId, effectiveThinking), providerOptions: resolveVeryfrontCloudThinkingProviderOptions( diff --git a/src/agent/hosted/veryfront-cloud-agent-service.test.ts b/src/agent/hosted/veryfront-cloud-agent-service.test.ts index f8fa333cf6..ca22f6a161 100644 --- a/src/agent/hosted/veryfront-cloud-agent-service.test.ts +++ b/src/agent/hosted/veryfront-cloud-agent-service.test.ts @@ -8,6 +8,7 @@ import { assertStrictEquals, } from "#veryfront/testing/assert.ts"; import { describe, it } from "#veryfront/testing/bdd.ts"; +import { useServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; import { resolve } from "node:path"; import { pathToFileURL } from "node:url"; import type { CreateSandboxBashTool } from "#veryfront/sandbox"; @@ -83,6 +84,7 @@ Deno.test("public agent service options expose deployment-owned remote MCP compo }); Deno.test("root and child runtimes use the deployment-owned remote MCP factory", async () => { + using _catalog = useServedCatalogForTests(); const createdConfigs: RemoteMCPToolSourceConfig[] = []; let failStudioListing = false; let modelCallCount = 0; diff --git a/src/agent/runtime/default-provider-options.test.ts b/src/agent/runtime/default-provider-options.test.ts index a34ed6aae6..c39ddef4b6 100644 --- a/src/agent/runtime/default-provider-options.test.ts +++ b/src/agent/runtime/default-provider-options.test.ts @@ -1,9 +1,16 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; describe("resolveProviderOptionsWithDefaults", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("enables Anthropic thinking by default for Anthropic models", () => { const result = resolveProviderOptionsWithDefaults( "anthropic/claude-sonnet-4-6", @@ -32,6 +39,27 @@ describe("resolveProviderOptionsWithDefaults", () => { }); }); + it("applies Anthropic defaults to a newly served provider on the Anthropic surface", () => { + __setVeryfrontCloudCatalogForTests({ + models: [{ + id: "m1", + modelId: "acme-labs/m1", + provider: "acme-labs", + surface: "anthropic", + operations: ["messages"], + aliases: [], + capabilities: { thinking: true, reasoning: true, reasoning_mode: "adaptive" }, + }], + }); + + const result = resolveProviderOptionsWithDefaults("veryfront-cloud/acme-labs/m1", undefined); + + assertEquals( + (result?.anthropic as { thinking?: unknown } | undefined)?.thinking, + { type: "adaptive", display: "summarized" }, + ); + }); + it("does not enable thinking for non-Anthropic models", () => { assertEquals( resolveProviderOptionsWithDefaults("openai/gpt-5.5", undefined), diff --git a/src/agent/runtime/default-provider-options.ts b/src/agent/runtime/default-provider-options.ts index 996d0d047f..a0d7f9dd3b 100644 --- a/src/agent/runtime/default-provider-options.ts +++ b/src/agent/runtime/default-provider-options.ts @@ -8,6 +8,7 @@ */ import { + isVeryfrontCloudAnthropicSurfaceModel, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudThinkingProviderOptions, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; @@ -15,11 +16,16 @@ import { const VERYFRONT_CLOUD_PREFIX = "veryfront-cloud/"; const ANTHROPIC_PREFIX = "anthropic/"; +/** + * Whether a model speaks the Anthropic protocol. A Veryfront Cloud model + * speaks the surface the served catalog gives its provider, so a newly served + * provider on the Anthropic surface gets the same defaults as `anthropic/*`. + */ function isAnthropicModel(modelString: string): boolean { - const normalized = modelString.startsWith(VERYFRONT_CLOUD_PREFIX) - ? modelString.slice(VERYFRONT_CLOUD_PREFIX.length) - : modelString; - return normalized.startsWith(ANTHROPIC_PREFIX); + if (modelString.startsWith(VERYFRONT_CLOUD_PREFIX)) { + return isVeryfrontCloudAnthropicSurfaceModel(modelString); + } + return modelString.startsWith(ANTHROPIC_PREFIX); } function hasAnthropicThinkingConfig(existing: Record | undefined): boolean { diff --git a/src/agent/runtime/model-resolution.test.ts b/src/agent/runtime/model-resolution.test.ts index e3cc6c7198..cda89d744e 100644 --- a/src/agent/runtime/model-resolution.test.ts +++ b/src/agent/runtime/model-resolution.test.ts @@ -7,7 +7,9 @@ import { } from "#veryfront/testing/assert.ts"; import { VeryfrontError } from "#veryfront/errors"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; -import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { resolveVeryfrontCloudModelId, VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES, @@ -45,6 +47,8 @@ function clearModelEnv(): void { } describe("agent/runtime/model-resolution", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { clearModelEnv(); }); diff --git a/src/agent/runtime/model-resolution.ts b/src/agent/runtime/model-resolution.ts index fbb1dbe190..9426bfcfee 100644 --- a/src/agent/runtime/model-resolution.ts +++ b/src/agent/runtime/model-resolution.ts @@ -5,9 +5,13 @@ import { getOpenAIEnvConfig, } from "#veryfront/config/env.ts"; import { + canVeryfrontCloudCatalogRefuse, createRetiredVeryfrontCloudModelError, - findVeryfrontCloudModelByModelId, + isListedInServedVeryfrontCloudCatalog, isRetiredVeryfrontCloudModelId, + isSupportedMistralModelId, + isVeryfrontCloudCatalogLoaded, + resolveServedVeryfrontCloudAlias, VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { DEFAULT_MODEL_CREDENTIAL_MISMATCH, NOT_SUPPORTED } from "#veryfront/errors"; @@ -104,7 +108,11 @@ export function resolveConfiguredAgentModel(model?: string): string { return normalized; } - return LEGACY_MODEL_ALIASES.get(normalized) ?? normalized; + // Known aliases first, so a bare vendor name keeps its direct-key meaning; + // then an alias only the loaded served catalog knows. + return LEGACY_MODEL_ALIASES.get(normalized) ?? + resolveServedVeryfrontCloudAlias(normalized) ?? + normalized; } /** Resolve the provider-options key used by the effective model runtime. */ @@ -147,12 +155,16 @@ function listAvailableDirectProviders(): string[] { } function isSupportedHostedMistralModel(modelId: string): boolean { - return Boolean(findVeryfrontCloudModelByModelId(`mistral/${modelId}`)); + // A stale served catalog cannot refuse: the platform answers for the model. + return !canVeryfrontCloudCatalogRefuse() || isSupportedMistralModelId(`mistral/${modelId}`); } function isUnsupportedVeryfrontCloudMistralModel(modelId: string): boolean { - return modelId.startsWith("veryfront-cloud/mistral/") && - !findVeryfrontCloudModelByModelId(modelId); + // An explicit Veryfront Cloud id is refused only against a served catalog: + // the shipped list cannot know a model the platform added since, and the + // model checks its own catalog once that has loaded. + return modelId.startsWith("veryfront-cloud/mistral/") && isVeryfrontCloudCatalogLoaded() && + canVeryfrontCloudCatalogRefuse() && !isSupportedMistralModelId(modelId); } function normalizeVeryfrontCloudRuntimeModel(modelId: string): string { @@ -249,7 +261,13 @@ export function resolveRuntimeModel(model?: string): string { const provider = configuredModel.slice(0, slashIndex); const modelId = configuredModel.slice(slashIndex + 1); - if (!HOSTED_PROVIDER_NAMES.has(provider) || !modelId) { + // A provider this package names, or any model the loaded served catalog + // lists (a provider the platform added since), is a Veryfront Cloud candidate. + if ( + !modelId || + (!HOSTED_PROVIDER_NAMES.has(provider) && + !isListedInServedVeryfrontCloudCatalog(configuredModel)) + ) { return configuredModel; } diff --git a/src/agent/runtime/model-transport.test.ts b/src/agent/runtime/model-transport.test.ts index e7914c6990..e71f86e254 100644 --- a/src/agent/runtime/model-transport.test.ts +++ b/src/agent/runtime/model-transport.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertStrictEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { ModelRuntime } from "#veryfront/provider"; import type { AgentConfig, ModelTransportRequest } from "../types.ts"; import { resolveAgentModelTransport } from "./model-transport.ts"; @@ -23,6 +25,8 @@ function createModel(modelId: string): ModelRuntime { } describe("resolveAgentModelTransport", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("resolves the configured runtime model when no host transport hook is present", async () => { const config: AgentConfig = { model: "local/qwen3.5-0.8b", diff --git a/src/agent/runtime/model-transport.ts b/src/agent/runtime/model-transport.ts index 3556c03e3c..72f1b33403 100644 --- a/src/agent/runtime/model-transport.ts +++ b/src/agent/runtime/model-transport.ts @@ -7,6 +7,9 @@ import { type AgentConfig, type RuntimeReasoningOption } from "../types.ts"; import { type ModelRuntime, resolveModel } from "#veryfront/provider"; import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; +import { warmVeryfrontCloudCatalog } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import { readVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; +import { isVeryfrontCloudEnabled } from "#veryfront/platform/cloud/resolver.ts"; import { resolveProviderOptionsWithDefaults } from "./default-provider-options.ts"; import { resolveConfiguredAgentModel, @@ -14,10 +17,10 @@ import { resolveRuntimeModel, } from "./model-resolution.ts"; import { + isVeryfrontCloudAnthropicSurfaceModel, resolveVeryfrontCloudModelThinking, resolveVeryfrontCloudReasoningOption, resolveVeryfrontCloudThinkingProviderOptions, - tryGetVeryfrontCloudProviderFromModelId, VERYFRONT_CLOUD_MODEL_PREFIX, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { hasDisabledThinking } from "./model-capabilities.ts"; @@ -198,7 +201,7 @@ function resolveReasoningWithDefaults( return { enabled: false }; } - if (tryGetVeryfrontCloudProviderFromModelId(modelString) === "anthropic") { + if (isVeryfrontCloudAnthropicSurfaceModel(modelString)) { return undefined; } @@ -209,10 +212,25 @@ function resolveReasoningWithDefaults( export async function resolveAgentModelTransport( input: ResolveAgentModelTransportInput, ): Promise { - const requestedModel = resolveConfiguredAgentModel(input.modelOverride || input.config.model); - const resolvedModelString = resolveRuntimeModel(input.modelOverride || input.config.model); - const privatelyResolvedModel = input.resolveModelRuntime && - IntrinsicReflectApply(StringStartsWith, resolvedModelString, [VERYFRONT_CLOUD_MODEL_PREFIX]) + const configuredModel = input.modelOverride || input.config.model; + const startsWithCloudPrefix = (model: string): boolean => + IntrinsicReflectApply(StringStartsWith, model, [VERYFRONT_CLOUD_MODEL_PREFIX]) as boolean; + // Every decision below reads the served catalog: which model an omitted or + // `auto` model means, whether an explicit provider model is served through + // Veryfront Cloud, and the thinking defaults. An ambient run that can reach + // Veryfront Cloud loads the catalog before any of them. A run with a private + // model resolver was prepared from the catalog as it stood then, and its call + // must keep what that preparation reserved, so it does not load it here. + if ( + !input.resolveModelRuntime && isVeryfrontCloudEnabled() && + !(typeof configuredModel === "string" && configuredModel.startsWith("local/")) + ) { + await warmVeryfrontCloudCatalog(); + } + const requestedModel = resolveConfiguredAgentModel(configuredModel); + const resolvedModelString = resolveRuntimeModel(configuredModel); + const usesVeryfrontCloud = startsWithCloudPrefix(resolvedModelString); + const privatelyResolvedModel = input.resolveModelRuntime && usesVeryfrontCloud ? input.resolveModelRuntime(resolvedModelString) : undefined; const transport = privatelyResolvedModel @@ -242,6 +260,16 @@ export async function resolveAgentModelTransport( : resolveProviderOptionsWithDefaults(resolvedModelString, transport?.providerOptions); const languageModel = privatelyResolvedModel ?? transport?.model ?? resolveModel(resolvedModelString); + // A Veryfront Cloud model settles its protocol and capabilities on its first + // async step. Settle it here, before anything reads them: the provider option + // key below and the runtime's tool-calling, structured-output and replay + // checks. A failure surfaces again when the model is called. + if ( + readVeryfrontCloudModelFacts(languageModel) !== undefined && + typeof languageModel.prepare === "function" + ) { + await Promise.resolve(languageModel.prepare()).catch(() => {}); + } const providerOptionKey = resolveModelProviderOptionKey(resolvedModelString, languageModel); return { diff --git a/src/eval/judges.ts b/src/eval/judges.ts index 8eb5dd0115..2754f81e44 100644 --- a/src/eval/judges.ts +++ b/src/eval/judges.ts @@ -1,5 +1,12 @@ import { resolveRuntimeModel } from "#veryfront/agent/runtime/model-resolution.ts"; -import { type ModelRuntime, resolveModel } from "#veryfront/provider"; +import { + loadVeryfrontCloudModelCatalog, + type ModelRuntime, + resolveModel, +} from "#veryfront/provider"; + +/** Longest a judge waits for the served catalog before resolving its model. */ +const CATALOG_MAX_WAIT_MS = 3_000; import { generateText } from "#veryfront/runtime/runtime-bridge.ts"; import { classifyEvalModelAccessDenial, isEvalModelAccessDeniedError } from "./model-access.ts"; @@ -83,8 +90,15 @@ function clampScore(score: number): number { return Math.max(0, Math.min(1, score)); } -function resolveJudgeModel(model: string | ModelRuntime | undefined): ModelRuntime { +async function resolveJudgeModel( + model: string | ModelRuntime | undefined, + signal?: AbortSignal, +): Promise { if (model && typeof model === "object") return model; + // Whether the judge model routes through Veryfront Cloud reads the served catalog. + // A cancelled evaluation stops waiting for it at once. + await loadVeryfrontCloudModelCatalog({ maxWaitMs: CATALOG_MAX_WAIT_MS, signal }); + signal?.throwIfAborted(); return resolveModel(resolveRuntimeModel(model ?? DEFAULT_JUDGE_MODEL)); } @@ -369,7 +383,7 @@ function createLlmRubricJudge( return async (input) => { try { - const model = resolveJudgeModel(options.model); + const model = await resolveJudgeModel(options.model, input.signal); const response = await generateText({ model, messages: [ @@ -434,7 +448,7 @@ function createLlmGroundednessJudge( return async (input) => { try { - const model = resolveJudgeModel(validatedOptions.model); + const model = await resolveJudgeModel(validatedOptions.model, input.signal); const response = await generateText({ model, messages: [{ diff --git a/src/platform/cloud/resolver.test.ts b/src/platform/cloud/resolver.test.ts index 010f035d54..985a18f3d7 100644 --- a/src/platform/cloud/resolver.test.ts +++ b/src/platform/cloud/resolver.test.ts @@ -13,6 +13,10 @@ import { createTestConfig, } from "#veryfront/config/runtime-config.ts"; import { runWithVeryfrontCloudContext } from "#veryfront/provider/veryfront-cloud/context.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { runWithProjectEnv } from "#veryfront/server/project-env"; import { runWithRequestContext } from "#veryfront/platform/adapters/fs/veryfront/request-context.ts"; import { __resetEnvLoaderForTests } from "#veryfront/utils/env-loader.ts"; @@ -116,6 +120,22 @@ describe("platform/cloud/resolver", () => { ); }); + it("defaults to the served default model once the catalog loads, unless overridden", () => { + assertEquals(getDefaultVeryfrontCloudModel(), "veryfront-cloud/mistral/mistral-small-2503"); + + __setVeryfrontCloudCatalogForTests({ + models: [], + defaultModelId: "anthropic/claude-sonnet-4-6", + }); + try { + assertEquals(getDefaultVeryfrontCloudModel(), "veryfront-cloud/anthropic/claude-sonnet-4-6"); + setEnv("VERYFRONT_DEFAULT_MODEL", "openai/gpt-5.2"); + assertEquals(getDefaultVeryfrontCloudModel(), "veryfront-cloud/openai/gpt-5.2"); + } finally { + __resetVeryfrontCloudCatalogForTests(); + } + }); + it("lets scoped cloud context override env bootstrap values", () => { setEnv("VERYFRONT_API_TOKEN", "vf_env_token"); setEnv("VERYFRONT_PROJECT_SLUG", "env-project"); diff --git a/src/platform/cloud/resolver.ts b/src/platform/cloud/resolver.ts index 18d03b91b4..c0acbc0a55 100644 --- a/src/platform/cloud/resolver.ts +++ b/src/platform/cloud/resolver.ts @@ -1,4 +1,8 @@ import { getRuntimeRequestContext } from "#veryfront/platform/runtime-request-context.ts"; +import { + peekVeryfrontCloudCatalog, + type VeryfrontCloudCatalogScopeKey, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { getHostEnv, getHostEnvExcludingEnvFile, @@ -92,6 +96,7 @@ function resolveHostCredentialApiBaseUrl(): string { DEFAULT_API_BASE_URL; } +/** Built-in default model, used until the served catalog names one. */ export const DEFAULT_VERYFRONT_CLOUD_MODEL = "veryfront-cloud/mistral/mistral-small-2503"; export const DEFAULT_VERYFRONT_CLOUD_EMBEDDING_MODEL = "veryfront-cloud/openai/text-embedding-3-small"; @@ -229,10 +234,27 @@ export function isVeryfrontCloudEnabled(): boolean { return Boolean(bootstrap.apiToken && hasProjectContext); } +/** + * Default Veryfront Cloud model: `VERYFRONT_DEFAULT_MODEL` when set, otherwise + * the default the served catalog names once it has loaded for the current + * credentials and project, otherwise the built-in default. + */ export function getDefaultVeryfrontCloudModel(): string { + const bootstrap = getVeryfrontCloudBootstrap(); + // A context without credentials may carry the key of the run's catalog. + const carriedKey = getCurrentVeryfrontCloudContext()?.catalogScopeKey; + const served = (bootstrap.apiToken + ? peekVeryfrontCloudCatalog({ + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + }) + : carriedKey + ? peekVeryfrontCloudCatalog(carriedKey as VeryfrontCloudCatalogScopeKey) + : peekVeryfrontCloudCatalog())?.defaultModelId; return normalizeCloudModelString( getHostEnv("VERYFRONT_DEFAULT_MODEL"), - DEFAULT_VERYFRONT_CLOUD_MODEL, + served?.includes("/") ? served : DEFAULT_VERYFRONT_CLOUD_MODEL, ); } diff --git a/src/provider/index.ts b/src/provider/index.ts index 28f2766ca3..0e7604bc7e 100644 --- a/src/provider/index.ts +++ b/src/provider/index.ts @@ -21,7 +21,11 @@ export { } from "./model-registry.ts"; export type { ModelProviderFactory, ModelProviderRegistrationDisposer } from "./model-registry.ts"; export type { ModelRuntime } from "./types.ts"; -export type { VeryfrontCloudProviderId } from "./veryfront-cloud/model-catalog.ts"; +export type { + VeryfrontCloudModelId, + VeryfrontCloudProviderId, + VeryfrontCloudRuntimeModelId, +} from "./veryfront-cloud/model-catalog.ts"; export { DEFAULT_VERYFRONT_CLOUD_MODEL_ID, findVeryfrontCloudModel, @@ -30,6 +34,7 @@ export { groupVeryfrontCloudModelsByProvider, normalizeVeryfrontCloudModelId, resolveHostedVeryfrontCloudModelId, + resolveVeryfrontCloudDefaultModelId, resolveVeryfrontCloudGatewayModelId, resolveVeryfrontCloudModelId, resolveVeryfrontCloudModelThinking, @@ -39,6 +44,7 @@ export { VERYFRONT_CLOUD_CHAT_MODELS, VERYFRONT_CLOUD_MODEL_PREFIX, } from "./veryfront-cloud/model-catalog.ts"; +export { loadVeryfrontCloudModelCatalog } from "./veryfront-cloud/shared.ts"; export type { VeryfrontCloudChatModel, VeryfrontCloudModelThinkingConfig, diff --git a/src/provider/veryfront-cloud/catalog-client.test-helpers.ts b/src/provider/veryfront-cloud/catalog-client.test-helpers.ts new file mode 100644 index 0000000000..71ccc87207 --- /dev/null +++ b/src/provider/veryfront-cloud/catalog-client.test-helpers.ts @@ -0,0 +1,326 @@ +/** + * Served model catalog fixtures for tests. + * + * `SERVED_MODEL_ROWS` copies the `/ai/models` rows the platform serves today + * for the models this package's shipped table lists, reduced to the fields + * the catalog client reads. `UNSERVED_TABLE_MODEL_ROWS` describes the models + * the shipped table still lists but the platform no longer serves, with the + * facts the table carries, so tests written against those IDs keep working. + * Models the gateway has retired are not listed here: they are refused. + */ +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "./catalog-client.ts"; + +/** Served rows for the models the shipped table lists. */ +export const SERVED_MODEL_ROWS = [ + { + "id": "claude-opus-4-8", + "modelId": "anthropic/claude-opus-4-8", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "opus", + "anthropic/claude-opus-4-8", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "adaptive", + }, + }, + { + "id": "claude-opus-4-6", + "modelId": "anthropic/claude-opus-4-6", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "anthropic/claude-opus-4-6", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "budget", + "reasoning_budget_tokens": 2048, + }, + }, + { + "id": "claude-sonnet-4-6", + "modelId": "anthropic/claude-sonnet-4-6", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "sonnet", + "anthropic/claude-sonnet-4-6", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "budget", + "reasoning_budget_tokens": 2048, + }, + }, + { + "id": "claude-haiku-4-5-20251001", + "modelId": "anthropic/claude-haiku-4-5-20251001", + "provider": "anthropic", + "surface": "anthropic", + "operations": [ + "messages", + ], + "aliases": [ + "haiku", + "anthropic/claude-haiku-4-5-20251001", + "claude-haiku-4-5", + "anthropic/claude-haiku-4-5", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "reasoning_mode": "budget", + "reasoning_budget_tokens": 1024, + }, + }, + { + "id": "gpt-5.5", + "modelId": "openai/gpt-5.5", + "provider": "openai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.5", + "gpt-5.5", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "transport": "chat-completions", + "chat_completions_reasoning_with_function_tools": false, + }, + }, + { + "id": "gpt-5.4-mini", + "modelId": "openai/gpt-5.4-mini", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.4-mini", + "gpt-5.4-mini", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gpt-5.4", + "modelId": "openai/gpt-5.4", + "provider": "openai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.4", + "gpt-5.4", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + "transport": "chat-completions", + "chat_completions_reasoning_with_function_tools": false, + }, + }, + { + "id": "gpt-5-nano", + "modelId": "openai/gpt-5-nano", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5-nano", + "gpt-5-nano", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gemini-3.5-flash", + "modelId": "google-ai-studio/gemini-3.5-flash", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-3.5-flash", + "gemini-3.5-flash", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + }, + }, + { + "id": "gemini-2.5-pro", + "modelId": "google-ai-studio/gemini-2.5-pro", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-2.5-pro", + "gemini-2.5-pro", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "gemini-2.5-flash", + "modelId": "google-ai-studio/gemini-2.5-flash", + "provider": "google", + "surface": "google", + "operations": [ + "generate-content", + "stream-generate-content", + "openai-responses", + "openai-chat-completions", + ], + "aliases": [ + "google-ai-studio/gemini-2.5-flash", + "gemini-2.5-flash", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + }, + }, + { + "id": "mistral-small-2503", + "modelId": "mistral/mistral-small-2503", + "provider": "mistral", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "mistral/mistral-small-2503", + "mistral-small-2503", + ], + "capabilities": { + "thinking": false, + "reasoning": false, + "chat_completions_consecutive_system_messages": true, + }, + }, + { + "id": "kimi-k2.6", + "modelId": "moonshotai/kimi-k2.6", + "provider": "moonshotai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "moonshotai/kimi-k2.6", + "kimi-k2.6", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, + { + "id": "kimi-k2.5", + "modelId": "moonshotai/kimi-k2.5", + "provider": "moonshotai", + "surface": "openai", + "operations": [ + "chat-completions", + ], + "aliases": [ + "moonshotai/kimi-k2.5", + "kimi-k2.5", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, +] as const; + +/** Rows for models the shipped table lists that the platform no longer serves. */ +export const UNSERVED_TABLE_MODEL_ROWS = [ + { + "id": "gpt-5.2", + "modelId": "openai/gpt-5.2", + "provider": "openai", + "surface": "openai", + "operations": [ + "responses", + "chat-completions", + ], + "aliases": [ + "openai/gpt-5.2", + ], + "capabilities": { + "thinking": true, + "reasoning": true, + }, + }, +] as const; + +/** Default model the platform serves. */ +export const SERVED_DEFAULT_MODEL_ID = "mistral/mistral-small-2503"; + +/** A `/ai/models` payload with every row above. */ +export function servedCatalogPayload(): Record { + return { + models: [...SERVED_MODEL_ROWS, ...UNSERVED_TABLE_MODEL_ROWS], + defaultModelId: SERVED_DEFAULT_MODEL_ID, + }; +} + +/** Serve {@link servedCatalogPayload} to every catalog read until reset. */ +export function seedServedCatalogForTests(): void { + __setVeryfrontCloudCatalogForTests(servedCatalogPayload()); +} + +/** + * Serve {@link servedCatalogPayload} until the returned handle is disposed: + * `using _catalog = useServedCatalogForTests();`. + */ +export function useServedCatalogForTests(): Disposable { + seedServedCatalogForTests(); + return { [Symbol.dispose]: __resetVeryfrontCloudCatalogForTests }; +} diff --git a/src/provider/veryfront-cloud/catalog-client.ts b/src/provider/veryfront-cloud/catalog-client.ts new file mode 100644 index 0000000000..c942eb0cc7 --- /dev/null +++ b/src/provider/veryfront-cloud/catalog-client.ts @@ -0,0 +1,520 @@ +/** + * Client for the model catalog Veryfront Cloud serves at `/ai/models`. + * + * Model facts (wire protocol, operations, thinking defaults and transport + * capabilities) come from the served catalog, not from a table shipped in this + * package. Loading is asynchronous and happens on the first async step of a + * model call; every synchronous reader uses {@link peekVeryfrontCloudCatalog} + * and degrades when nothing is loaded yet. + * + * - Entries are cached per API base URL, project and credential, because the + * served list is filtered by the project the credential or header selects. + * A synchronous read names the same scope, so one project never reads + * another project's list. + * - Concurrent loads for one key share a single request. A caller's abort + * signal only stops that caller waiting; it never cancels the shared request. + * - An entry is fresh for {@link VERYFRONT_CLOUD_CATALOG_TTL_MS}. A stale entry + * is returned at once while one refresh runs in the background, and it is + * kept when that refresh fails. + * - A failed load never throws. It is logged once and retried after + * {@link VERYFRONT_CLOUD_CATALOG_RETRY_MS}. + */ +import { createVeryfrontApiOriginBoundOutboundFetch } from "#veryfront/security/http/outbound-fetch.ts"; +import { logger } from "#veryfront/utils/logger/logger.ts"; + +/** How long a loaded catalog is used before it is refreshed. */ +export const VERYFRONT_CLOUD_CATALOG_TTL_MS = 5 * 60_000; +/** How long a failed load waits before the next attempt for the same key. */ +export const VERYFRONT_CLOUD_CATALOG_RETRY_MS = 30_000; +/** + * Most catalogs kept at once. Run-scoped credentials rotate, so each run can + * add a key; the least recently used entry goes first, and a load in flight is + * never evicted. + */ +export const VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES = 256; +/** Upper bound on one catalog request. */ +const VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS = 10_000; +/** Header naming the project a catalog request is scoped to. */ +const PROJECT_SLUG_HEADER = "x-veryfront-project-slug"; +/** Path of the served catalog, relative to the API base URL. */ +const VERYFRONT_CLOUD_CATALOG_PATH = "ai/models"; + +/** One model of the served catalog, reduced to the facts this package reads. */ +export interface VeryfrontCloudCatalogModel { + /** Short model id, for example `claude-sonnet-4-6`. */ + readonly id: string; + /** Provider-qualified model id, for example `anthropic/claude-sonnet-4-6`. */ + readonly modelId: string; + /** Canonical provider the model belongs to. */ + readonly provider: string; + /** Other ids that select this model. */ + readonly aliases: readonly string[]; + /** Wire protocol the model is served on. Absent when the platform declares none. */ + readonly surface?: string; + /** Operations of `surface` the model is served on. Absent on an older API. */ + readonly operations?: readonly string[]; + /** Whether the model takes thinking controls. */ + readonly thinking?: boolean; + /** Which reasoning control the model takes, for example `budget` or `adaptive`. */ + readonly reasoningMode?: string; + /** OpenAI wire API the model must use, when the platform constrains it. */ + readonly transport?: string; + /** Thinking budget to send when the caller sets none. */ + readonly reasoningBudgetTokens?: number; + /** Whether `chat-completions` accepts reasoning together with function tools. */ + readonly chatCompletionsReasoningWithFunctionTools?: boolean; + /** Whether `chat-completions` accepts adjacent system messages separately. */ + readonly chatCompletionsConsecutiveSystemMessages?: boolean; +} + +/** A loaded catalog. */ +export interface VeryfrontCloudCatalog { + readonly models: readonly VeryfrontCloudCatalogModel[]; + /** Provider-qualified id of the default model, when the platform names one. */ + readonly defaultModelId?: string; +} + +/** Credentials and project a catalog is loaded and read for: the same ones inference uses. */ +export interface VeryfrontCloudCatalogScope { + readonly apiBaseUrl: string; + readonly apiToken: string; + readonly projectSlug?: string; +} + +/** Options for one catalog load. */ +export interface VeryfrontCloudCatalogLoadOptions extends VeryfrontCloudCatalogScope { + /** Stops this caller waiting. The shared request keeps running for other callers. */ + readonly signal?: AbortSignal; + /** Longest this caller waits for a request in flight before it goes on without it. */ + readonly maxWaitMs?: number; + /** + * Wait for the refresh of a stale entry instead of answering with it at once, + * for a caller about to make a decision the catalog must be current for. The + * stale entry still answers when the refresh fails or the wait ends. + */ + readonly fresh?: boolean; +} + +interface CatalogEntry { + catalog: VeryfrontCloudCatalog; + fetchedAt: number; +} + +const entries = new Map(); +const inflight = new Map>(); +const failedAt = new Map(); +/** Key of the scope a synchronous read uses, while {@link withVeryfrontCloudCatalogScope} runs. */ +let activeKey: string | undefined; +let seeded: VeryfrontCloudCatalog | undefined; +/** Scopes whose current failure has been logged, so each scope logs once per outage. */ +const loggedFailures = new Set(); +let now: () => number = Date.now; +/** Bumped by a test reset, so a load that settles afterwards changes nothing. */ +let generation = 0; + +function isRecord(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +function optionalString(value: unknown): string | undefined { + return typeof value === "string" && value.length > 0 ? value : undefined; +} + +function optionalBoolean(value: unknown): boolean | undefined { + return typeof value === "boolean" ? value : undefined; +} + +function optionalPositiveInteger(value: unknown): number | undefined { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0 ? value : undefined; +} + +function stringList(value: unknown): string[] | undefined { + if (!Array.isArray(value)) return undefined; + return value.filter((item): item is string => typeof item === "string" && item.length > 0); +} + +function parseModel(value: unknown): VeryfrontCloudCatalogModel | undefined { + if (!isRecord(value)) return undefined; + const id = optionalString(value.id); + const modelId = optionalString(value.modelId); + const provider = optionalString(value.provider); + if (!id || !modelId || !provider) return undefined; + const capabilities = isRecord(value.capabilities) ? value.capabilities : {}; + const model: VeryfrontCloudCatalogModel = { + id, + modelId, + provider, + aliases: Object.freeze(stringList(value.aliases) ?? []), + surface: optionalString(value.surface), + operations: ((operations) => operations && Object.freeze(operations))( + stringList(value.operations), + ), + thinking: optionalBoolean(capabilities.thinking), + reasoningMode: optionalString(capabilities.reasoning_mode), + transport: optionalString(capabilities.transport), + reasoningBudgetTokens: optionalPositiveInteger(capabilities.reasoning_budget_tokens), + chatCompletionsReasoningWithFunctionTools: optionalBoolean( + capabilities.chat_completions_reasoning_with_function_tools, + ), + chatCompletionsConsecutiveSystemMessages: optionalBoolean( + capabilities.chat_completions_consecutive_system_messages, + ), + }; + return Object.freeze(model); +} + +/** + * Read a served `/ai/models` payload. Rows without an id, model id or provider + * are skipped, and a field an older API does not serve reads as absent. + * Returns undefined when the payload carries no model list at all. + */ +export function parseVeryfrontCloudCatalog(payload: unknown): VeryfrontCloudCatalog | undefined { + if (!isRecord(payload) || !Array.isArray(payload.models)) return undefined; + const models: VeryfrontCloudCatalogModel[] = []; + for (const row of payload.models) { + const model = parseModel(row); + if (model) models.push(model); + } + return Object.freeze({ + models: Object.freeze(models), + defaultModelId: optionalString(payload.defaultModelId), + }); +} + +/** + * Non-reversible fingerprint of a credential, so entries for different + * credentials never share a key and the key never holds the credential. + */ +/** + * Per-process salt for credential fingerprints, so a fingerprint means nothing + * outside this process and cannot be matched against a known credential. + */ +const FINGERPRINT_SALT = Array.from( + crypto.getRandomValues(new Uint8Array(16)), + (byte) => byte.toString(16).padStart(2, "0"), +).join(""); + +function credentialFingerprint(token: string): string { + let a = 0x811c9dc5; + let b = 0x01000193; + const salted = `${FINGERPRINT_SALT}:${token}`; + for (let index = 0; index < salted.length; index++) { + const code = salted.charCodeAt(index); + a = Math.imul(a ^ code, 0x01000193) >>> 0; + b = Math.imul(b ^ code, 0x85ebca6b) >>> 0; + } + return `${a.toString(16)}${b.toString(16)}`; +} + +/** Store an entry as the most recently used, evicting the least recently used over the cap. */ +function rememberEntry(key: string, entry: CatalogEntry): void { + entries.delete(key); + entries.set(key, entry); + evictOldest(entries); +} + +/** Mark an entry as just used, so eviction keeps it longest. */ +function touchEntry(key: string, entry: CatalogEntry): void { + entries.delete(key); + entries.set(key, entry); +} + +/** Drop the oldest keys over the cap, skipping any key with a load in flight. */ +function evictOldest(map: Map): void { + if (map.size <= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) return; + for (const key of [...map.keys()]) { + if (map.size <= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) return; + if (!inflight.has(key)) map.delete(key); + } +} + +/** Record a failed load, forgetting failures whose retry window has passed. */ +function rememberFailure(key: string, at: number): void { + for (const [failedKey, failedTime] of failedAt) { + if (at - failedTime >= VERYFRONT_CLOUD_CATALOG_RETRY_MS) failedAt.delete(failedKey); + } + failedAt.delete(key); + failedAt.set(key, at); + evictOldest(failedAt); +} + +/** + * A catalog scope key: the cache key for a scope. It carries the API base URL, + * the project and a salted credential fingerprint, never the credential, so a + * context that must not hold the credential can still name the catalog loaded + * for it. + */ +export type VeryfrontCloudCatalogScopeKey = string & { readonly __catalogScopeKey: true }; + +/** The non-secret scope key for a scope. */ +export function veryfrontCloudCatalogScopeKey( + scope: VeryfrontCloudCatalogScope, +): VeryfrontCloudCatalogScopeKey { + return cacheKey(scope) as VeryfrontCloudCatalogScopeKey; +} + +function keyOf(scope: VeryfrontCloudCatalogScope | VeryfrontCloudCatalogScopeKey): string { + return typeof scope === "string" ? scope : cacheKey(scope); +} + +function cacheKey(scope: VeryfrontCloudCatalogScope): string { + return `${scope.apiBaseUrl}\n${scope.projectSlug ?? ""}\n${ + credentialFingerprint(scope.apiToken) + }`; +} + +/** + * The catalog URL under an API base URL. Like the gateway URLs, it keeps the + * base URL's query, which a setup can use to scope or sign requests. + */ +function catalogUrl(apiBaseUrl: string): string { + const url = new URL(apiBaseUrl); + url.pathname = `${url.pathname.replace(/\/+$/, "")}/${VERYFRONT_CLOUD_CATALOG_PATH}`; + url.hash = ""; + return url.toString(); +} + +/** An API base URL without its query or fragment, which can carry signed values. */ +function loggableBaseUrl(apiBaseUrl: string): string { + try { + const url = new URL(apiBaseUrl); + return `${url.origin}${url.pathname}`; + } catch { + return "[invalid URL]"; + } +} + +/** Strip the query and fragment from every URL an error message quotes. */ +function loggableErrorMessage(error: unknown): string { + const message = error instanceof Error ? error.message : String(error); + return message.replace(/(https?:\/\/[^\s?#"'<>)]*)[?#][^\s"'<>)]*/g, "$1"); +} + +async function fetchCatalog( + options: VeryfrontCloudCatalogScope, +): Promise { + const headers = new Headers({ + Accept: "application/json", + Authorization: `Bearer ${options.apiToken}`, + }); + if (options.projectSlug) headers.set(PROJECT_SLUG_HEADER, options.projectSlug); + // Only the internal timeout bounds the shared request: one caller giving up + // must not fail the load for every other caller on the same key. + const timeout = new AbortController(); + const timer = setTimeout(() => timeout.abort(), VERYFRONT_CLOUD_CATALOG_TIMEOUT_MS); + const signal = timeout.signal; + try { + const response = await createVeryfrontApiOriginBoundOutboundFetch(options.apiBaseUrl)( + catalogUrl(options.apiBaseUrl), + { method: "GET", headers, signal }, + ); + if (!response.ok) { + await response.body?.cancel(); + throw new Error( + `Veryfront Cloud model catalog request failed with status ${response.status}`, + ); + } + const catalog = parseVeryfrontCloudCatalog(await response.json()); + if (!catalog) throw new Error("Veryfront Cloud model catalog response has no model list"); + return catalog; + } finally { + clearTimeout(timer); + } +} + +function refresh( + key: string, + options: VeryfrontCloudCatalogScope, +): Promise { + const pending = inflight.get(key); + if (pending) return pending; + const started = generation; + const request = fetchCatalog(options).then( + (catalog) => { + if (started !== generation) return catalog; + rememberEntry(key, { catalog, fetchedAt: now() }); + failedAt.delete(key); + loggedFailures.delete(key); + return catalog; + }, + (error: unknown) => { + if (started !== generation) return undefined; + rememberFailure(key, now()); + const stale = entries.get(key)?.catalog; + if (!loggedFailures.has(key)) { + if (loggedFailures.size >= VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES) loggedFailures.clear(); + loggedFailures.add(key); + // Names the scope, never the credential or a signed query value. + logger.warn( + stale + ? "Veryfront Cloud model catalog refresh failed; the last loaded catalog stays in use" + : "Veryfront Cloud model catalog is unavailable; model facts fall back to the built-in list", + { + apiBaseUrl: loggableBaseUrl(options.apiBaseUrl), + projectSlug: options.projectSlug, + error: loggableErrorMessage(error), + }, + ); + } + return stale; + }, + ).finally(() => { + if (started !== generation) return; + inflight.delete(key); + // Eviction skips keys with a load in flight; retry now that this one has + // settled, so a burst of concurrent scopes cannot leave the maps over the cap. + evictOldest(entries); + evictOldest(failedAt); + }); + inflight.set(key, request); + return request; +} + +/** Resolve with what `request` resolves to, or with `fallback` once this caller stops waiting. */ +function waitFor( + request: Promise, + fallback: VeryfrontCloudCatalog | undefined, + signal: AbortSignal | undefined, + maxWaitMs: number | undefined, +): Promise { + if (!signal && maxWaitMs === undefined) return request; + if (signal?.aborted) return Promise.resolve(fallback); + return new Promise((resolve) => { + let timer: ReturnType | undefined; + const finish = (value: VeryfrontCloudCatalog | undefined) => { + if (timer !== undefined) clearTimeout(timer); + signal?.removeEventListener("abort", onAbort); + resolve(value); + }; + const onAbort = () => finish(fallback); + signal?.addEventListener("abort", onAbort, { once: true }); + if (maxWaitMs !== undefined) timer = setTimeout(() => finish(fallback), maxWaitMs); + request.then(finish); + }); +} + +/** + * Load the served catalog for a scope. Resolves to the cached catalog when it + * is fresh, to a stale one while a refresh runs, and to undefined when no + * catalog could be loaded or the caller stopped waiting first. Never rejects. + */ +export function loadVeryfrontCloudCatalog( + options: VeryfrontCloudCatalogLoadOptions, +): Promise { + if (seeded) return Promise.resolve(seeded); + const key = cacheKey(options); + const entry = entries.get(key); + const current = now(); + if (entry && current - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS) { + touchEntry(key, entry); + return Promise.resolve(entry.catalog); + } + const lastFailure = failedAt.get(key); + if (lastFailure !== undefined && current - lastFailure < VERYFRONT_CLOUD_CATALOG_RETRY_MS) { + return Promise.resolve(entry?.catalog); + } + if (lastFailure !== undefined) failedAt.delete(key); + const request = refresh(key, { + apiBaseUrl: options.apiBaseUrl, + apiToken: options.apiToken, + ...(options.projectSlug ? { projectSlug: options.projectSlug } : {}), + }); + // Stale while revalidate: the stale entry answers now, the refresh replaces it. + if (entry && !options.fresh) return Promise.resolve(entry.catalog); + return waitFor(request, entry?.catalog, options.signal, options.maxWaitMs).then((catalog) => + catalog ?? entry?.catalog + ); +} + +/** + * Whether the catalog for a scope is fresh: loaded within the TTL. Without a + * scope, reads the one {@link withVeryfrontCloudCatalogScope} names. A stale + * catalog may miss models the platform has enabled since, so it must not be + * used to refuse one. + */ +export function isVeryfrontCloudCatalogFresh( + scope?: VeryfrontCloudCatalogScope | VeryfrontCloudCatalogScopeKey, +): boolean { + if (seeded) return true; + const key = scope ? keyOf(scope) : activeKey; + if (key === undefined) return false; + const entry = entries.get(key); + return entry !== undefined && now() - entry.fetchedAt < VERYFRONT_CLOUD_CATALOG_TTL_MS; +} + +/** + * Run `fn` synchronously with {@link peekVeryfrontCloudCatalog} reading the + * catalog loaded for `scope`, so a model's facts come from its own project and + * credential whatever the ambient request carries. + */ +export function withVeryfrontCloudCatalogScope( + scope: VeryfrontCloudCatalogScope, + fn: () => T, +): T { + const previous = activeKey; + activeKey = cacheKey(scope); + try { + return fn(); + } finally { + activeKey = previous; + } +} + +/** Whether {@link withVeryfrontCloudCatalogScope} names the scope reads use right now. */ +export function hasActiveVeryfrontCloudCatalogScope(): boolean { + return activeKey !== undefined; +} + +/** + * The catalog loaded for a scope, stale or not, or undefined before any load + * for it. Without a scope, reads the one {@link withVeryfrontCloudCatalogScope} + * names, and undefined outside it. + */ +export function peekVeryfrontCloudCatalog( + scope?: VeryfrontCloudCatalogScope | VeryfrontCloudCatalogScopeKey, +): VeryfrontCloudCatalog | undefined { + if (seeded) return seeded; + const key = scope ? keyOf(scope) : activeKey; + if (key === undefined) return undefined; + const entry = entries.get(key); + if (!entry) return undefined; + touchEntry(key, entry); + return entry.catalog; +} + +/** @internal Serve a fixed catalog for every key, as if freshly loaded. `undefined` clears it. */ +export function __setVeryfrontCloudCatalogForTests(payload: unknown): void { + seeded = payload === undefined ? undefined : parseVeryfrontCloudCatalog(payload); +} + +/** @internal Store a catalog for one scope, as if it had just loaded for it. */ +export function __setVeryfrontCloudCatalogForScopeForTests( + scope: VeryfrontCloudCatalogScope, + payload: unknown, +): void { + const catalog = parseVeryfrontCloudCatalog(payload); + if (!catalog) throw new TypeError("Test catalog payload has no model list"); + rememberEntry(cacheKey(scope), { catalog, fetchedAt: now() }); +} + +/** @internal How many catalogs and recorded failures the cache holds. */ +export function __veryfrontCloudCatalogSizesForTests(): { entries: number; failures: number } { + return { entries: entries.size, failures: failedAt.size }; +} + +/** @internal Forget every loaded catalog, pending load and failure. */ +export function __resetVeryfrontCloudCatalogForTests(): void { + generation++; + entries.clear(); + inflight.clear(); + failedAt.clear(); + activeKey = undefined; + seeded = undefined; + loggedFailures.clear(); + now = Date.now; +} + +/** @internal Replace the clock the cache reads. */ +export function __setVeryfrontCloudCatalogClockForTests(clock: () => number): void { + now = clock; +} diff --git a/src/provider/veryfront-cloud/context.ts b/src/provider/veryfront-cloud/context.ts index 5f8a95ff0e..2a32842de6 100644 --- a/src/provider/veryfront-cloud/context.ts +++ b/src/provider/veryfront-cloud/context.ts @@ -14,6 +14,12 @@ export interface VeryfrontCloudContext { billingGroupRequestAdmitted?: boolean; projectSlug?: string; serviceLayer?: string; + /** + * Names the model catalog loaded for credentials this context does not hold, + * so synchronous model reads in a credential-free context use the same + * catalog as the run. It carries no credential. + */ + catalogScopeKey?: string; } const veryfrontCloudContextStorage = new AsyncLocalStorage(); diff --git a/src/provider/veryfront-cloud/gateway-routing.test.ts b/src/provider/veryfront-cloud/gateway-routing.test.ts index 819bba1879..3d155203fc 100644 --- a/src/provider/veryfront-cloud/gateway-routing.test.ts +++ b/src/provider/veryfront-cloud/gateway-routing.test.ts @@ -7,7 +7,9 @@ */ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { resolveGenAiProviderName } from "#veryfront/agent/hosted/trace-attributes.ts"; import { getProviderToolProfile } from "#veryfront/agent/runtime/provider-tool-compat.ts"; import { @@ -48,6 +50,8 @@ function routingRow(modelId: string): RoutingRow { } describe("provider/veryfront-cloud gateway routing", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("keeps the routing facts of every catalog model", () => { assertEquals(VERYFRONT_CLOUD_CHAT_MODELS.map((model) => routingRow(model.modelId)), [ { diff --git a/src/provider/veryfront-cloud/model-catalog.deprecated.test.ts b/src/provider/veryfront-cloud/model-catalog.deprecated.test.ts new file mode 100644 index 0000000000..3d506184e9 --- /dev/null +++ b/src/provider/veryfront-cloud/model-catalog.deprecated.test.ts @@ -0,0 +1,61 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import * as barrel from "#veryfront/provider"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForTests, +} from "./catalog-client.ts"; +import * as catalog from "./model-catalog.ts"; +import * as shim from "./model-catalog.deprecated.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, +} from "./model-catalog.data.ts"; + +describe("provider/veryfront-cloud/model-catalog deprecated exports", () => { + afterEach(__resetVeryfrontCloudCatalogForTests); + + it("keeps the shipped model list whatever the served catalog says", () => { + __setVeryfrontCloudCatalogForTests({ models: [], defaultModelId: "acme/acme-1" }); + + assertEquals( + shim.VERYFRONT_CLOUD_CHAT_MODELS.map((model) => model.modelId), + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => model.modelId), + ); + assertEquals(shim.DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL.id, TABLE_DEFAULT_MODEL_ID); + assertEquals(shim.findVeryfrontCloudModel("opus")?.modelId, "anthropic/claude-opus-4-8"); + assertEquals( + shim.findVeryfrontCloudModelByModelId("veryfront-cloud/google/gemini-2.5-pro")?.id, + "gemini-2.5-pro", + ); + assertEquals( + shim.groupVeryfrontCloudModelsByProvider().map((group) => group.provider), + ["anthropic", "openai", "google", "mistral", "moonshotai"], + ); + }); + + it("is what the catalog module and the public barrel export under the same names", () => { + for ( + const name of [ + "VERYFRONT_CLOUD_CHAT_MODELS", + "DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL", + "findVeryfrontCloudModel", + "findVeryfrontCloudModelByModelId", + "groupVeryfrontCloudModelsByProvider", + ] as const + ) { + assertEquals(catalog[name], shim[name], name); + } + for ( + const name of [ + "VERYFRONT_CLOUD_CHAT_MODELS", + "findVeryfrontCloudModel", + "findVeryfrontCloudModelByModelId", + "groupVeryfrontCloudModelsByProvider", + ] as const + ) { + assertEquals(barrel[name], shim[name], name); + } + }); +}); diff --git a/src/provider/veryfront-cloud/model-catalog.deprecated.ts b/src/provider/veryfront-cloud/model-catalog.deprecated.ts new file mode 100644 index 0000000000..38bd479277 --- /dev/null +++ b/src/provider/veryfront-cloud/model-catalog.deprecated.ts @@ -0,0 +1,221 @@ +/** + * Deprecated model list exports, backed by the table shipped in this package. + * + * Model facts now come from the served catalog (`catalog-client.ts`). This + * module keeps the public exports that only make sense with a shipped list + * working for one release, and it is the only module that imports the shipped + * table. While it ships, the resolvers also read {@link SHIPPED_VERYFRONT_CLOUD_CATALOG} + * for a scope whose catalog has not loaded, so a process that has not reached + * `/ai/models` keeps the behaviour of the previous release. + */ +import type { KnownVeryfrontCloudProviderId, VeryfrontCloudChatModel } from "./model-catalog.ts"; +import type { VeryfrontCloudCatalog, VeryfrontCloudCatalogModel } from "./catalog-client.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, + VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, + VERYFRONT_CLOUD_PROVIDER_ALIASES, + VERYFRONT_CLOUD_PROVIDER_LABELS as PROVIDER_LABELS, + VERYFRONT_CLOUD_PROVIDER_ORDER as PROVIDER_ORDER, + VERYFRONT_CLOUD_PROVIDER_ROUTING, +} from "./model-catalog.data.ts"; + +const MODEL_PREFIX = "veryfront-cloud/"; + +/** + * Every provider name the shipped catalog data routes: accepted aliases, + * providers with a routing row, and providers of a listed chat model. Runtime + * model resolution sends `/` for these through the gateway + * when no direct provider credential applies; a provider only the served + * catalog lists is checked against that catalog instead. + */ +export const VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES: readonly string[] = Object.freeze([ + ...new Set([ + ...VERYFRONT_CLOUD_PROVIDER_ALIASES.map(([alias]) => alias), + ...VERYFRONT_CLOUD_PROVIDER_ROUTING.map(([provider]) => provider), + ...VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => model.provider), + ]), +]); +const providerAliases: ReadonlyMap = new Map(VERYFRONT_CLOUD_PROVIDER_ALIASES); + +function isPositiveSafeInteger(value: unknown): value is number { + return typeof value === "number" && Number.isSafeInteger(value) && value > 0; +} + +/** `/` by the shipped alias list, prefix removed. */ +function tableModelKey(modelId: string): string { + const normalized = modelId.startsWith(MODEL_PREFIX) + ? modelId.slice(MODEL_PREFIX.length) + : modelId; + const slashIndex = normalized.indexOf("/"); + if (slashIndex <= 0) return normalized; + const provider = normalized.slice(0, slashIndex); + return `${providerAliases.get(provider) ?? provider}/${normalized.slice(slashIndex + 1)}`; +} + +/** + * Chat models shipped with this package. + * + * @deprecated Read the served catalog instead. This list is removed in a later release. + */ +export const VERYFRONT_CLOUD_CHAT_MODELS: readonly VeryfrontCloudChatModel[] = Object.freeze( + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => { + if ( + model.thinkingBudgetTokens !== undefined && + !isPositiveSafeInteger(model.thinkingBudgetTokens) + ) { + throw new TypeError( + `Veryfront Cloud model "${model.id}" thinkingBudgetTokens must be a positive safe integer`, + ); + } + return Object.freeze(model); + }), +); + +const defaultVeryfrontCloudChatModel = VERYFRONT_CLOUD_CHAT_MODELS.find( + (model) => model.id === TABLE_DEFAULT_MODEL_ID, +); +if (!defaultVeryfrontCloudChatModel) { + throw new Error( + `Veryfront Cloud default model "${TABLE_DEFAULT_MODEL_ID}" is missing from the catalog`, + ); +} + +/** + * Shipped descriptor of the built-in default model. + * + * @deprecated Read the served catalog instead. This descriptor is removed in a later release. + */ +export const DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL = defaultVeryfrontCloudChatModel; + +/** + * Find a shipped chat model by its short id. + * + * @deprecated Read the served catalog instead. This lookup is removed in a later release. + */ +export function findVeryfrontCloudModel( + id: string, +): VeryfrontCloudChatModel | undefined { + return VERYFRONT_CLOUD_CHAT_MODELS.find((model) => model.id === id); +} + +/** + * Find a shipped chat model by its provider-qualified id, in any provider spelling. + * + * @deprecated Read the served catalog instead. This lookup is removed in a later release. + */ +export function findVeryfrontCloudModelByModelId( + modelId: string, +): VeryfrontCloudChatModel | undefined { + const key = tableModelKey(modelId); + return VERYFRONT_CLOUD_CHAT_MODELS.find((model) => tableModelKey(model.modelId) === key); +} + +/** + * Group the shipped chat models by provider, in display order. + * + * @deprecated Read the served catalog instead. This grouping is removed in a later release. + */ +export function groupVeryfrontCloudModelsByProvider(): Array<{ + readonly provider: KnownVeryfrontCloudProviderId; + readonly label: string; + readonly models: readonly VeryfrontCloudChatModel[]; +}> { + return PROVIDER_ORDER.map((provider) => ({ + provider, + label: PROVIDER_LABELS[provider], + models: Object.freeze( + VERYFRONT_CLOUD_CHAT_MODELS.filter((model) => model.provider === provider), + ), + })).filter((group) => group.models.length > 0); +} + +const providerRouting = new Map(VERYFRONT_CLOUD_PROVIDER_ROUTING); +const transportCapabilities = new Map(VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES); + +/** The operations the shipped table implies for a model, in the served catalog's terms. */ +function shippedOperations(provider: string, key: string): readonly string[] | undefined { + const routing = providerRouting.get(provider); + switch (routing?.surface) { + case "anthropic": + return ["messages"]; + case "google": + return ["generate-content", "stream-generate-content"]; + case "openai": + return routing.native === true && + transportCapabilities.get(key)?.openAITransport !== "chat-completions" + ? ["responses", "chat-completions"] + : ["chat-completions"]; + default: + return undefined; + } +} + +function shippedModel( + id: string, + modelId: string, + provider: string, + thinking: boolean | undefined, + budget: number | undefined, +): VeryfrontCloudCatalogModel { + const key = tableModelKey(modelId); + const capabilities = transportCapabilities.get(key); + const surface = providerRouting.get(provider)?.surface; + const operations = shippedOperations(provider, key); + return Object.freeze({ + id, + modelId, + provider, + aliases: Object.freeze([]), + ...(surface === undefined ? {} : { surface }), + ...(operations === undefined ? {} : { operations: Object.freeze([...operations]) }), + ...(thinking === undefined ? {} : { thinking }), + ...(capabilities?.anthropicThinkingMode + ? { reasoningMode: capabilities.anthropicThinkingMode } + : {}), + ...(capabilities?.openAITransport ? { transport: capabilities.openAITransport } : {}), + ...(budget === undefined ? {} : { reasoningBudgetTokens: budget }), + ...(capabilities?.openAIChatReasoningWithFunctionTools === undefined ? {} : { + chatCompletionsReasoningWithFunctionTools: capabilities.openAIChatReasoningWithFunctionTools, + }), + ...(capabilities?.openAIChatPreserveSystemMessages === undefined ? {} : { + chatCompletionsConsecutiveSystemMessages: capabilities.openAIChatPreserveSystemMessages, + }), + }); +} + +/** + * The shipped table in the served catalog's shape. The resolvers read it for a + * scope whose catalog has not loaded. A later release removes it with the table. + * + * @deprecated Internal fallback that is removed with the shipped table. + */ +export const SHIPPED_VERYFRONT_CLOUD_CATALOG: VeryfrontCloudCatalog = Object.freeze({ + models: Object.freeze([ + ...VERYFRONT_CLOUD_CHAT_MODELS.map((model) => + shippedModel( + model.id, + model.modelId, + model.provider, + model.thinking === true || model.thinkingBudgetTokens !== undefined ? true : undefined, + model.thinkingBudgetTokens, + ) + ), + // Transport rows for models without a chat entry keep their facts too. + ...VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES + .filter(([key]) => + !VERYFRONT_CLOUD_CHAT_MODELS.some((model) => tableModelKey(model.modelId) === key) + ) + .map(([key]) => { + const slashIndex = key.indexOf("/"); + return shippedModel( + key.slice(slashIndex + 1), + key, + key.slice(0, slashIndex), + undefined, + undefined, + ); + }), + ]), + defaultModelId: DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL.modelId, +}); diff --git a/src/provider/veryfront-cloud/model-catalog.served.test.ts b/src/provider/veryfront-cloud/model-catalog.served.test.ts new file mode 100644 index 0000000000..b23d2698e9 --- /dev/null +++ b/src/provider/veryfront-cloud/model-catalog.served.test.ts @@ -0,0 +1,462 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals, assertThrows } from "#veryfront/testing/assert.ts"; +import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogForScopeForTests, + __setVeryfrontCloudCatalogForTests, + peekVeryfrontCloudCatalog, + withVeryfrontCloudCatalogScope, +} from "./catalog-client.ts"; +import { + seedServedCatalogForTests, + SERVED_MODEL_ROWS, + servedCatalogPayload, + UNSERVED_TABLE_MODEL_ROWS, +} from "./catalog-client.test-helpers.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, + isRetiredVeryfrontCloudModelId, + isSupportedMistralModelId, + resolveVeryfrontCloudDefaultModelId, + resolveVeryfrontCloudModelId, + resolveVeryfrontCloudModelThinking, + resolveVeryfrontCloudOpenAIChatFunctionToolReasoning, + resolveVeryfrontCloudOpenAIChatSystemMessages, + resolveVeryfrontCloudOpenAITransport, + resolveVeryfrontCloudOpenAITransportPlan, + resolveVeryfrontCloudProviderId, + resolveVeryfrontCloudProviderRouting, + resolveVeryfrontCloudReasoningOption, + resolveVeryfrontCloudThinkingProviderOptions, +} from "./model-catalog.ts"; +import { + DEFAULT_VERYFRONT_CLOUD_MODEL_ID as TABLE_DEFAULT_MODEL_ID, + VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, + VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, + VERYFRONT_CLOUD_PROVIDER_ALIASES, + VERYFRONT_CLOUD_PROVIDER_ROUTING, +} from "./model-catalog.data.ts"; +import { isOpenAIReasoningModel } from "../shared/openai-reasoning.ts"; + +type ServedRow = { + id: string; + modelId: string; + provider: string; + surface?: string; + operations?: readonly string[]; + aliases: readonly string[]; + capabilities: Record; +}; + +/** A served payload of the given rows, each with the fields a row always carries. */ +function payload(rows: readonly ServedRow[], defaultModelId?: string): Record { + return { models: rows, ...(defaultModelId ? { defaultModelId } : {}) }; +} + +function row(modelId: string, fields: Partial = {}): ServedRow { + const [provider = "", id = ""] = modelId.split("/"); + return { id, modelId, provider, aliases: [], capabilities: {}, ...fields }; +} + +describe("provider/veryfront-cloud/model-catalog served facts", () => { + afterEach(__resetVeryfrontCloudCatalogForTests); + + describe("before the catalog is loaded", () => { + it("routes on the shipped list: protocol providers natively, Mistral on the OpenAI protocol", () => { + assertEquals(resolveVeryfrontCloudProviderRouting("openai"), { + surface: "openai", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("anthropic"), { + surface: "anthropic", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("google"), { + surface: "google", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("mistral").surface, "openai"); + assertEquals(resolveVeryfrontCloudProviderRouting("mistral").native, false); + assertEquals(resolveVeryfrontCloudProviderRouting("acme-labs"), { surface: "openai" }); + }); + + it("routes google-ai-studio as Google", () => { + assertEquals(resolveVeryfrontCloudProviderId("google-ai-studio"), "google"); + assertEquals(resolveVeryfrontCloudProviderRouting("google-ai-studio"), { + surface: "google", + native: true, + }); + }); + + it("reads the facts shipped with this package, so behaviour matches the previous release", () => { + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), { + enabled: true, + budgetTokens: 2048, + }); + assertEquals(resolveVeryfrontCloudOpenAITransport("openai/gpt-5.5"), "chat-completions"); + assertEquals( + resolveVeryfrontCloudOpenAIChatSystemMessages("mistral/mistral-small-2503"), + true, + ); + assertEquals(resolveVeryfrontCloudModelId("opus"), "anthropic/claude-opus-4-8"); + assertEquals( + resolveVeryfrontCloudDefaultModelId(), + DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, + ); + assertEquals(resolveVeryfrontCloudModelId(), DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID); + assertEquals(isSupportedMistralModelId("mistral/mistral-small-2503"), true); + assertEquals(isSupportedMistralModelId("mistral/not-listed"), false); + }); + }); + + describe("scope", () => { + const projectA = { + apiBaseUrl: "https://api.example.test", + apiToken: "token-a", + projectSlug: "a", + }; + const projectB = { + apiBaseUrl: "https://api.example.test", + apiToken: "token-b", + projectSlug: "b", + }; + const sameProjectOtherToken = { ...projectA, apiToken: "token-c" }; + + it("reads each project's own catalog, never one another project loaded", () => { + __setVeryfrontCloudCatalogForScopeForTests( + projectA, + payload([ + row("openai/gpt-a", { surface: "openai", operations: ["chat-completions"] }), + ], "openai/gpt-a"), + ); + __setVeryfrontCloudCatalogForScopeForTests( + projectB, + payload([ + row("mistral/mistral-b", { surface: "openai", operations: ["chat-completions"] }), + ], "mistral/mistral-b"), + ); + + withVeryfrontCloudCatalogScope(projectA, () => { + assertEquals(isSupportedMistralModelId("mistral/mistral-b"), false); + assertEquals(resolveVeryfrontCloudDefaultModelId(), "openai/gpt-a"); + assertEquals(resolveVeryfrontCloudProviderRouting("openai").native, false); + }); + withVeryfrontCloudCatalogScope(projectB, () => { + assertEquals(isSupportedMistralModelId("mistral/mistral-b"), true); + assertEquals(resolveVeryfrontCloudDefaultModelId(), "mistral/mistral-b"); + }); + }); + + it("keeps a credential's catalog apart from another credential's for the same project", () => { + __setVeryfrontCloudCatalogForScopeForTests(projectA, payload([], "openai/gpt-a")); + + assertEquals(peekVeryfrontCloudCatalog(projectA)?.defaultModelId, "openai/gpt-a"); + assertEquals(peekVeryfrontCloudCatalog(sameProjectOtherToken), undefined); + withVeryfrontCloudCatalogScope(sameProjectOtherToken, () => { + // Nothing loaded for this credential: the shipped list applies. + assertEquals( + resolveVeryfrontCloudDefaultModelId(), + DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, + ); + assertEquals(resolveVeryfrontCloudModelId("opus"), "anthropic/claude-opus-4-8"); + }); + }); + + it("keeps google-ai-studio as Google when a loaded catalog lists no Google model", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-x", { surface: "openai", operations: ["chat-completions"] }), + ])); + + assertEquals(resolveVeryfrontCloudProviderId("google-ai-studio"), "google"); + assertEquals(resolveVeryfrontCloudProviderRouting("google-ai-studio").surface, "google"); + }); + }); + + describe("once the catalog is loaded", () => { + it("reads native providers from the operations each model is served on", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-chat-only", { surface: "openai", operations: ["chat-completions"] }), + row("acme/acme-1", { surface: "openai", operations: ["responses", "chat-completions"] }), + row("anthropic/claude-x", { surface: "anthropic", operations: ["messages"] }), + ])); + + assertEquals(resolveVeryfrontCloudProviderRouting("openai"), { + surface: "openai", + native: false, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("acme"), { + surface: "openai", + native: true, + }); + assertEquals(resolveVeryfrontCloudProviderRouting("anthropic"), { + surface: "anthropic", + native: true, + }); + }); + + it("keeps a model the platform does not serve on Responses on chat completions", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-5.9", { + surface: "openai", + operations: ["responses", "chat-completions"], + capabilities: { thinking: true }, + }), + row("openai/gpt-5.9-chat", { + surface: "openai", + operations: ["chat-completions"], + capabilities: { thinking: true }, + }), + ])); + + assertEquals(resolveVeryfrontCloudOpenAITransportPlan("openai", "gpt-5.9"), { + transport: "responses", + pinned: true, + }); + assertEquals(resolveVeryfrontCloudOpenAITransportPlan("openai", "gpt-5.9-chat"), { + transport: "chat-completions", + pinned: true, + }); + }); + + it("reads the thinking budget from reasoning_budget_tokens", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("anthropic/claude-sonnet-4-6", { + surface: "anthropic", + aliases: ["sonnet"], + capabilities: { thinking: true, reasoning_mode: "budget", reasoning_budget_tokens: 4096 }, + }), + row("openai/gpt-think", { surface: "openai", capabilities: { thinking: true } }), + row("openai/gpt-plain", { surface: "openai", capabilities: { thinking: false } }), + ])); + + assertEquals(resolveVeryfrontCloudModelThinking("anthropic/claude-sonnet-4-6"), { + enabled: true, + budgetTokens: 4096, + }); + assertEquals(resolveVeryfrontCloudModelThinking("sonnet"), { + enabled: true, + budgetTokens: 4096, + }); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-think"), { enabled: true }); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-plain"), undefined); + }); + + it("reads both chat completions flags from their capability fields", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-5.4", { + surface: "openai", + capabilities: { + transport: "chat-completions", + chat_completions_reasoning_with_function_tools: true, + }, + }), + row("mistral/mistral-small-2503", { + surface: "openai", + capabilities: { chat_completions_consecutive_system_messages: false }, + }), + ])); + + assertEquals(resolveVeryfrontCloudOpenAITransport("openai/gpt-5.4"), "chat-completions"); + assertEquals(resolveVeryfrontCloudOpenAIChatFunctionToolReasoning("openai/gpt-5.4"), true); + assertEquals( + resolveVeryfrontCloudOpenAIChatSystemMessages("mistral/mistral-small-2503"), + false, + ); + }); + + it("resolves a provider alias from the served provider field", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("vendor-studio/vendor-model", { provider: "vendor", surface: "google" }), + ])); + + assertEquals(resolveVeryfrontCloudProviderId("vendor-studio"), "vendor"); + assertEquals(resolveVeryfrontCloudProviderRouting("vendor-studio").surface, "google"); + }); + + it("uses the default model the catalog names", () => { + __setVeryfrontCloudCatalogForTests( + payload([row("anthropic/claude-sonnet-4-6")], "anthropic/claude-sonnet-4-6"), + ); + + assertEquals(resolveVeryfrontCloudDefaultModelId(), "anthropic/claude-sonnet-4-6"); + assertEquals(resolveVeryfrontCloudModelId(), "anthropic/claude-sonnet-4-6"); + }); + + it("refuses a Mistral model the catalog does not list", () => { + seedServedCatalogForTests(); + + assertEquals(isSupportedMistralModelId("mistral/mistral-small-2503"), true); + assertEquals(isSupportedMistralModelId("mistral/not-served"), false); + }); + }); + + describe("retired models", () => { + const retired = [ + "openai/gpt-5.4-nano", + "mistral/mistral-large-2512", + "google-ai-studio/gemini-3.1-pro-preview", + "google/gemini-3.1-pro-preview", + ]; + + it("refuses a retired model through Veryfront Cloud before the catalog loads", () => { + for (const modelId of retired) { + assertEquals(isRetiredVeryfrontCloudModelId(modelId), true, modelId); + assertEquals(isRetiredVeryfrontCloudModelId(`veryfront-cloud/${modelId}`), true, modelId); + // Mistral ids the list does not carry keep the Mistral refusal first. + assertThrows(() => resolveVeryfrontCloudModelId(modelId), Error); + } + assertThrows( + () => resolveVeryfrontCloudModelId("openai/gpt-5.4-nano"), + Error, + "no longer available", + ); + }); + + it("refuses a retired model even when a loaded catalog still lists it", () => { + __setVeryfrontCloudCatalogForTests(payload([ + row("openai/gpt-5.4-nano", { surface: "openai", operations: ["chat-completions"] }), + ])); + + assertThrows( + () => resolveVeryfrontCloudModelId("openai/gpt-5.4-nano"), + Error, + "no longer available", + ); + }); + + it("keeps retired models out of the shipped fallback and the parity fixtures", () => { + const fixtureIds: string[] = [...SERVED_MODEL_ROWS, ...UNSERVED_TABLE_MODEL_ROWS].map(( + model, + ) => model.modelId); + for (const modelId of retired) { + assertEquals(fixtureIds.includes(modelId), false, modelId); + } + assertThrows( + () => resolveVeryfrontCloudModelId("gpt-5.4-nano"), + Error, + "Unknown model alias", + ); + }); + }); + + describe("parity with the shipped table for today's models", () => { + const tableAliases = new Map(VERYFRONT_CLOUD_PROVIDER_ALIASES); + const tableKey = (modelId: string): string => { + const slash = modelId.indexOf("/"); + const provider = modelId.slice(0, slash); + return `${tableAliases.get(provider) ?? provider}/${modelId.slice(slash + 1)}`; + }; + const tableRouting = new Map(VERYFRONT_CLOUD_PROVIDER_ROUTING); + const tableCapabilities = new Map(VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES); + const servedRows = SERVED_MODEL_ROWS as readonly ServedRow[]; + + /** The transport plan the table-backed resolver chose, from the table's own inputs. */ + function tablePlan(provider: string, upstreamModelId: string, modelId: string) { + const routing = tableRouting.get(provider); + if (routing?.surface !== "openai" || routing.native !== true) { + return { transport: "chat-completions", pinned: true }; + } + const declared = tableCapabilities.get(tableKey(modelId))?.openAITransport; + if (declared !== undefined) return { transport: declared, pinned: true }; + const entry = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + tableKey(model.modelId) === tableKey(modelId) + ); + if (entry?.thinking === true || entry?.thinkingBudgetTokens !== undefined) { + return { transport: "responses", pinned: true }; + } + if (isOpenAIReasoningModel(upstreamModelId, "veryfront-cloud")) { + return { transport: "responses", pinned: true }; + } + return { transport: "chat-completions", pinned: false }; + } + + it("covers every served row with a shipped table entry", () => { + for (const served of servedRows) { + const entry = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + tableKey(model.modelId) === tableKey(served.modelId) + ); + assertEquals(entry !== undefined, true, `${served.modelId} is not in the shipped table`); + } + }); + + for (const served of servedRows) { + const entry = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + tableKey(model.modelId) === tableKey(served.modelId) + ); + if (!entry) continue; + + it(`serves the shipped facts of ${entry.modelId}`, () => { + __setVeryfrontCloudCatalogForTests(servedCatalogPayload()); + const key = tableKey(entry.modelId); + const [provider = "", upstreamModelId = ""] = [ + key.slice(0, key.indexOf("/")), + key.slice(key.indexOf("/") + 1), + ]; + const capabilities = tableCapabilities.get(key); + const routing = tableRouting.get(provider); + + assertEquals(resolveVeryfrontCloudProviderId(entry.modelId.split("/")[0] ?? ""), provider); + assertEquals(resolveVeryfrontCloudProviderRouting(provider).surface, routing?.surface); + assertEquals( + resolveVeryfrontCloudProviderRouting(provider).native, + routing?.native === true, + ); + assertEquals(resolveVeryfrontCloudModelId(entry.id), entry.modelId); + assertEquals( + resolveVeryfrontCloudOpenAITransport(entry.modelId), + capabilities?.openAITransport, + ); + assertEquals( + resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(entry.modelId), + capabilities?.openAIChatReasoningWithFunctionTools, + ); + assertEquals( + resolveVeryfrontCloudOpenAIChatSystemMessages(entry.modelId), + capabilities?.openAIChatPreserveSystemMessages, + ); + if (routing?.surface === "openai") { + assertEquals( + resolveVeryfrontCloudOpenAITransportPlan(provider, upstreamModelId), + tablePlan(provider, upstreamModelId, entry.modelId), + ); + } + + const tableThinking = entry.thinking === true || entry.thinkingBudgetTokens !== undefined + ? { + enabled: true, + ...(entry.thinkingBudgetTokens === undefined + ? {} + : { budgetTokens: entry.thinkingBudgetTokens }), + } + : undefined; + const servedThinking = resolveVeryfrontCloudModelThinking(entry.modelId); + if (capabilities?.anthropicThinkingMode === "adaptive") { + // An adaptive model takes no budget, so the served catalog declares + // none; what is sent is the same with or without the shipped one. + assertEquals(servedThinking?.enabled, true); + assertEquals( + resolveVeryfrontCloudThinkingProviderOptions(entry.modelId, servedThinking), + resolveVeryfrontCloudThinkingProviderOptions(entry.modelId, tableThinking), + ); + assertEquals( + resolveVeryfrontCloudReasoningOption(entry.modelId, servedThinking), + resolveVeryfrontCloudReasoningOption(entry.modelId, tableThinking), + ); + } else { + assertEquals(servedThinking, tableThinking); + } + }); + } + + it("serves the shipped default model", () => { + seedServedCatalogForTests(); + const tableDefault = VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.find((model) => + model.id === TABLE_DEFAULT_MODEL_ID + ); + + assertEquals(resolveVeryfrontCloudDefaultModelId(), tableDefault?.modelId); + assertEquals(DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID, tableDefault?.modelId); + }); + }); +}); diff --git a/src/provider/veryfront-cloud/model-catalog.test.ts b/src/provider/veryfront-cloud/model-catalog.test.ts index 45b69c1e25..763e36ddca 100644 --- a/src/provider/veryfront-cloud/model-catalog.test.ts +++ b/src/provider/veryfront-cloud/model-catalog.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertExists, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { canonicalVeryfrontCloudModelKey, DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL, @@ -29,6 +31,8 @@ import { import { VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES } from "./model-catalog.data.ts"; describe("provider/veryfront-cloud/model-catalog", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("retires DeepSeek from managed selections while retaining Mistral as default", () => { assertEquals(findVeryfrontCloudModelByModelId("deepseek/deepseek-v4-flash"), undefined); assertEquals( @@ -669,7 +673,6 @@ describe("provider/veryfront-cloud/model-catalog", () => { it("keeps adaptive Anthropic thinking out of provider-neutral reasoning", () => { for ( const modelId of [ - "anthropic/claude-opus-4-7", "anthropic/claude-opus-4-8", "veryfront-cloud/anthropic/claude-opus-4-8", ] diff --git a/src/provider/veryfront-cloud/model-catalog.ts b/src/provider/veryfront-cloud/model-catalog.ts index 15d16c2116..110a44f836 100644 --- a/src/provider/veryfront-cloud/model-catalog.ts +++ b/src/provider/veryfront-cloud/model-catalog.ts @@ -1,20 +1,28 @@ import { INVALID_ARGUMENT, NOT_SUPPORTED } from "#veryfront/errors"; import { isOpenAIReasoningModel } from "../shared/openai-reasoning.ts"; +import { getVeryfrontCloudBootstrap } from "#veryfront/platform/cloud/resolver.ts"; +import { createPrivateWeakStore } from "#veryfront/security/private-weak-store.ts"; +import type { ModelRuntime } from "../types.ts"; +import { getCurrentVeryfrontCloudContext } from "./context.ts"; import { - DEFAULT_VERYFRONT_CLOUD_GATEWAY_API_VERSION, - DEFAULT_VERYFRONT_CLOUD_MODEL_ID as CATALOG_DEFAULT_MODEL_ID, - DEFAULT_VERYFRONT_CLOUD_SURFACE, - VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES, - VERYFRONT_CLOUD_GATEWAY_PATH_PREFIX, - VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, - VERYFRONT_CLOUD_PROVIDER_ALIASES, - VERYFRONT_CLOUD_PROVIDER_LABELS as PROVIDER_LABELS, - VERYFRONT_CLOUD_PROVIDER_ORDER as PROVIDER_ORDER, - VERYFRONT_CLOUD_PROVIDER_ROUTING, - VERYFRONT_CLOUD_SURFACE_GATEWAY_API_VERSIONS, - type VeryfrontCloudModelTransportCapabilities, - type VeryfrontCloudProviderRouting, -} from "./model-catalog.data.ts"; + hasActiveVeryfrontCloudCatalogScope, + isVeryfrontCloudCatalogFresh, + peekVeryfrontCloudCatalog, + type VeryfrontCloudCatalog, + type VeryfrontCloudCatalogModel, + type VeryfrontCloudCatalogScopeKey, + veryfrontCloudCatalogScopeKey, +} from "./catalog-client.ts"; +import { SHIPPED_VERYFRONT_CLOUD_CATALOG } from "./model-catalog.deprecated.ts"; + +export { + DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL, + findVeryfrontCloudModel, + findVeryfrontCloudModelByModelId, + groupVeryfrontCloudModelsByProvider, + VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES, + VERYFRONT_CLOUD_CHAT_MODELS, +} from "./model-catalog.deprecated.ts"; /** * Veryfront Cloud providers listed in the catalog of this package. @@ -43,6 +51,16 @@ export type VeryfrontCloudProviderId = | KnownVeryfrontCloudProviderId | (string & Record); +/** + * A provider-qualified Veryfront Cloud model ID, for example + * `anthropic/claude-sonnet-4-6`. Any provider and model the platform serves + * fits, so a new model needs no release of this package. + */ +export type VeryfrontCloudModelId = `${string}/${string}`; + +/** A model ID routed through Veryfront Cloud: `veryfront-cloud//`. */ +export type VeryfrontCloudRuntimeModelId = `veryfront-cloud/${string}/${string}`; + /** Wire format a Veryfront Cloud gateway endpoint speaks. */ export type VeryfrontCloudWireSurface = "openai" | "anthropic" | "google"; @@ -55,6 +73,24 @@ export type VeryfrontCloudSurfaceId = | VeryfrontCloudWireSurface | (string & Record); +/** + * Gateway routing for one provider: the wire format its endpoint speaks, and + * whether it implements that format natively. On the OpenAI surface, only a + * native provider can use the Responses transport. + */ +export type VeryfrontCloudProviderRouting = { + readonly surface: VeryfrontCloudSurfaceId; + readonly native?: boolean; +}; + +/** Model-specific transport capabilities that cannot be inferred from the provider family. */ +type VeryfrontCloudModelTransportCapabilities = { + readonly anthropicThinkingMode?: "adaptive"; + readonly openAITransport?: "chat-completions" | "responses"; + readonly openAIChatReasoningWithFunctionTools?: boolean; + readonly openAIChatPreserveSystemMessages?: boolean; +}; + /** Configuration used by Veryfront Cloud model thinking. */ export type VeryfrontCloudModelThinkingConfig = { enabled: boolean; @@ -88,32 +124,219 @@ function requireThinkingBudgetTokens(value: unknown): number | undefined { } /** - * Default Veryfront Cloud model ID used when no model is configured. - * Update this when the current default is deprecated — otherwise the default - * path silently breaks for users who have not set an explicit model. + * Short ID of the built-in default model, used when no model is configured + * and the served catalog has not been loaded. */ -export const DEFAULT_VERYFRONT_CLOUD_MODEL_ID = CATALOG_DEFAULT_MODEL_ID; +export const DEFAULT_VERYFRONT_CLOUD_MODEL_ID = "mistral-small-2503"; /** Shared Veryfront Cloud model prefix value. */ export const VERYFRONT_CLOUD_MODEL_PREFIX = "veryfront-cloud/"; +/** Provider-qualified ID of the built-in default model. */ +export const DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID: VeryfrontCloudModelId = + "mistral/mistral-small-2503"; +/** Veryfront Cloud runtime ID of the built-in default model. */ +export const DEFAULT_VERYFRONT_CLOUD_RUNTIME_MODEL_ID: VeryfrontCloudRuntimeModelId = + `veryfront-cloud/${DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID}`; + +/** + * Provider-qualified ID of the default model: the one the served catalog + * names once it is loaded, otherwise the built-in default. + */ +export function resolveVeryfrontCloudDefaultModelId(): VeryfrontCloudModelId { + const served = loadedCatalog()?.defaultModelId; + return served !== undefined && served.includes("/") + ? served as VeryfrontCloudModelId + : DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID; +} + +/** Leading gateway path segments of a vendor-scoped route, shared by every surface. */ +const VENDOR_GATEWAY_PATH_PREFIX = "ai/gateway"; +/** Vendor-scoped gateway API version per wire protocol. */ +const VENDOR_GATEWAY_API_VERSIONS: ReadonlyMap = new Map([ + ["anthropic", "v1"], + ["openai", "v1"], + ["google", "v1beta"], +]); +/** Vendor-scoped gateway API version for a protocol without its own entry. */ +const DEFAULT_VENDOR_GATEWAY_API_VERSION = "v1"; +/** Surface used for a provider the served catalog does not describe. */ +const DEFAULT_VERYFRONT_CLOUD_SURFACE = "openai"; +/** + * Providers named after the wire protocol they implement. When no catalog + * describes a provider, one of these speaks its own protocol natively and any + * other provider speaks the default surface. + */ +const PROTOCOL_NAMED_PROVIDERS: ReadonlySet = new Set(["openai", "anthropic", "google"]); +/** + * Provider spellings that name a protocol-named provider. A protocol fact, not + * a model fact: it holds whether or not a catalog has loaded. + */ +const PROTOCOL_PROVIDER_ALIASES: ReadonlyMap = new Map([ + ["google-ai-studio", "google"], +]); + +/** Lookups built once per loaded catalog. */ +interface ServedCatalogIndex { + /** Provider segment of a served model ID -> the canonical provider it belongs to. */ + readonly providerAliases: ReadonlyMap; + /** Canonical provider -> routing derived from its served models. */ + readonly routing: ReadonlyMap>; + /** `/` -> served model. */ + readonly byKey: ReadonlyMap; + /** Short ID or bare alias -> served model. */ + readonly byShortId: ReadonlyMap; + /** Exact provider-qualified model ID -> served model. */ + readonly byModelId: ReadonlyMap; +} + +const servedIndexes = new WeakMap(); + +function modelSegment(modelId: string): string { + return modelId.slice(modelId.indexOf("/") + 1); +} + +function buildServedIndex(catalog: VeryfrontCloudCatalog): ServedCatalogIndex { + const providerAliases = new Map(); + const surfaces = new Map(); + const operationsKnown = new Set(); + const servesResponses = new Set(); + const byKey = new Map(); + const byShortId = new Map(); + const byModelId = new Map(); + + for (const model of catalog.models) { + const slashIndex = model.modelId.indexOf("/"); + if (slashIndex <= 0) continue; + const segment = model.modelId.slice(0, slashIndex); + if (!providerAliases.has(segment)) providerAliases.set(segment, model.provider); + if (!providerAliases.has(model.provider)) providerAliases.set(model.provider, model.provider); + if (model.surface !== undefined && !surfaces.has(model.provider)) { + surfaces.set(model.provider, model.surface); + } + if (model.operations !== undefined) { + operationsKnown.add(model.provider); + if (model.operations.includes("responses")) servesResponses.add(model.provider); + } + const key = `${model.provider}/${modelSegment(model.modelId)}`; + if (!byKey.has(key)) byKey.set(key, model); + if (!byModelId.has(model.modelId)) byModelId.set(model.modelId, model); + for (const shortId of [model.id, ...model.aliases]) { + if (!shortId.includes("/") && !byShortId.has(shortId)) byShortId.set(shortId, model); + } + } + + const routing = new Map>(); + for (const [provider, surface] of surfaces) { + // A provider is native to the OpenAI surface when the platform serves the + // Responses operation for one of its models. An API that serves no + // operations yet leaves the protocol-named rule in place. + const native = surface !== "openai" + ? true + : operationsKnown.has(provider) + ? servesResponses.has(provider) + : PROTOCOL_NAMED_PROVIDERS.has(provider); + routing.set(provider, Object.freeze({ surface, native })); + } + + return { providerAliases, routing, byKey, byShortId, byModelId }; +} + +/** + * The scope synchronous reads use: the one a model build names, otherwise the + * ambient Veryfront Cloud credentials. Undefined without credentials. + */ +function ambientScope(): + | { apiBaseUrl: string; apiToken: string; projectSlug?: string } + | undefined { + let bootstrap: ReturnType; + try { + bootstrap = getVeryfrontCloudBootstrap(); + } catch { + return undefined; + } + if (!bootstrap.apiToken || !bootstrap.apiBaseUrl) return undefined; + return { + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + }; +} + +/** + * @internal The non-secret key of the catalog synchronous reads use outside a + * model build: the ambient credentials' scope, or, in a context that does not + * hold credentials, the scope key it carries. Undefined when neither applies. + */ +export function currentVeryfrontCloudCatalogScopeKey(): VeryfrontCloudCatalogScopeKey | undefined { + const scope = ambientScope(); + if (scope) return veryfrontCloudCatalogScopeKey(scope); + const carried = getCurrentVeryfrontCloudContext()?.catalogScopeKey; + return carried ? carried as VeryfrontCloudCatalogScopeKey : undefined; +} -/** Private runtime Map for alias lookups, built from the frozen data entries. */ -const _providerAliasMap = new Map(VERYFRONT_CLOUD_PROVIDER_ALIASES); -/** Private runtime Map for transport-capability lookups, built from the frozen data entries. */ -const _transportCapabilitiesMap = new Map( - VERYFRONT_CLOUD_MODEL_TRANSPORT_CAPABILITIES, -); -/** Private runtime Map for provider routing lookups, built from the frozen data entries. */ -const _providerRoutingMap = new Map(VERYFRONT_CLOUD_PROVIDER_ROUTING); -/** Private runtime Map for gateway API version lookups, built from the frozen data entries. */ -const _surfaceGatewayApiVersionMap = new Map( - VERYFRONT_CLOUD_SURFACE_GATEWAY_API_VERSIONS, -); - -/** Resolve a supported gateway provider alias without consulting object prototypes. */ +/** The served catalog loaded for the scope reads use, or undefined before it loads. */ +function loadedCatalog(): VeryfrontCloudCatalog | undefined { + if (hasActiveVeryfrontCloudCatalogScope()) return peekVeryfrontCloudCatalog(); + const key = currentVeryfrontCloudCatalogScopeKey(); + // A scope-less read still sees a catalog fixed by a test hook. + return key ? peekVeryfrontCloudCatalog(key) : peekVeryfrontCloudCatalog(); +} + +/** + * Whether the catalog reads use may refuse a model it does not list: a fresh + * served catalog, or the shipped list while none has loaded. A stale served + * catalog may miss a model the platform has enabled since, so a refusal waits + * until it is refreshed and the platform answers for the model meanwhile. + */ +export function canVeryfrontCloudCatalogRefuse(): boolean { + if (loadedCatalog() === undefined) return true; + if (hasActiveVeryfrontCloudCatalogScope()) return isVeryfrontCloudCatalogFresh(); + const key = currentVeryfrontCloudCatalogScopeKey(); + return key ? isVeryfrontCloudCatalogFresh(key) : isVeryfrontCloudCatalogFresh(); +} + +/** + * Whether the served catalog loaded for the current scope lists this model, + * under any accepted spelling. A listed model is a Veryfront Cloud candidate + * even when its provider is one this package does not name. + */ +export function isListedInServedVeryfrontCloudCatalog(modelId: string): boolean { + if (loadedCatalog() === undefined) return false; + const model = servedIndex().byKey.get(canonicalVeryfrontCloudModelKey(modelId)); + return model !== undefined && !isRetiredVeryfrontCloudModelId(model.modelId); +} + +/** + * Whether a served catalog has loaded for the scope reads use right now. While + * it has not, reads fall back to the shipped list, which cannot know models + * the platform added since, so a caller should not refuse a model on it alone. + */ +export function isVeryfrontCloudCatalogLoaded(): boolean { + return loadedCatalog() !== undefined; +} + +/** + * The index reads use: the served catalog loaded for the current scope, or the + * shipped list while none has loaded for it. + */ +function servedIndex(): ServedCatalogIndex { + const catalog = loadedCatalog() ?? SHIPPED_VERYFRONT_CLOUD_CATALOG; + let index = servedIndexes.get(catalog); + if (!index) { + index = buildServedIndex(catalog); + servedIndexes.set(catalog, index); + } + return index; +} + +/** + * Canonical provider for a provider segment the catalog spells, for example + * `google-ai-studio` for `google`. Undefined when neither the catalog nor the + * protocol aliases name the segment. + */ export function normalizeVeryfrontCloudProviderAlias( provider: string, -): KnownVeryfrontCloudProviderId | undefined { - return _providerAliasMap.get(provider); +): VeryfrontCloudProviderId | undefined { + return servedIndex().providerAliases.get(provider) ?? PROTOCOL_PROVIDER_ALIASES.get(provider); } /** @@ -147,18 +370,33 @@ export function resolveVeryfrontCloudProviderId( : undefined; } -/** Routing used for a provider the catalog data does not list. */ +/** Routing used for a provider the served catalog does not describe. */ const DEFAULT_PROVIDER_ROUTING: Readonly = Object .freeze({ surface: DEFAULT_VERYFRONT_CLOUD_SURFACE, }); -/** Gateway routing declared for a provider, or the default for an unlisted one. */ +/** Routing of a provider named after its protocol, used until the catalog describes it. */ +const PROTOCOL_NAMED_ROUTING: ReadonlyMap> = + new Map( + [...PROTOCOL_NAMED_PROVIDERS].map(( + protocol, + ) => [protocol, Object.freeze({ surface: protocol, native: true })]), + ); + +/** + * Gateway routing for a provider, as the served catalog describes it. Before + * the catalog is loaded, and for a provider it does not describe, a provider + * named after a protocol speaks that protocol natively and any other provider + * speaks the default surface. + */ export function resolveVeryfrontCloudProviderRouting( provider: string, ): Readonly { const canonical = normalizeVeryfrontCloudProviderAlias(provider) ?? provider; - return _providerRoutingMap.get(canonical) ?? DEFAULT_PROVIDER_ROUTING; + return servedIndex().routing.get(canonical) ?? + PROTOCOL_NAMED_ROUTING.get(canonical) ?? + DEFAULT_PROVIDER_ROUTING; } /** Wire format the given provider's gateway endpoint speaks. */ @@ -200,10 +438,10 @@ export function resolveVeryfrontCloudGatewayPath( ): string | undefined { const providerId = resolveVeryfrontCloudProviderId(provider); if (!providerId) return undefined; - const apiVersion = _surfaceGatewayApiVersionMap.get( + const apiVersion = VENDOR_GATEWAY_API_VERSIONS.get( resolveVeryfrontCloudSurface(providerId), - ) ?? DEFAULT_VERYFRONT_CLOUD_GATEWAY_API_VERSION; - return `${VERYFRONT_CLOUD_GATEWAY_PATH_PREFIX}/${providerId}/${apiVersion}`; + ) ?? DEFAULT_VENDOR_GATEWAY_API_VERSION; + return `${VENDOR_GATEWAY_PATH_PREFIX}/${providerId}/${apiVersion}`; } /** @@ -241,12 +479,38 @@ export function canonicalVeryfrontCloudModelKey(modelId: string): string { : `${provider}/${normalizedModelId.slice(slashIndex + 1)}`; } +/** + * The served model a model ID names: by provider-qualified ID in any provider + * spelling, or by short ID or bare alias. Undefined before the catalog is + * loaded, or when the catalog does not list the model. + */ +function findServedModel(modelId: string): VeryfrontCloudCatalogModel | undefined { + const index = servedIndex(); + return index.byKey.get(canonicalVeryfrontCloudModelKey(modelId)) ?? + index.byShortId.get(normalizeVeryfrontCloudModelId(modelId)); +} + +function isOpenAITransport(value: string | undefined): value is "chat-completions" | "responses" { + return value === "chat-completions" || value === "responses"; +} + function getVeryfrontCloudModelTransportCapabilities( modelId: string, ): Readonly | undefined { - return _transportCapabilitiesMap.get( - canonicalVeryfrontCloudModelKey(modelId), - ); + const model = findServedModel(modelId); + if (!model) return undefined; + return { + ...(model.surface === "anthropic" && model.reasoningMode === "adaptive" + ? { anthropicThinkingMode: "adaptive" as const } + : {}), + ...(isOpenAITransport(model.transport) ? { openAITransport: model.transport } : {}), + ...(model.chatCompletionsReasoningWithFunctionTools === undefined ? {} : { + openAIChatReasoningWithFunctionTools: model.chatCompletionsReasoningWithFunctionTools, + }), + ...(model.chatCompletionsConsecutiveSystemMessages === undefined ? {} : { + openAIChatPreserveSystemMessages: model.chatCompletionsConsecutiveSystemMessages, + }), + }; } /** Resolves a model-specific OpenAI transport override for Veryfront Cloud. */ @@ -324,6 +588,12 @@ export function resolveVeryfrontCloudOpenAITransportPlan( if (declared !== undefined) { return declared === "responses" ? RESPONSES_PINNED : CHAT_COMPLETIONS_PINNED; } + // A model the platform does not serve on Responses keeps to chat completions, + // even when its provider serves Responses for other models. + const operations = findServedModel(catalogModelId)?.operations; + if (operations !== undefined && !operations.includes("responses")) { + return CHAT_COMPLETIONS_PINNED; + } if (resolveVeryfrontCloudModelThinking(catalogModelId)?.enabled === true) { return RESPONSES_PINNED; } @@ -335,6 +605,41 @@ export function resolveVeryfrontCloudOpenAITransportPlan( return CHAT_COMPLETIONS_ADAPTIVE; } +/** @internal The catalog facts one built Veryfront Cloud model was built with. */ +export interface VeryfrontCloudModelFacts { + readonly provider: string; + readonly surface: VeryfrontCloudSurfaceId; + readonly native: boolean; + readonly transportPlan: VeryfrontCloudOpenAITransportPlan; + readonly openAITransport?: "chat-completions" | "responses"; + readonly openAIChatReasoningWithFunctionTools?: boolean; + readonly openAIChatPreserveSystemMessages?: boolean; +} + +const builtModelFacts = createPrivateWeakStore VeryfrontCloudModelFacts>(); + +/** @internal Record where a built model's current facts are read from. */ +export function registerVeryfrontCloudModelFacts( + model: ModelRuntime, + read: () => VeryfrontCloudModelFacts, +): void { + builtModelFacts.set(model, read); +} + +/** + * @internal The facts a Veryfront Cloud model built by this package currently + * calls with, so a record of a call describes the request actually sent. + * Undefined for any other object. + */ +export function readVeryfrontCloudModelFacts( + model: unknown, +): VeryfrontCloudModelFacts | undefined { + if (model === null || (typeof model !== "object" && typeof model !== "function")) { + return undefined; + } + return builtModelFacts.get(model as ModelRuntime)?.(); +} + /** Transport one call uses, given whether that call carries a hosted tool. */ export function resolveVeryfrontCloudOpenAICallTransport( provider: string, @@ -349,13 +654,6 @@ export function resolveVeryfrontCloudOpenAICallTransport( return usesHostedTool ? "responses" : "chat-completions"; } -/** - * Returns true if the given model ID is a Mistral model in the catalog. - * - * Compared by canonical key on both sides, so a catalog entry served under a - * provider alias and a request spelling the canonical provider (or carrying - * the gateway prefix) still meet. - */ /** * Whether an id is a Mistral id under any spelling the runtime accepts: the * gateway prefix stripped and the provider segment resolved through the alias @@ -366,17 +664,17 @@ function isMistralModelId(modelId: string): boolean { } /** - * Model ids the gateway no longer serves. Removing them from the catalog is not - * enough: explicit provider ids pass through unlisted, so the gateway boundary - * rejects these by name. They stay usable with the vendor's own key. + * Model ids the gateway no longer serves, keyed by canonical provider. + * Removing them from the catalog is not enough: explicit provider ids pass + * through unlisted, and the shipped list still backs reads before the served + * catalog loads, so the gateway boundary rejects these by name. They stay + * usable with the vendor's own key. */ -const RETIRED_VERYFRONT_CLOUD_MODEL_KEYS: ReadonlySet = new Set( - [ - "openai/gpt-5.4-nano", - "google-ai-studio/gemini-3.1-pro-preview", - "mistral/mistral-large-2512", - ].map(canonicalVeryfrontCloudModelKey), -); +const RETIRED_VERYFRONT_CLOUD_MODEL_KEYS: ReadonlySet = new Set([ + "openai/gpt-5.4-nano", + "google/gemini-3.1-pro-preview", + "mistral/mistral-large-2512", +]); /** Whether the gateway has retired this model id, under any accepted spelling. */ export function isRetiredVeryfrontCloudModelId(modelId: string): boolean { @@ -391,66 +689,13 @@ export function createRetiredVeryfrontCloudModelError(modelId: string): Error { }); } +/** + * Whether a Mistral model ID is one the catalog lists: the served catalog once + * it has loaded for the current scope, otherwise the shipped list. + */ export function isSupportedMistralModelId(modelId: string): boolean { - const key = canonicalVeryfrontCloudModelKey(modelId); - return VERYFRONT_CLOUD_CHAT_MODELS.some( - (model) => - model.provider === "mistral" && - canonicalVeryfrontCloudModelKey(model.modelId) === key, - ); -} - -/** Shared Veryfront Cloud chat models value. */ -export const VERYFRONT_CLOUD_CHAT_MODELS: readonly VeryfrontCloudChatModel[] = Object.freeze( - VERYFRONT_CLOUD_CHAT_MODEL_ENTRIES.map((model) => { - if ( - model.thinkingBudgetTokens !== undefined && - !isPositiveSafeInteger(model.thinkingBudgetTokens) - ) { - throw new TypeError( - `Veryfront Cloud model "${model.id}" thinkingBudgetTokens must be a positive safe integer`, - ); - } - return Object.freeze(model); - }), -); - -/** - * Every provider name the catalog data routes: accepted aliases, providers - * with a routing row, and providers of a listed chat model. Runtime model - * resolution sends `/` for these through the gateway when no - * direct provider credential applies. - */ -export const VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES: readonly string[] = Object.freeze([ - ...new Set([ - ...VERYFRONT_CLOUD_PROVIDER_ALIASES.map(([alias]) => alias), - ...VERYFRONT_CLOUD_PROVIDER_ROUTING.map(([provider]) => provider), - ...VERYFRONT_CLOUD_CHAT_MODELS.map((model) => model.provider), - ]), -]); - -const defaultVeryfrontCloudChatModel = VERYFRONT_CLOUD_CHAT_MODELS.find( - (model) => model.id === DEFAULT_VERYFRONT_CLOUD_MODEL_ID, -); -if (!defaultVeryfrontCloudChatModel) { - throw new Error( - `Veryfront Cloud default model "${DEFAULT_VERYFRONT_CLOUD_MODEL_ID}" is missing from the catalog`, - ); -} - -/** Catalog-backed default model descriptor. */ -export const DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL = defaultVeryfrontCloudChatModel; -/** Canonical direct provider/model ID for the default chat model. */ -export const DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID = DEFAULT_VERYFRONT_CLOUD_CHAT_MODEL.modelId; -/** Canonical hosted runtime ID for the default chat model. */ -export const DEFAULT_VERYFRONT_CLOUD_RUNTIME_MODEL_ID = - `${VERYFRONT_CLOUD_MODEL_PREFIX}${DEFAULT_VERYFRONT_CLOUD_PROVIDER_MODEL_ID}`; - -/** Find Veryfront Cloud model. */ -export function findVeryfrontCloudModel( - id: string, -): VeryfrontCloudChatModel | undefined { - return VERYFRONT_CLOUD_CHAT_MODELS.find((model) => model.id === id); + const index = servedIndex(); + return index.byKey.get(canonicalVeryfrontCloudModelKey(modelId))?.provider === "mistral"; } /** Normalizes Veryfront Cloud model ID. */ @@ -460,19 +705,6 @@ export function normalizeVeryfrontCloudModelId(modelId: string): string { : modelId; } -/** Find Veryfront Cloud model by model ID. */ -export function findVeryfrontCloudModelByModelId( - modelId: string, -): VeryfrontCloudChatModel | undefined { - // Compared by canonical key on both sides: the catalog may publish a model - // under a provider alias while a caller spells the canonical provider, or - // the other way round once the catalog moves on and the alias is retained. - const key = canonicalVeryfrontCloudModelKey(modelId); - return VERYFRONT_CLOUD_CHAT_MODELS.find( - (model) => canonicalVeryfrontCloudModelKey(model.modelId) === key, - ); -} - /** * Return the Veryfront Cloud provider named by a model ID, including a provider this package does not list. * @@ -495,6 +727,16 @@ export function getVeryfrontCloudProviderFromModelId( } /** Return the Veryfront Cloud provider named by a model ID, including one this package does not list, or `undefined` when the ID names none. */ +/** + * Whether a Veryfront Cloud model ID speaks the Anthropic protocol: its + * provider is served on the Anthropic surface. A newly served provider on that + * surface counts, not only `anthropic/*`. + */ +export function isVeryfrontCloudAnthropicSurfaceModel(modelId: string): boolean { + const provider = tryGetVeryfrontCloudProviderFromModelId(modelId); + return provider !== undefined && resolveVeryfrontCloudSurface(provider) === "anthropic"; +} + export function tryGetVeryfrontCloudProviderFromModelId( modelId: string, ): VeryfrontCloudProviderId | undefined { @@ -505,13 +747,37 @@ export function tryGetVeryfrontCloudProviderFromModelId( } } -/** Resolves Veryfront Cloud model ID. */ +/** + * Provider-qualified ID of a short alias only the served catalog knows, for + * example one the platform added after this release. Reads a catalog loaded + * for the current scope, never the shipped list, and never a retired model. + * Undefined when no served catalog has loaded or it does not name the alias. + */ +export function resolveServedVeryfrontCloudAlias(alias: string): string | undefined { + if (alias.includes("/") || loadedCatalog() === undefined) return undefined; + const model = servedIndex().byShortId.get(alias); + if (!model || isRetiredVeryfrontCloudModelId(model.modelId)) return undefined; + return model.modelId; +} + +/** + * Resolve a model ID or short alias to a provider-qualified model ID. + * + * No value resolves to the default model. A provider-qualified ID is returned + * as written. A short ID or alias resolves through the served catalog, or + * through the shipped list before the catalog has loaded; use + * `loadVeryfrontCloudModelCatalog()` first to resolve an alias the platform + * added since this release. + */ export function resolveVeryfrontCloudModelId(alias?: string): string { - const requestedModel = alias || DEFAULT_VERYFRONT_CLOUD_MODEL_ID; - const catalogModel = VERYFRONT_CLOUD_CHAT_MODELS.find((model) => - model.modelId === requestedModel - ); + const requestedModel = alias || resolveVeryfrontCloudDefaultModelId(); + const index = servedIndex(); + const catalogModel = index.byModelId.get(requestedModel); if (catalogModel) { + // A stale served list may still name a model the gateway has retired. + if (isRetiredVeryfrontCloudModelId(catalogModel.modelId)) { + throw createRetiredVeryfrontCloudModelError(catalogModel.modelId); + } return catalogModel.modelId; } @@ -520,6 +786,7 @@ export function resolveVeryfrontCloudModelId(alias?: string): string { // list so callers get a clear error rather than a gateway-side failure. if ( isMistralModelId(requestedModel) && + canVeryfrontCloudCatalogRefuse() && !isSupportedMistralModelId(requestedModel) ) { throw NOT_SUPPORTED.create({ @@ -532,7 +799,7 @@ export function resolveVeryfrontCloudModelId(alias?: string): string { return requestedModel; } - const model = findVeryfrontCloudModel(requestedModel); + const model = index.byShortId.get(requestedModel); if (!model) { throw INVALID_ARGUMENT.create({ detail: `Unknown model alias "${requestedModel}"`, @@ -575,7 +842,10 @@ export function resolveVeryfrontCloudGatewayModelId( // Unsupported Mistral ids are passed through unprefixed (not routed through // the Veryfront Cloud gateway prefix). - if (isMistralModelId(modelId) && !isSupportedMistralModelId(modelId)) { + if ( + isMistralModelId(modelId) && canVeryfrontCloudCatalogRefuse() && + !isSupportedMistralModelId(modelId) + ) { return modelId; } @@ -595,9 +865,8 @@ export function resolveVeryfrontCloudModelThinking( return undefined; } - const model = findVeryfrontCloudModelByModelId(modelId) ?? - findVeryfrontCloudModel(modelId); - const budgetTokens = requireThinkingBudgetTokens(model?.thinkingBudgetTokens); + const model = findServedModel(modelId); + const budgetTokens = requireThinkingBudgetTokens(model?.reasoningBudgetTokens); if (model?.thinking !== true && budgetTokens === undefined) { return undefined; } @@ -686,21 +955,6 @@ export function resolveVeryfrontCloudThinkingProviderOptions( }; } -/** Group Veryfront Cloud models by provider. */ -export function groupVeryfrontCloudModelsByProvider(): Array<{ - readonly provider: KnownVeryfrontCloudProviderId; - readonly label: string; - readonly models: readonly VeryfrontCloudChatModel[]; -}> { - return PROVIDER_ORDER.map((provider) => ({ - provider, - label: PROVIDER_LABELS[provider], - models: Object.freeze( - VERYFRONT_CLOUD_CHAT_MODELS.filter((model) => model.provider === provider), - ), - })).filter((group) => group.models.length > 0); -} - /** * Prefix a model ID for a hosted run. Alias of * {@link resolveVeryfrontCloudGatewayModelId}, with the same contract: call it diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index 3c1f8754dd..1c12e697d8 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -1,7 +1,15 @@ import "#veryfront/schemas/_test-setup.ts"; import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { assertEquals, assertThrows } from "#veryfront/testing/assert.ts"; -import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests, servedCatalogPayload } from "./catalog-client.test-helpers.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogClockForTests, + __setVeryfrontCloudCatalogForScopeForTests, + VERYFRONT_CLOUD_CATALOG_RETRY_MS, + VERYFRONT_CLOUD_CATALOG_TTL_MS, +} from "./catalog-client.ts"; import { agent } from "#veryfront/agent"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; import { clearEmbeddingProviders, resolveEmbeddingModel } from "#veryfront/embedding/index.ts"; @@ -9,7 +17,19 @@ import { ensureBuiltinLLMProviders } from "#veryfront/extensions/builtin-extensi import { clearModelProviders, resolveModel } from "#veryfront/provider"; import type { ModelRuntime } from "#veryfront/provider/types.ts"; import { getVeryfrontCloudAuthToken } from "#veryfront/platform/cloud/resolver.ts"; -import { createVeryfrontCloudInferenceModel, createVeryfrontCloudModel } from "./provider.ts"; +import { + createVeryfrontCloudInferenceModel, + createVeryfrontCloudModel, + warmVeryfrontCloudCatalog, +} from "./provider.ts"; +import { + readVeryfrontCloudModelFacts, + resolveVeryfrontCloudModelThinking, +} from "./model-catalog.ts"; +import { loadVeryfrontCloudModelCatalog } from "./shared.ts"; +import { generateText } from "#veryfront/runtime/runtime-bridge.ts"; +import { runWithMandatoryRunEventSink } from "#veryfront/runtime/run-event-sink-context.ts"; +import type { AgentRunEvent } from "#veryfront/runtime/model-call-context.ts"; import { createVeryfrontCloudFetch, getVeryfrontCloudGatewayBaseUrl, @@ -113,6 +133,8 @@ function setCloudBootstrap(): void { } describe("provider/veryfront-cloud", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { restoreMockFetch(); clearCloudEnv(); @@ -1437,6 +1459,8 @@ describe("provider/veryfront-cloud", () => { }); describe("provider/veryfront-cloud vendor-neutral routes", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { restoreMockFetch(); clearCloudEnv(); @@ -1805,3 +1829,355 @@ describe("provider/veryfront-cloud vendor-neutral routes", () => { assertEquals(sent, ['{"input":"x"}', "[1,2]", "not json", '{"model":7}']); }); }); + +describe("provider/veryfront-cloud served catalog loading", () => { + beforeEach(__resetVeryfrontCloudCatalogForTests); + afterEach(() => { + __resetVeryfrontCloudCatalogForTests(); + restoreMockFetch(); + clearCloudEnv(); + clearModelProviders(); + }); + + type CapturedRequest = { + method: string; + url: string; + authorization: string | null; + projectSlug: string | null; + }; + + /** Answer the catalog request from `catalog` and every other request with a finished chat stream. */ + function installGateway(catalog: () => Response, catalogGate?: Promise): CapturedRequest[] { + const requests: CapturedRequest[] = []; + const encoder = new TextEncoder(); + installMockFetch( + ((input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + requests.push({ + method: request.method, + url: request.url, + authorization: request.headers.get("authorization"), + projectSlug: request.headers.get("x-veryfront-project-slug"), + }); + if (request.url.endsWith("/ai/models")) { + return (catalogGate ?? Promise.resolve()).then(catalog); + } + return Promise.resolve( + new Response( + readableStreamFrom([ + encoder.encode('data: {"choices":[{"delta":{"content":"Hi"}}]}\n\n'), + encoder.encode('data: {"choices":[{"finish_reason":"stop"}]}\n\n'), + encoder.encode("data: [DONE]\n\n"), + ]), + { status: 200, headers: { "content-type": "text/event-stream" } }, + ), + ); + }) as typeof fetch, + ); + return requests; + } + + /** Stream once; a response the model cannot parse still leaves its request captured. */ + async function streamOnce(model: ModelRuntime): Promise { + try { + const result = await model.doStream({ + prompt: [{ role: "user", content: [{ type: "text", text: "Hi" }] }], + } as never); + await drainStream(result.stream); + } catch { + // expected for a Responses request answered with a chat stream + } + } + + const calls = (requests: CapturedRequest[]) => + requests.map(({ method, url }) => `${method} ${url.replace("https://api.veryfront.com", "")}`); + + /** A catalog serving one OpenAI-protocol model only on chat completions. */ + const chatOnlyCatalog = () => + Response.json({ + models: [{ + id: "gpt-5.9-chat", + modelId: "openai/gpt-5.9-chat", + provider: "openai", + surface: "openai", + operations: ["chat-completions"], + aliases: [], + capabilities: { thinking: true }, + }], + }); + + it("builds synchronously, then loads the catalog on the first call and follows it", async () => { + setCloudBootstrap(); + const requests = installGateway(chatOnlyCatalog); + + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + // Unlisted and reasoning-style: before the catalog loads it would use Responses. + assertEquals(readVeryfrontCloudModelFacts(model)?.transportPlan.transport, "responses"); + assertEquals(requests.length, 0); + await streamOnce(model); + await streamOnce(model); + + assertEquals(calls(requests), [ + "GET /ai/models", + "POST /ai/v1/chat/completions", + "POST /ai/v1/chat/completions", + ]); + assertEquals(requests[0]?.authorization, "Bearer vf_test_provider"); + assertEquals(requests[0]?.projectSlug, "provider-test-project"); + }); + + it("loads the catalog in prepare, before the first call", async () => { + setCloudBootstrap(); + const requests = installGateway(() => Response.json(servedCatalogPayload())); + + const model = resolveModel("veryfront-cloud/mistral/mistral-small-2503") as ModelRuntime; + await model.prepare?.(); + + assertEquals(requests.map(({ url }) => url), ["https://api.veryfront.com/ai/models"]); + }); + + it("retries the catalog on a later call after a failed first load", async () => { + setCloudBootstrap(); + let now = 1_000_000; + __setVeryfrontCloudCatalogClockForTests(() => now); + let available = false; + const requests = installGateway(() => + available ? chatOnlyCatalog() : Response.json({ error: "unavailable" }, { status: 503 }) + ); + + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + await streamOnce(model); + available = true; + // Inside the retry window no new catalog request is made. + await streamOnce(model); + now += VERYFRONT_CLOUD_CATALOG_RETRY_MS; + await streamOnce(model); + await streamOnce(model); + + assertEquals(calls(requests), [ + "GET /ai/models", + "POST /ai/v1/responses", + "POST /ai/v1/responses", + "GET /ai/models", + "POST /ai/v1/chat/completions", + "POST /ai/v1/chat/completions", + ]); + }); + + it("does not settle a model on a caller that stopped waiting", async () => { + setCloudBootstrap(); + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const requests = installGateway(chatOnlyCatalog, gate); + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + + const controller = new AbortController(); + controller.abort(); + await model.prepare?.(controller.signal); + release?.(); + await streamOnce(model); + + // The abandoned wait left the model unsettled; the call used the catalog. + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + }); + + it("lets one caller give up without deciding for a concurrent caller", async () => { + setCloudBootstrap(); + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const requests = installGateway(chatOnlyCatalog, gate); + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + + const controller = new AbortController(); + const abandoned = model.prepare?.(controller.signal); + const waiting = streamOnce(model); + controller.abort(); + await abandoned; + release?.(); + await waiting; + + // The waiting caller used the served catalog, fetched once for both. + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + }); + + it("does not refuse a model against a stale catalog, and serves it after the refresh", async () => { + setCloudBootstrap(); + let now = 1_000_000; + __setVeryfrontCloudCatalogClockForTests(() => now); + const scope = { + apiBaseUrl: "https://api.veryfront.com", + apiToken: "vf_test_provider", + projectSlug: "provider-test-project", + }; + // A catalog loaded before the platform enabled mistral/mistral-new. + __setVeryfrontCloudCatalogForScopeForTests(scope, { + models: [{ + id: "mistral-small-2503", + modelId: "mistral/mistral-small-2503", + provider: "mistral", + surface: "openai", + operations: ["chat-completions"], + }], + }); + now += VERYFRONT_CLOUD_CATALOG_TTL_MS; + const requests = installGateway(() => + Response.json({ + models: [{ + id: "mistral-new", + modelId: "mistral/mistral-new", + provider: "mistral", + surface: "openai", + operations: ["chat-completions"], + }], + }) + ); + + const model = resolveModel("veryfront-cloud/mistral/mistral-new") as ModelRuntime; + await streamOnce(model); + + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + }); + + it("still refuses a model a fresh catalog does not list", () => { + setCloudBootstrap(); + __setVeryfrontCloudCatalogForScopeForTests({ + apiBaseUrl: "https://api.veryfront.com", + apiToken: "vf_test_provider", + projectSlug: "provider-test-project", + }, { + models: [{ + id: "mistral-small-2503", + modelId: "mistral/mistral-small-2503", + provider: "mistral", + surface: "openai", + }], + }); + + assertThrows( + () => resolveModel("veryfront-cloud/mistral/mistral-new"), + Error, + 'Unsupported Mistral model "mistral/mistral-new"', + ); + }); + + it("validates a response format against the protocol the served catalog settles", async () => { + setCloudBootstrap(); + const requests = installGateway(() => + Response.json({ + models: [{ + id: "acme-claude", + modelId: "acme/acme-claude", + provider: "acme", + surface: "anthropic", + operations: ["messages"], + aliases: [], + capabilities: {}, + }], + }) + ); + const call = (model: ModelRuntime) => + generateText({ + model, + messages: [{ role: "user", content: "Hi" }], + responseFormat: { type: "json" }, + }); + + // Cold, the unlisted provider would look like an OpenAI-protocol model. + const cold = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + const coldError = await call(cold).then(() => undefined, (error: unknown) => error); + const warm = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + const warmError = await call(warm).then(() => undefined, (error: unknown) => error); + + assertEquals(warmError instanceof Error, true); + assertEquals((coldError as Error | undefined)?.message, (warmError as Error).message); + // Refused before any inference request. + assertEquals(calls(requests), ["GET /ai/models"]); + }); + + it("forwards metadata to the model rebuilt from the catalog", async () => { + setCloudBootstrap(); + installGateway(() => + Response.json({ + models: [{ + id: "acme-claude", + modelId: "acme/acme-claude", + provider: "acme", + surface: "anthropic", + operations: ["messages"], + aliases: [], + capabilities: {}, + }], + }) + ); + + const model = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + const coldCapabilities = model.runtimeCapabilities; + await model.prepare?.(); + + // A model constructed now, with the catalog loaded, is the reference. + const warm = resolveModel("veryfront-cloud/acme/acme-claude") as ModelRuntime; + assertEquals(coldCapabilities, { structuredOutput: true }); + assertEquals( + JSON.stringify(warm.runtimeCapabilities) === JSON.stringify(coldCapabilities), + false, + ); + assertEquals(model.runtimeCapabilities, warm.runtimeCapabilities); + assertEquals(model.modelProvider, "acme"); + assertEquals(readVeryfrontCloudModelFacts(model)?.surface, "anthropic"); + }); + + it("records the transport the request is sent with, from a cold start", async () => { + setCloudBootstrap(); + const bodies: Record[] = []; + const requests = installGateway(chatOnlyCatalog); + const captureBodies = globalThis.fetch; + installMockFetch( + (async (input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + if (request.method === "POST") bodies.push(await request.clone().json()); + return captureBodies(request); + }) as typeof fetch, + ); + const recorded: AgentRunEvent[] = []; + + const model = resolveModel("veryfront-cloud/openai/gpt-5.9-chat") as ModelRuntime; + assertEquals(readVeryfrontCloudModelFacts(model)?.transportPlan.transport, "responses"); + await runWithMandatoryRunEventSink( + (event) => { + recorded.push(event); + }, + () => generateText({ model, messages: [{ role: "user", content: "Hi" }], seed: 7 }), + ); + + const context = recorded.find((event) => + event.type === "AGENT_RUN_MODEL_CALL_CONTEXT_RECORDED" + ) as + | { request?: { seed?: number } } + | undefined; + assertEquals(calls(requests), ["GET /ai/models", "POST /ai/v1/chat/completions"]); + // Chat completions carries the seed; Responses would not. Record and request agree. + assertEquals(bodies[0]?.seed, 7); + assertEquals(context?.request?.seed, 7); + }); + + it("loads the catalog with the ambient credentials for synchronous reads", async () => { + setCloudBootstrap(); + const requests = installGateway(chatOnlyCatalog); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-5.9-chat"), undefined); + + assertEquals(await loadVeryfrontCloudModelCatalog(), true); + await warmVeryfrontCloudCatalog(); + + assertEquals(requests.map(({ url }) => url), ["https://api.veryfront.com/ai/models"]); + assertEquals(resolveVeryfrontCloudModelThinking("openai/gpt-5.9-chat"), { enabled: true }); + }); + + it("skips the ambient load without credentials", async () => { + const requests = installGateway(() => Response.json(servedCatalogPayload())); + + assertEquals(await loadVeryfrontCloudModelCatalog(), false); + await warmVeryfrontCloudCatalog(); + + assertEquals(requests, []); + }); +}); diff --git a/src/provider/veryfront-cloud/provider.ts b/src/provider/veryfront-cloud/provider.ts index 47432ee3fd..e37b39ff0f 100644 --- a/src/provider/veryfront-cloud/provider.ts +++ b/src/provider/veryfront-cloud/provider.ts @@ -8,7 +8,9 @@ import { getHostSecret } from "#veryfront/platform/compat/process/env.ts"; import type { ModelRuntime } from "../types.ts"; import { getCurrentVeryfrontCloudContext } from "./context.ts"; import { + assertVeryfrontCloudModelListed, createVeryfrontCloudFetch, + loadVeryfrontCloudModelCatalog, parseVeryfrontCloudModelId, requireVeryfrontCloudBootstrap, resolveVeryfrontCloudGatewayRoute, @@ -18,12 +20,20 @@ import { createVeryfrontCloudOpenAIResponsesModel, } from "./openai.ts"; import { + registerVeryfrontCloudModelFacts, requireVeryfrontCloudWireSurface, resolveVeryfrontCloudOpenAIChatFunctionToolReasoning, resolveVeryfrontCloudOpenAIChatSystemMessages, + resolveVeryfrontCloudOpenAITransport, resolveVeryfrontCloudOpenAITransportPlan, resolveVeryfrontCloudProviderRouting, + type VeryfrontCloudModelFacts, } from "./model-catalog.ts"; +import { + isVeryfrontCloudCatalogFresh, + loadVeryfrontCloudCatalog, + withVeryfrontCloudCatalogScope, +} from "./catalog-client.ts"; const IntrinsicReflectApply = Reflect.apply; const HostCrypto = globalThis.crypto; @@ -37,6 +47,7 @@ const ObjectGetPrototypeOf = Object.getPrototypeOf; const ObjectHasOwn = Object.hasOwn; const ObjectPrototype = Object.prototype; const ReflectOwnKeys = Reflect.ownKeys; +const ReflectGet = Reflect.get; function bindModelMethod unknown>( method: T, @@ -82,34 +93,131 @@ function wrapVeryfrontCloudModel( return wrapped; } +/** Upper bound on how long a warm-up before a model call waits for the catalog. */ +const CATALOG_WARM_UP_MAX_WAIT_MS = 3_000; + +/** + * @internal Load the catalog with the ambient credentials before a model call + * reads facts synchronously, waiting at most a few seconds. + */ +export async function warmVeryfrontCloudCatalog(abortSignal?: AbortSignal): Promise { + await loadVeryfrontCloudModelCatalog({ + ...(abortSignal ? { signal: abortSignal } : {}), + maxWaitMs: CATALOG_WARM_UP_MAX_WAIT_MS, + }); +} + +/** Metadata keys a wrapped model forwards to the model it currently calls. */ +const NON_FORWARDED_KEYS: ReadonlySet = new Set([ + "prepare", + "doGenerate", + "doStream", + "constructor", +]); + +/** + * Optional members a model rebuilt onto another protocol can gain even when + * the model it was first built as lacks them (the Google protocol adds + * `_reconcileProviderMetadata`), so the wrapper forwards them regardless. + */ +const OPTIONAL_FORWARDED_KEYS: readonly PropertyKey[] = [ + "_reconcileProviderMetadata", + "_generateViaStream", + "runtimeCapabilities", + "executionMode", + "modelProvider", + "specificationVersion", +]; + +/** + * Wrap a built model so its first async step loads the served catalog. When + * the catalog changes how the model is built, calls and metadata go to the + * model rebuilt from it. Once the catalog is settled, calls go straight to the + * current model. + * + * A settled model keeps the facts it settled with for its lifetime: a catalog + * refreshed later applies to models constructed after the refresh. + */ +function withServedCatalog( + model: ModelRuntime, + current: () => ModelRuntime, + settled: () => ModelRuntime | undefined, + ready: (abortSignal?: AbortSignal) => Promise, +): ModelRuntime { + const readSignal = (options: unknown): AbortSignal | undefined => + options !== null && typeof options === "object" + ? (options as { abortSignal?: AbortSignal }).abortSignal + : undefined; + const wrapped = ObjectCreate(model, { + prepare: { + value: async (abortSignal?: AbortSignal): Promise => { + const target = settled() ?? await ready(abortSignal); + if (target.prepare) await target.prepare(abortSignal); + }, + }, + doGenerate: { + value: (options: unknown) => { + const target = settled(); + if (target) return target.doGenerate(options); + return (async () => await (await ready(readSignal(options))).doGenerate(options))(); + }, + }, + doStream: { + value: (options: unknown) => { + const target = settled(); + if (target) return target.doStream(options); + return (async () => await (await ready(readSignal(options))).doStream(options))(); + }, + }, + }); + + // Metadata (provider attribution, capabilities, model ID) follows the model + // the calls go to, so a rebuild never leaves the construction-time values. + const forwarded = new Set(); + const forward = (key: PropertyKey): void => { + if (forwarded.has(key) || NON_FORWARDED_KEYS.has(key)) return; + forwarded.add(key); + ObjectDefineProperty(wrapped, key, { + configurable: false, + enumerable: true, + get: () => { + const target = current(); + const value: unknown = IntrinsicReflectApply(ReflectGet, undefined, [target, key]); + return typeof value === "function" + ? IntrinsicReflectApply(FunctionBind, value, [target]) + : value; + }, + }); + }; + let source: object | null = model; + while (source && source !== ObjectPrototype) { + for (const key of ReflectOwnKeys(source)) forward(key); + source = ObjectGetPrototypeOf(source); + } + for (const key of OPTIONAL_FORWARDED_KEYS) forward(key); + return wrapped; +} + +type VeryfrontCloudModelOptions = { + apiBaseUrl?: string; + assertInferenceCredentialActive?: () => void; + credentialSource?: "application"; + providerSelection?: "first-party"; + assertCredentialActive?: () => void; +}; + function createVeryfrontCloudModelInternal( modelId: string, inferenceCredential?: string, - options: { - apiBaseUrl?: string; - assertInferenceCredentialActive?: () => void; - credentialSource?: "application"; - providerSelection?: "first-party"; - assertCredentialActive?: () => void; - } = {}, + options: VeryfrontCloudModelOptions = {}, ): ModelRuntime { - const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language"); + // Parsed here so a malformed or retired ID fails at construction. Whether the + // catalog lists the model is checked against this model's own catalog: now + // when it has loaded, otherwise once the first async step has loaded it. + parseVeryfrontCloudModelId(modelId, "language", { catalogChecks: false }); const { apiBaseUrl, apiToken, projectSlug } = options.credentialSource === "application" ? requireApplicationBootstrap() : requireVeryfrontCloudBootstrap(inferenceCredential, options.apiBaseUrl); - // Builders keep the upstream model id; on a vendor-neutral route the fetch - // wrapper sends it as `/`. - const { baseURL, wireModelProvider } = resolveVeryfrontCloudGatewayRoute(apiBaseUrl, provider); - const fetch = createVeryfrontCloudFetch(apiToken, baseURL, projectSlug, { - inferenceCredential: inferenceCredential !== undefined, - ...(wireModelProvider ? { wireModelProvider } : {}), - ...(options.assertCredentialActive || options.assertInferenceCredentialActive - ? { - assertInferenceCredentialActive: options.assertCredentialActive ?? - options.assertInferenceCredentialActive, - } - : {}), - }); const usesHostPrivateCredential = inferenceCredential === undefined && options.credentialSource !== "application" && getHostSecret("VERYFRONT_API_TOKEN") === apiToken; const usesPrivateCredential = inferenceCredential !== undefined || usesHostPrivateCredential || @@ -125,6 +233,156 @@ function createVeryfrontCloudModelInternal( // credential therefore uses only first-party transports that project code // cannot replace; ordinary project credentials retain extension behavior. const registry = useFirstPartyTransport ? undefined : ensureBuiltinLLMProviders(); + + const catalogScope = { apiBaseUrl, apiToken, ...(projectSlug ? { projectSlug } : {}) }; + + // The catalog facts a build reads, from this model's own credentials and + // project. A model built before the catalog loaded is rebuilt at its first + // async step when these differ. + function readFacts(): VeryfrontCloudModelFacts { + return withVeryfrontCloudCatalogScope(catalogScope, () => { + const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId( + modelId, + "language", + { catalogChecks: false }, + ); + const catalogModelId = `${provider}/${upstreamModelId}`; + const routing = resolveVeryfrontCloudProviderRouting(provider); + const openAIChatReasoningWithFunctionTools = + resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogModelId); + const openAIChatPreserveSystemMessages = resolveVeryfrontCloudOpenAIChatSystemMessages( + catalogModelId, + ); + const openAITransport = resolveVeryfrontCloudOpenAITransport(catalogModelId); + return Object.freeze({ + provider, + surface: routing.surface, + native: routing.native === true, + transportPlan: resolveVeryfrontCloudOpenAITransportPlan(provider, upstreamModelId), + ...(openAITransport === undefined ? {} : { openAITransport }), + ...(openAIChatReasoningWithFunctionTools === undefined + ? {} + : { openAIChatReasoningWithFunctionTools }), + ...(openAIChatPreserveSystemMessages === undefined + ? {} + : { openAIChatPreserveSystemMessages }), + }); + }); + } + const factsKey = (value: VeryfrontCloudModelFacts): string => + [ + value.provider, + value.surface, + String(value.native), + value.transportPlan.transport, + String(value.transportPlan.pinned), + String(value.openAITransport), + String(value.openAIChatReasoningWithFunctionTools), + String(value.openAIChatPreserveSystemMessages), + ].join("\n"); + + const build = (): ModelRuntime => + withVeryfrontCloudCatalogScope(catalogScope, () => + buildVeryfrontCloudModel({ + modelId, + inferenceCredential, + options, + apiBaseUrl, + apiToken, + projectSlug, + providerCredential, + registry, + useFirstPartyTransport, + })); + // The listing check runs in this model's own scope, against a catalog + // actually loaded for it; before that, the platform answers for the model. + const assertListed = (): void => + withVeryfrontCloudCatalogScope(catalogScope, () => { + const parsed = parseVeryfrontCloudModelId(modelId, "language", { catalogChecks: false }); + assertVeryfrontCloudModelListed(parsed.provider, parsed.modelId); + }); + // Refuse at construction only against a fresh catalog: a stale one may miss a + // model enabled since, so the check waits for the refresh on the first call. + if (isVeryfrontCloudCatalogFresh(catalogScope)) assertListed(); + let facts = readFacts(); + const built = build(); + let current = built; + let isSettled = false; + const rebuildIfChanged = (settle: boolean): ModelRuntime => { + const next = readFacts(); + if (settle) assertListed(); + if (factsKey(next) !== factsKey(facts)) current = build(); + facts = next; + // Only a catalog actually obtained settles the model: after a failed or + // abandoned load, the next call tries again. + if (settle) isSettled = true; + return current; + }; + // A fresh cached catalog settles the model without waiting on anything. + const settled = (): ModelRuntime | undefined => { + if (isSettled) return current; + if (!isVeryfrontCloudCatalogFresh(catalogScope)) return undefined; + return rebuildIfChanged(true); + }; + // Each caller waits on its own signal. Concurrent callers share only the + // underlying catalog request (one per credentials and project, in the + // catalog client), so one caller giving up never decides for another. + const ready = async (abortSignal?: AbortSignal): Promise => { + const catalog = await loadVeryfrontCloudCatalog({ + ...catalogScope, + fresh: true, + ...(abortSignal ? { signal: abortSignal } : {}), + }); + // Settle (and run the listing check) only on a fresh catalog. A stale one + // answers this call, and a later call tries the refresh again. + return rebuildIfChanged(catalog !== undefined && isVeryfrontCloudCatalogFresh(catalogScope)); + }; + const wrapped = withServedCatalog(built, () => current, settled, ready); + registerVeryfrontCloudModelFacts(wrapped, () => facts); + return wrapped; +} + +interface VeryfrontCloudModelBuild { + readonly modelId: string; + readonly inferenceCredential: string | undefined; + readonly options: VeryfrontCloudModelOptions; + readonly apiBaseUrl: string; + readonly apiToken: string; + readonly projectSlug: string | undefined; + readonly providerCredential: string; + readonly registry: ReturnType | undefined; + readonly useFirstPartyTransport: boolean; +} + +/** Build a model from the served facts as they stand now. */ +function buildVeryfrontCloudModel(build: VeryfrontCloudModelBuild): ModelRuntime { + const { + modelId, + inferenceCredential, + options, + apiBaseUrl, + apiToken, + projectSlug, + providerCredential, + registry, + useFirstPartyTransport, + } = build; + const { provider, modelId: upstreamModelId } = parseVeryfrontCloudModelId(modelId, "language", { + catalogChecks: false, + }); + // Builders keep the upstream model id; on a vendor-neutral route the fetch + // wrapper sends it as `/`. + const { baseURL, wireModelProvider } = resolveVeryfrontCloudGatewayRoute(apiBaseUrl, provider); + const fetch = createVeryfrontCloudFetch(apiToken, baseURL, projectSlug, { + inferenceCredential: inferenceCredential !== undefined, + ...(wireModelProvider ? { wireModelProvider } : {}), + ...(options.assertCredentialActive || options.assertInferenceCredentialActive + ? { + assertInferenceCredentialActive: options.assertCredentialActive ?? + options.assertInferenceCredentialActive, + } + : {}), + }); const routing = resolveVeryfrontCloudProviderRouting(provider); // A provider that only speaks the OpenAI wire format is promised the chat diff --git a/src/provider/veryfront-cloud/shared.test.ts b/src/provider/veryfront-cloud/shared.test.ts index 0e6415af4b..b0788e163d 100644 --- a/src/provider/veryfront-cloud/shared.test.ts +++ b/src/provider/veryfront-cloud/shared.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "./catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "./catalog-client.ts"; import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { isVeryfrontGatewayResponse } from "#veryfront/provider/runtime-loader/provider-http.ts"; import { @@ -15,6 +17,8 @@ import { } from "./shared.ts"; describe("provider/veryfront-cloud/shared", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("normalizes provider aliases when parsing model IDs", () => { assertEquals( parseVeryfrontCloudModelId("google-ai-studio/gemini-2.0-flash", "embedding"), diff --git a/src/provider/veryfront-cloud/shared.ts b/src/provider/veryfront-cloud/shared.ts index e076b2e48c..15a53b8366 100644 --- a/src/provider/veryfront-cloud/shared.ts +++ b/src/provider/veryfront-cloud/shared.ts @@ -18,6 +18,7 @@ import { markCurrentVeryfrontCloudBillingGroupUsed, } from "./context.ts"; import { + canVeryfrontCloudCatalogRefuse, createRetiredVeryfrontCloudModelError, isRetiredVeryfrontCloudModelId, isSupportedMistralModelId, @@ -26,6 +27,7 @@ import { resolveVeryfrontCloudSurface, type VeryfrontCloudProviderId, } from "./model-catalog.ts"; +import { loadVeryfrontCloudCatalog } from "./catalog-client.ts"; import { requireInferenceProviderCredential, requireProviderCredential, @@ -212,6 +214,14 @@ function createInvalidModelIdError(modelId: string): Error { export function parseVeryfrontCloudModelId( modelId: string, kind: "language" | "embedding", + options: { + /** + * Also refuse a model the catalog in effect does not list (the Mistral + * check). A model built before its catalog loaded passes `false` and runs + * the check once the catalog for its own credentials is known. + */ + catalogChecks?: boolean; + } = {}, ): ParsedVeryfrontCloudModelId { const slashIndex = modelId.indexOf("/"); if (slashIndex === -1) { @@ -242,16 +252,8 @@ export function parseVeryfrontCloudModelId( ); } - if ( - kind === "language" && normalizedProvider === "mistral" && - !isSupportedMistralModelId(`mistral/${upstreamModelId}`) - ) { - throw toError( - createError({ - type: "config", - message: `Unsupported Mistral model "mistral/${upstreamModelId}"`, - }), - ); + if (kind === "language" && options.catalogChecks !== false) { + assertVeryfrontCloudModelListed(normalizedProvider, upstreamModelId); } if (kind === "language" && isRetiredVeryfrontCloudModelId(modelId)) { @@ -264,6 +266,24 @@ export function parseVeryfrontCloudModelId( }; } +/** + * Refuse a Mistral model the catalog in effect does not list, so a caller gets + * a clear error rather than a gateway-side failure. + */ +export function assertVeryfrontCloudModelListed(provider: string, upstreamModelId: string): void { + if ( + provider === "mistral" && canVeryfrontCloudCatalogRefuse() && + !isSupportedMistralModelId(`mistral/${upstreamModelId}`) + ) { + throw toError( + createError({ + type: "config", + message: `Unsupported Mistral model "mistral/${upstreamModelId}"`, + }), + ); + } +} + export function requireVeryfrontCloudBootstrap( apiTokenOverride?: string, inferenceApiBaseUrlOverride?: string, @@ -304,6 +324,35 @@ export function requireVeryfrontCloudBootstrap( }; } +/** + * Load the model catalog Veryfront Cloud serves, with the Veryfront Cloud + * credentials and project in effect, so model facts read synchronously + * afterward (thinking defaults, short aliases such as `opus`, the default + * model) come from it. Resolves to whether a catalog is available. Never + * throws. When a refresh fails, the last catalog loaded for these credentials + * stays in use; only when none has loaded (no credentials, or no load has + * succeeded yet) do the facts shipped with this package apply. + */ +export async function loadVeryfrontCloudModelCatalog( + options: { signal?: AbortSignal; maxWaitMs?: number } = {}, +): Promise { + let bootstrap: ReturnType; + try { + bootstrap = requireVeryfrontCloudBootstrap(); + } catch { + return false; + } + const catalog = await loadVeryfrontCloudCatalog({ + fresh: true, + apiBaseUrl: bootstrap.apiBaseUrl, + apiToken: bootstrap.apiToken, + ...(bootstrap.projectSlug ? { projectSlug: bootstrap.projectSlug } : {}), + ...(options.signal ? { signal: options.signal } : {}), + ...(options.maxWaitMs === undefined ? {} : { maxWaitMs: options.maxWaitMs }), + }); + return catalog !== undefined; +} + /** * Host environment variable that restores the vendor-scoped gateway routes. * diff --git a/src/runtime/model-call-context-request.test.ts b/src/runtime/model-call-context-request.test.ts index 10d3833f3f..c6c975787b 100644 --- a/src/runtime/model-call-context-request.test.ts +++ b/src/runtime/model-call-context-request.test.ts @@ -1,5 +1,7 @@ import { assertEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import type { ModelRuntimeCallOptions } from "#veryfront/provider/types.ts"; import { createWarningCollector } from "#veryfront/provider/shared/index.ts"; import { @@ -7,6 +9,7 @@ import { resolveVeryfrontCloudOpenAITransport, } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { buildModelCallContextRequest } from "#veryfront/runtime/model-call-context-request.ts"; +import { registerVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { buildOpenAIChatRequest } from "../../extensions/ext-llm-openai/src/openai-chat-request-builder.ts"; import { buildOpenAIResponsesRequest } from "../../extensions/ext-llm-openai/src/openai-responses-request-builder.ts"; import { buildAnthropicMessagesRequest } from "../../extensions/ext-llm-anthropic/src/anthropic-request-builder.ts"; @@ -34,6 +37,8 @@ const samplingFields = [ ] as const; describe("model call request projection", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("matches OpenAI-compatible Cloud controls including Kimi fixed sampling", () => { for ( const [modelProvider, modelId] of [["mistral", "mistral-large"], [ @@ -324,6 +329,30 @@ describe("model call request projection", () => { assertEquals((body.output_config as Record).effort, undefined); }); + it("records native controls by served surface for a newly served provider", () => { + const options: ModelRuntimeCallOptions = { + prompt, + providerOptions: { + anthropic: { thinking: { type: "adaptive" }, output_config: { effort: "high" } }, + }, + }; + const project = (modelProvider: string) => { + const model = { provider: "veryfront-cloud", modelProvider, modelId: "m1" }; + registerVeryfrontCloudModelFacts(model as never, () => + ({ + provider: modelProvider, + surface: "anthropic", + native: false, + transportPlan: "chat-completions", + }) as never); + return buildModelCallContextRequest(model, options); + }; + + // A provider served on the Anthropic surface records what anthropic/* records. + assertEquals(project("acme"), project("anthropic")); + assertEquals(project("acme")?.reasoning?.enabled, true); + }); + for (const modelId of ["gpt-5.4", "gpt-5.5"]) { it(`matches ${modelId} Cloud Chat reasoning with and without function tools`, () => { const catalogId = `openai/${modelId}`; diff --git a/src/runtime/model-call-context-request.ts b/src/runtime/model-call-context-request.ts index 717268c720..9c417fa7e3 100644 --- a/src/runtime/model-call-context-request.ts +++ b/src/runtime/model-call-context-request.ts @@ -9,6 +9,7 @@ import { } from "#veryfront/provider/shared/openai-reasoning.ts"; import { readProviderOptions } from "#veryfront/provider/runtime-loader.ts"; import { + readVeryfrontCloudModelFacts, resolveVeryfrontCloudOpenAICallTransport, resolveVeryfrontCloudOpenAIChatFunctionToolReasoning, resolveVeryfrontCloudOpenAITransport, @@ -52,7 +53,8 @@ function readProviderControl( ): PropertyDescriptor | undefined { const provider = resolveModelCallProvider(model); let selected: PropertyDescriptor | undefined; - for (const name of [provider, model.provider ?? provider]) { + // The protocol's bucket first, so a provider-named bucket still takes precedence. + for (const name of [resolveModelCallProtocol(model), provider, model.provider ?? provider]) { if (!name) continue; const bucket = readOwnEnumerableDataDescriptor(options.providerOptions, name)?.value; if (Array.isArray(bucket)) continue; @@ -74,8 +76,10 @@ function stopControl(value: unknown): string[] | undefined { function usesOpenAIBuilder(model: ModelCallRuntimeMetadata): boolean { const provider = resolveModelCallProvider(model); if (provider === "openai") return true; - return model.provider === "veryfront-cloud" && provider !== undefined && - resolveVeryfrontCloudProviderRouting(provider).surface === "openai"; + if (model.provider !== "veryfront-cloud" || provider === undefined) return false; + // A model built by this package records the facts it was built with. + const built = readVeryfrontCloudModelFacts(model); + return (built?.surface ?? resolveVeryfrontCloudProviderRouting(provider).surface) === "openai"; } function managedOpenAITransport( @@ -90,12 +94,15 @@ function managedOpenAITransport( // against the call is the one the request is built with. A provider that is // not native to the OpenAI surface never reaches the Responses transport, // whatever its model IDs look like. - return resolveVeryfrontCloudOpenAICallTransport( - provider, - model.modelId, + const usesHostedTool = options.tools?.some((tool) => tool.type === "provider" && tool.id.startsWith("openai.")) === - true, - ); + true; + const built = readVeryfrontCloudModelFacts(model); + if (built) { + if (built.transportPlan.pinned) return built.transportPlan.transport; + return usesHostedTool ? "responses" : "chat-completions"; + } + return resolveVeryfrontCloudOpenAICallTransport(provider, model.modelId, usesHostedTool); } function openAIProviderOptions( @@ -124,9 +131,9 @@ function resolvePersistedControls( model: ModelCallRuntimeMetadata, options: ModelCallRequestSource, ): ModelCallRequestSource { - const provider = resolveModelCallProvider(model); - if (provider === "anthropic") return resolveAnthropicControls(model, options); - if (provider === "google") return resolveGoogleControls(model, options); + const protocol = resolveModelCallProtocol(model); + if (protocol === "anthropic") return resolveAnthropicControls(model, options); + if (protocol === "google") return resolveGoogleControls(model, options); if (!usesOpenAIBuilder(model)) { return options; } @@ -272,6 +279,18 @@ function buildModelCallRequest( } /** Resolve the canonical provider recorded by the existing durable contract. */ +/** + * The wire protocol a model's request is built for. A Veryfront Cloud model + * speaks the surface it settled on, so a newly served provider on the Anthropic + * or Google surface records the same native controls as `anthropic/*` or + * `google/*`. Other models are identified by their provider name. + */ +function resolveModelCallProtocol(model: ModelCallRuntimeMetadata): string | undefined { + const surface = readVeryfrontCloudModelFacts(model)?.surface; + if (surface === "anthropic" || surface === "google") return surface; + return resolveModelCallProvider(model); +} + export function resolveModelCallProvider(model: ModelCallRuntimeMetadata): string | undefined { if (typeof model.modelProvider === "string" && model.modelProvider !== "") { return model.modelProvider; @@ -283,8 +302,7 @@ function resolvePersistedReasoning( model: ModelCallRuntimeMetadata, options: ModelCallRequestSource, ): RuntimeReasoningOption | undefined { - const modelProvider = resolveModelCallProvider(model); - if (modelProvider === "google") return resolveGoogleReasoning(model, options); + if (resolveModelCallProtocol(model) === "google") return resolveGoogleReasoning(model, options); if (usesOpenAIBuilder(model) && typeof model.modelId === "string") { const neutral = resolveOpenAINeutralReasoning(model, options); const transport = managedOpenAITransport(model, options); @@ -314,10 +332,17 @@ function suppressOpenAIFunctionToolReasoning( // must not apply it either. if (resolveModelCallProvider(model) !== "openai") return false; const catalogId = `openai/${model.modelId}`; + const built = readVeryfrontCloudModelFacts(model); + const openAITransport = built + ? built.openAITransport + : resolveVeryfrontCloudOpenAITransport(catalogId); + const reasoningWithFunctionTools = built + ? built.openAIChatReasoningWithFunctionTools + : resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogId); if ( model.provider === "veryfront-cloud" && - resolveVeryfrontCloudOpenAITransport(catalogId) === "chat-completions" && - resolveVeryfrontCloudOpenAIChatFunctionToolReasoning(catalogId) === false + openAITransport === "chat-completions" && + reasoningWithFunctionTools === false ) { // Match the Chat builder's native bucket precedence, including an own // tools value that clears the neutral list with [] or undefined. @@ -353,10 +378,9 @@ function resolveNonOpenAIReasoning( model: ModelCallRuntimeMetadata, options: ModelCallRequestSource, ): RuntimeReasoningOption | undefined { - const modelProvider = resolveModelCallProvider(model); // The Anthropic request builder only gives neutral reasoning precedence when // it enables thinking; otherwise a raw provider thinking config remains effective. - if (modelProvider !== "anthropic" || options.reasoning?.enabled === true) { + if (resolveModelCallProtocol(model) !== "anthropic" || options.reasoning?.enabled === true) { return options.reasoning; } diff --git a/src/runtime/runtime-bridge.ts b/src/runtime/runtime-bridge.ts index 190ce72b31..72597f9819 100644 --- a/src/runtime/runtime-bridge.ts +++ b/src/runtime/runtime-bridge.ts @@ -1,3 +1,4 @@ +import { readVeryfrontCloudModelFacts } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; import { mapPrivateArray, pushPrivateArray } from "#veryfront/security/private-array.ts"; import { createPrivateMap } from "#veryfront/security/private-map.ts"; import { getPrivateAsyncIterator } from "#veryfront/security/private-iterator.ts"; @@ -731,6 +732,21 @@ function buildDirectModelOptions( }; } +/** + * Settle a Veryfront Cloud model before anything reads its protocol or + * capabilities: it settles how it is built on its first async step, so the + * options validated and built for it, and the request recorded for it, match + * the request then sent. + */ +async function settleVeryfrontCloudModel(options: DirectTextOptions): Promise { + if ( + readVeryfrontCloudModelFacts(options.model) !== undefined && + typeof options.model.prepare === "function" + ) { + await options.model.prepare(options.abortSignal); + } +} + async function emitModelCallContextEvent( options: DirectTextOptions, directOptions: DirectModelOptions, @@ -1248,6 +1264,7 @@ async function* textDeltasFromStream(stream: ReadableStream): AsyncIter export function generateText(options: GenerateTextOptions): PromiseLike { return resolveDirectTools(options.tools).then(async (tools) => { + await settleVeryfrontCloudModel(options); const directOptions = buildDirectModelOptions(options, tools); await emitModelCallContextEvent(options, directOptions); if (shouldGenerateViaStream(options.model)) { @@ -1262,6 +1279,7 @@ export function generateText(options: GenerateTextOptions): PromiseLike { + await settleVeryfrontCloudModel(options); const directOptions = buildDirectModelOptions(options, tools); await emitModelCallContextEvent(options, directOptions); return options.model.doStream(directOptions); diff --git a/tests/integration/agent/hosted-application-model-resolver.test.ts b/tests/integration/agent/hosted-application-model-resolver.test.ts index f1dec2a23b..58899d37fd 100644 --- a/tests/integration/agent/hosted-application-model-resolver.test.ts +++ b/tests/integration/agent/hosted-application-model-resolver.test.ts @@ -1,6 +1,8 @@ import "#veryfront/schemas/_test-setup.ts"; import { assert, assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { withEnv } from "#veryfront/testing/deno-compat.ts"; import { @@ -74,6 +76,8 @@ async function drain(stream: ReadableStream) { } describe("hosted ordinary application model resolver", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("audits adaptive Cloud transport for generation and streaming on the same model", async () => { const resolver = createHostedApplicationModelResolver(resolverOptions()); try { diff --git a/tests/integration/agent/run-scoped-inference-credential.test.ts b/tests/integration/agent/run-scoped-inference-credential.test.ts index 393aef0dec..796a0cbc33 100644 --- a/tests/integration/agent/run-scoped-inference-credential.test.ts +++ b/tests/integration/agent/run-scoped-inference-credential.test.ts @@ -24,7 +24,9 @@ import { import { parseAgUiJsonBody } from "#veryfront/agent/ag-ui/request-shared.ts"; import { readBodyWithLimit } from "#veryfront/security/input-validation/limits.ts"; import { assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; -import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { installMockFetch, restoreMockFetch, @@ -73,6 +75,8 @@ function runtimeAgentInvocation(inferenceAuthToken: string): Record { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); afterEach(() => { restoreMockFetch(); clearModelProviders(); diff --git a/tests/integration/agent/veryfront-cloud-served-default-model.test.ts b/tests/integration/agent/veryfront-cloud-served-default-model.test.ts new file mode 100644 index 0000000000..0fd7e32082 --- /dev/null +++ b/tests/integration/agent/veryfront-cloud-served-default-model.test.ts @@ -0,0 +1,73 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; +import { clearModelProviders } from "#veryfront/provider"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { resolveAgentModelTransport } from "#veryfront/agent/runtime/model-transport.ts"; +import type { AgentConfig } from "#veryfront/agent/types.ts"; + +const SERVED_DEFAULT = "anthropic/claude-sonnet-4-6"; + +function servedCatalog(): Response { + return Response.json({ + models: [{ + id: "claude-sonnet-4-6", + modelId: SERVED_DEFAULT, + provider: "anthropic", + surface: "anthropic", + operations: ["messages"], + aliases: ["sonnet"], + capabilities: { thinking: true, reasoning_mode: "budget", reasoning_budget_tokens: 2048 }, + }], + defaultModelId: SERVED_DEFAULT, + }); +} + +describe("agent model transport with the served default model", () => { + let catalogRequests = 0; + + beforeEach(() => { + __resetVeryfrontCloudCatalogForTests(); + setEnv("VERYFRONT_API_TOKEN", "vf_default_model_test"); + setEnv("VERYFRONT_PROJECT_SLUG", "default-model-project"); + catalogRequests = 0; + installMockFetch( + ((input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + if (new URL(request.url).pathname === "/ai/models") { + catalogRequests++; + return Promise.resolve(servedCatalog()); + } + return Promise.resolve(new Response("unexpected", { status: 500 })); + }) as typeof fetch, + ); + }); + + afterEach(() => { + restoreMockFetch(); + __resetVeryfrontCloudCatalogForTests(); + deleteEnv("VERYFRONT_API_TOKEN"); + deleteEnv("VERYFRONT_PROJECT_SLUG"); + clearModelProviders(); + }); + + for (const model of [undefined, "auto"]) { + it(`resolves ${model ?? "an omitted model"} to the served default on the first request`, async () => { + const config: AgentConfig = { system: "You are concise.", ...(model ? { model } : {}) }; + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(catalogRequests, 1); + assertEquals(transport.resolvedModelString, `veryfront-cloud/${SERVED_DEFAULT}`); + assertEquals(transport.requestedModel, `veryfront-cloud/${SERVED_DEFAULT}`); + }); + } +}); diff --git a/tests/integration/agent/veryfront-cloud-served-only-models.test.ts b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts new file mode 100644 index 0000000000..77a362ab34 --- /dev/null +++ b/tests/integration/agent/veryfront-cloud-served-only-models.test.ts @@ -0,0 +1,392 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals, assertRejects, assertThrows } from "#veryfront/testing/assert.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; +import { + clearModelProviders, + loadVeryfrontCloudModelCatalog, + registerModelProvider, +} from "#veryfront/provider"; +import { + getCurrentVeryfrontCloudContext, + runWithVeryfrontCloudContext, + type VeryfrontCloudContext, +} from "#veryfront/provider/veryfront-cloud/context.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { resolveVeryfrontCloudModelId } from "#veryfront/provider/veryfront-cloud/model-catalog.ts"; +import { createVeryfrontCloudInferenceModel } from "#veryfront/provider/veryfront-cloud/provider.ts"; +import type { ModelRuntime } from "#veryfront/provider/types.ts"; +import { resolveAgentModelTransport } from "#veryfront/agent/runtime/model-transport.ts"; +import { resolveRuntimeModel } from "#veryfront/agent/runtime/model-resolution.ts"; +import { createDefaultHostedChatRuntime } from "#veryfront/agent/hosted/default-chat-runtime.ts"; +import type { + RemoteMCPToolSourceConfig, + RemoteToolSource, + ToolExecutionContext, +} from "#veryfront/tool"; +import { defineSchema } from "#veryfront/schemas/define.ts"; + +/** + * A process that has loaded no catalog meets a model and an alias only the + * served catalog knows. Each case drives one entry point cold, with ambient or + * explicit run credentials, and checks it routes through Veryfront Cloud with + * the catalog loaded for those same credentials. + */ + +const SERVED_ONLY_MODEL = "mistral/mistral-medium-2609"; +const SERVED_ONLY_ALIAS = "medium"; +const NEW_PROVIDER_MODEL = "acme-labs/m1"; +const NEW_PROVIDER_ALIAS = "acme-m1"; +const AMBIENT_TOKEN = "vf_ambient_token"; +const RUN_TOKEN = "vf_run_token"; + +type Captured = { method: string; path: string; authorization: string | null; model?: unknown }; + +/** A catalog that knows the served-only model and alias, for one credential. */ +function catalogFor(authorization: string | null): Response { + const knowsModel = authorization === `Bearer ${RUN_TOKEN}` || + authorization === `Bearer ${AMBIENT_TOKEN}`; + return Response.json({ + models: knowsModel + ? [{ + id: "mistral-medium-2609", + modelId: SERVED_ONLY_MODEL, + provider: "mistral", + surface: "openai", + operations: ["chat-completions"], + aliases: [SERVED_ONLY_ALIAS], + capabilities: {}, + }, { + id: "m1", + modelId: NEW_PROVIDER_MODEL, + provider: "acme-labs", + surface: "openai", + operations: ["chat-completions"], + aliases: [NEW_PROVIDER_ALIAS], + capabilities: {}, + }] + : [], + }); +} + +function installGateway(onlyRunCredentialKnowsModel = false): Captured[] { + const captured: Captured[] = []; + const encoder = new TextEncoder(); + installMockFetch( + (async (input: URL | Request | string, init?: RequestInit) => { + const request = new Request(input, init); + const authorization = request.headers.get("authorization"); + const entry: Captured = { + method: request.method, + path: new URL(request.url).pathname, + authorization, + }; + captured.push(entry); + if (entry.path === "/ai/models") { + return catalogFor( + onlyRunCredentialKnowsModel && authorization !== `Bearer ${RUN_TOKEN}` + ? null + : authorization, + ); + } + if (request.method === "POST") entry.model = (await request.json()).model; + return new Response( + new ReadableStream({ + start(controller) { + controller.enqueue(encoder.encode('data: {"choices":[{"finish_reason":"stop"}]}\n\n')); + controller.enqueue(encoder.encode("data: [DONE]\n\n")); + controller.close(); + }, + }), + { status: 200, headers: { "content-type": "text/event-stream" } }, + ); + }) as typeof fetch, + ); + return captured; +} + +async function streamOnce(model: ModelRuntime): Promise { + const result = await model.doStream({ prompt: [] } as never); + const reader = result.stream.getReader(); + while (!(await reader.read()).done) { + // drain + } +} + +describe("served-only models from a cold process", () => { + beforeEach(() => { + __resetVeryfrontCloudCatalogForTests(); + }); + + afterEach(() => { + restoreMockFetch(); + __resetVeryfrontCloudCatalogForTests(); + deleteEnv("VERYFRONT_API_TOKEN"); + deleteEnv("VERYFRONT_PROJECT_SLUG"); + clearModelProviders(); + }); + + describe("decided before the catalog loads", () => { + it("routes an explicit served-only model through Veryfront Cloud with ambient credentials", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + const captured = installGateway(); + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config: { model: SERVED_ONLY_MODEL, system: "You are concise." }, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(transport.resolvedModelString, `veryfront-cloud/${SERVED_ONLY_MODEL}`); + assertEquals(captured.map(({ path }) => path), ["/ai/models"]); + }); + + it("routes a served-only short alias through Veryfront Cloud from the agent transport", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + const captured = installGateway(); + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config: { model: SERVED_ONLY_ALIAS, system: "You are concise." }, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(transport.resolvedModelString, `veryfront-cloud/${SERVED_ONLY_MODEL}`); + assertEquals(transport.requestedModel, SERVED_ONLY_MODEL); + assertEquals(captured.map(({ path }) => path), ["/ai/models"]); + }); + + it("resolves a served-only alias in runtime model resolution only once a catalog loaded", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + installGateway(); + + // Cold: nothing names the alias, so it is left as written. + assertEquals(resolveRuntimeModel(SERVED_ONLY_ALIAS), SERVED_ONLY_ALIAS); + await loadVeryfrontCloudModelCatalog(); + assertEquals(resolveRuntimeModel(SERVED_ONLY_ALIAS), `veryfront-cloud/${SERVED_ONLY_MODEL}`); + // A known alias keeps its meaning. + assertEquals(resolveRuntimeModel("sonnet"), "veryfront-cloud/anthropic/claude-sonnet-4-6"); + }); + + for (const model of [NEW_PROVIDER_MODEL, NEW_PROVIDER_ALIAS]) { + it(`routes ${model}, served for a provider this package does not name, through Veryfront Cloud`, async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + installGateway(); + + const transport = await resolveAgentModelTransport({ + agentId: "agent-1", + config: { model, system: "You are concise." }, + context: undefined, + modelOverride: undefined, + mode: "stream", + }); + + assertEquals(transport.resolvedModelString, `veryfront-cloud/${NEW_PROVIDER_MODEL}`); + }); + } + + it("resolves a served-only alias once the ambient catalog is loaded", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "cold-project"); + installGateway(); + + assertEquals(await loadVeryfrontCloudModelCatalog(), true); + assertEquals(resolveVeryfrontCloudModelId(SERVED_ONLY_ALIAS), SERVED_ONLY_MODEL); + }); + + it("builds a served-only model with run credentials and serves it after its own load", async () => { + const captured = installGateway(); + + const model = createVeryfrontCloudInferenceModel(SERVED_ONLY_MODEL, RUN_TOKEN); + await streamOnce(model); + + assertEquals( + captured.map(({ method, path, authorization }) => `${method} ${path} ${authorization}`), + [ + `GET /ai/models Bearer ${RUN_TOKEN}`, + `POST /ai/v1/chat/completions Bearer ${RUN_TOKEN}`, + ], + ); + assertEquals(captured[1]?.model, SERVED_ONLY_MODEL); + }); + + it("refuses a Mistral model the run's own catalog does not list, on the first call", async () => { + installGateway(); + + const model = createVeryfrontCloudInferenceModel("mistral/not-served", RUN_TOKEN); + await assertRejects( + () => streamOnce(model), + Error, + 'Unsupported Mistral model "mistral/not-served"', + ); + }); + }); + + describe("read in the scope the catalog was loaded for", () => { + it("resolves a served-only alias in a credential-free hosted tool context, without the credential", async () => { + const captured = installGateway(true); + let resolvedInTool: string | undefined; + let toolContext: VeryfrontCloudContext | undefined; + let modelCalls = 0; + registerModelProvider("test", () => ({ + provider: "test", + modelId: "test/stripped-context", + doGenerate: () => Promise.reject(new Error("unused")), + doStream() { + modelCalls++; + return Promise.resolve({ + stream: new ReadableStream({ + start(controller) { + if (modelCalls === 1) { + controller.enqueue({ + type: "tool-call", + toolCallId: "child-1", + toolName: "invoke_child", + input: {}, + }); + controller.enqueue({ type: "finish", finishReason: "tool-calls", usage: {} }); + } else { + controller.enqueue({ type: "text-delta", text: "done" }); + controller.enqueue({ type: "finish", finishReason: "stop", usage: {} }); + } + controller.close(); + }, + }), + }); + }, + })); + + const runtime = await createDefaultHostedChatRuntime({ + sourceIntegrationPolicy: { schemaVersion: 1, mode: "unrestricted" }, + options: { + projectId: "project-1", + projectSlug: "run-project", + authToken: RUN_TOKEN, + instructions: "Invoke the child.", + model: "test/stripped-context", + allowedTools: ["invoke_child"], + }, + config: { + apiUrl: "https://api.veryfront.com", + apiMcpUrl: "https://api.veryfront.com/mcp", + }, + buildLocalTools: () => ({ + invoke_child: { + description: "Resolve a child model the way invoke_agent does", + inputSchema: defineSchema((v) => v.object({}))(), + execute: () => { + toolContext = getCurrentVeryfrontCloudContext(); + // invoke_agent resolves the child's model with this function. + resolvedInTool = resolveVeryfrontCloudModelId(SERVED_ONLY_ALIAS); + return { ok: true }; + }, + }, + }), + createRemoteToolSource: emptyRemoteSource, + preloadLatestConversationUserText: false, + }); + try { + const result = await runtime.agent.stream({ + messages: [], + abortSignal: new AbortController().signal, + }); + for await (const _chunk of result.toUIMessageStream()) { + // Consume the tool round trip. + } + } finally { + await runtime.cleanup?.(); + } + + assertEquals(resolvedInTool, SERVED_ONLY_MODEL); + assertEquals(toolContext?.apiToken, undefined); + assertEquals(typeof toolContext?.catalogScopeKey, "string"); + assertEquals(toolContext?.catalogScopeKey?.includes(RUN_TOKEN), false); + assertEquals(JSON.stringify(toolContext).includes(RUN_TOKEN), false); + assertEquals( + captured.filter(({ path }) => path === "/ai/models").map(({ authorization }) => + authorization + ), + [`Bearer ${RUN_TOKEN}`], + ); + }); + + it("falls back to the shipped aliases in a credential-free context whose run loaded nothing", () => { + const stripped: VeryfrontCloudContext = { + apiBaseUrl: "https://api.veryfront.com", + projectSlug: "run-project", + serviceLayer: "cloud", + }; + + runWithVeryfrontCloudContext(stripped, () => { + assertEquals(resolveVeryfrontCloudModelId("opus"), "anthropic/claude-opus-4-8"); + assertThrows( + () => resolveVeryfrontCloudModelId(SERVED_ONLY_ALIAS), + Error, + "Unknown model alias", + ); + }); + }); + + it("resolves a hosted alias from the run's catalog, not the ambient one", async () => { + setEnv("VERYFRONT_API_TOKEN", AMBIENT_TOKEN); + setEnv("VERYFRONT_PROJECT_SLUG", "ambient-project"); + const captured = installGateway(true); + + const runtime = await createDefaultHostedChatRuntime({ + sourceIntegrationPolicy: { schemaVersion: 1, mode: "unrestricted" }, + options: { + projectId: "project-1", + projectSlug: "run-project", + branchId: "branch-1", + authToken: RUN_TOKEN, + instructions: "Base instructions", + model: SERVED_ONLY_ALIAS, + allowedTools: [], + conversationId: "conversation-1", + userId: "user-1", + }, + config: { + apiUrl: "https://api.veryfront.com", + apiMcpUrl: "https://api.veryfront.com/mcp", + studioMcpUrl: "https://studio.example.com/mcp", + }, + buildLocalTools: () => ({ noop: localTool("No-op") }), + createRemoteToolSource: emptyRemoteSource, + preloadLatestConversationUserText: false, + }); + + assertEquals(runtime.modelId, SERVED_ONLY_MODEL); + const catalogRequests = captured.filter(({ path }) => path === "/ai/models"); + assertEquals(catalogRequests.map(({ authorization }) => authorization), [ + `Bearer ${RUN_TOKEN}`, + ]); + await runtime.cleanup?.(); + }); + }); +}); + +function localTool(description: string) { + return { + description, + inputSchema: defineSchema((v) => v.object({}))(), + execute: () => ({ ok: true }), + }; +} + +function emptyRemoteSource(config: RemoteMCPToolSourceConfig): RemoteToolSource { + return { + id: config.id ?? "source", + listTools: () => Promise.resolve([]), + executeTool: (_toolName: string, _args: unknown, _context?: ToolExecutionContext) => + Promise.resolve({ ok: true }), + }; +} diff --git a/tests/integration/eval/judge-catalog-cancellation.test.ts b/tests/integration/eval/judge-catalog-cancellation.test.ts new file mode 100644 index 0000000000..60770059d1 --- /dev/null +++ b/tests/integration/eval/judge-catalog-cancellation.test.ts @@ -0,0 +1,41 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { describe, it } from "#veryfront/testing/bdd.ts"; +import { withEnv } from "#veryfront/testing"; +import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { judges } from "veryfront/eval"; + +describe("eval judges with Veryfront Cloud", () => { + it("stops waiting for the model catalog when the evaluation is cancelled", async () => { + const controller = new AbortController(); + controller.abort(); + // The catalog request does not answer until the test ends; a cancelled judge + // must not wait for it. + let release: (response: Response) => void = () => {}; + const pending = new Promise((resolve) => release = resolve); + const hanging = () => pending; + try { + const started = Date.now(); + const result = await withEnv( + { VERYFRONT_API_TOKEN: "vf_judge_test", VERYFRONT_PROJECT_SLUG: "judge-test" }, + () => + withMockFetch(hanging, () => + judges.llm.rubric({ model: "openai/gpt-5-nano" })({ + input: "Question", + output: { text: "Answer" }, + metadata: {}, + rubric: "Correct.", + signal: controller.signal, + })), + ); + assertEquals(result.pass, false); + assertEquals(Date.now() - started < 1_000, true); + } finally { + // Let the shared request settle so its timeout is cleared. + release(new Response(null, { status: 503 })); + await new Promise((resolve) => setTimeout(resolve, 0)); + __resetVeryfrontCloudCatalogForTests(); + } + }); +}); diff --git a/tests/integration/provider/veryfront-cloud-catalog-client.test.ts b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts new file mode 100644 index 0000000000..a081f95ae5 --- /dev/null +++ b/tests/integration/provider/veryfront-cloud-catalog-client.test.ts @@ -0,0 +1,448 @@ +import "#veryfront/schemas/_test-setup.ts"; +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { afterEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { withMockFetch } from "#veryfront/testing/mock-fetch.ts"; +import { + __resetVeryfrontCloudCatalogForTests, + __setVeryfrontCloudCatalogClockForTests, + __setVeryfrontCloudCatalogForScopeForTests, + __veryfrontCloudCatalogSizesForTests, + loadVeryfrontCloudCatalog, + parseVeryfrontCloudCatalog, + peekVeryfrontCloudCatalog, + VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES, + VERYFRONT_CLOUD_CATALOG_RETRY_MS, + VERYFRONT_CLOUD_CATALOG_TTL_MS, +} from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; +import { servedCatalogPayload } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { logger } from "#veryfront/utils/logger/logger.ts"; + +const API_BASE_URL = "https://api.veryfront.com"; +const LOAD = { apiBaseUrl: API_BASE_URL, apiToken: "vf_catalog_test", projectSlug: "catalog-test" }; + +function catalogPayload(defaultModelId: string): Record { + return { ...servedCatalogPayload(), defaultModelId }; +} + +function jsonResponse(body: unknown, status = 200): Response { + return new Response(JSON.stringify(body), { + status, + headers: { "content-type": "application/json" }, + }); +} + +/** A fetch stub that answers every request from `respond` and records each one. */ +function recordingFetch(respond: () => Response | Promise): { + fetch: typeof fetch; + requests: Request[]; +} { + const requests: Request[] = []; + return { + requests, + fetch: ((input: URL | Request | string, init?: RequestInit) => { + requests.push(new Request(input, init)); + return Promise.resolve(respond()); + }) as typeof fetch, + }; +} + +/** Let a background refresh finish: it settles within one macrotask on a stub fetch. */ +function settleBackgroundRefresh(): Promise { + return new Promise((resolve) => setTimeout(resolve, 0)); +} + +function useClock(start = 1_000_000): { advance(ms: number): void } { + let current = start; + __setVeryfrontCloudCatalogClockForTests(() => current); + return { + advance(ms: number) { + current += ms; + }, + }; +} + +describe("provider/veryfront-cloud/catalog-client", () => { + afterEach(__resetVeryfrontCloudCatalogForTests); + + describe("parseVeryfrontCloudCatalog", () => { + it("reads the served model facts and the default model", () => { + const catalog = parseVeryfrontCloudCatalog(servedCatalogPayload()); + const sonnet = catalog?.models.find((model) => model.id === "claude-sonnet-4-6"); + const gpt = catalog?.models.find((model) => model.id === "gpt-5.5"); + const mistral = catalog?.models.find((model) => model.id === "mistral-small-2503"); + + assertEquals(catalog?.defaultModelId, "mistral/mistral-small-2503"); + assertEquals(sonnet?.surface, "anthropic"); + assertEquals(sonnet?.operations, ["messages"]); + assertEquals(sonnet?.reasoningBudgetTokens, 2048); + assertEquals(sonnet?.aliases.includes("sonnet"), true); + assertEquals(gpt?.transport, "chat-completions"); + assertEquals(gpt?.chatCompletionsReasoningWithFunctionTools, false); + assertEquals(mistral?.chatCompletionsConsecutiveSystemMessages, true); + }); + + it("reads a field an older API does not serve as absent", () => { + const catalog = parseVeryfrontCloudCatalog({ + models: [{ + id: "gpt-x", + modelId: "openai/gpt-x", + provider: "openai", + surface: "openai", + capabilities: { thinking: true }, + }], + }); + + assertEquals(catalog?.defaultModelId, undefined); + assertEquals(catalog?.models.length, 1); + assertEquals(catalog?.models[0]?.operations, undefined); + assertEquals(catalog?.models[0]?.reasoningBudgetTokens, undefined); + assertEquals(catalog?.models[0]?.chatCompletionsConsecutiveSystemMessages, undefined); + assertEquals(catalog?.models[0]?.aliases, []); + }); + + it("skips malformed rows and ignores invalid field values", () => { + const catalog = parseVeryfrontCloudCatalog({ + models: [ + null, + { id: "no-model-id", provider: "openai" }, + { + id: "gpt-y", + modelId: "openai/gpt-y", + provider: "openai", + operations: ["responses", 7], + capabilities: { reasoning_budget_tokens: 1.5, transport: 3 }, + }, + ], + }); + + assertEquals(catalog?.models.map((model) => model.id), ["gpt-y"]); + assertEquals(catalog?.models[0]?.operations, ["responses"]); + assertEquals(catalog?.models[0]?.reasoningBudgetTokens, undefined); + assertEquals(catalog?.models[0]?.transport, undefined); + }); + + it("returns undefined for a payload without a model list", () => { + assertEquals(parseVeryfrontCloudCatalog({ error: "nope" }), undefined); + assertEquals(parseVeryfrontCloudCatalog("[]"), undefined); + }); + }); + + describe("loadVeryfrontCloudCatalog", () => { + it("is cold until the first load, then serves the loaded catalog synchronously", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + assertEquals(peekVeryfrontCloudCatalog(LOAD), undefined); + + const loaded = await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(LOAD)); + + assertEquals(loaded?.defaultModelId, "mistral/mistral-small-2503"); + assertEquals(peekVeryfrontCloudCatalog(LOAD), loaded); + assertEquals(stub.requests.length, 1); + const [request] = stub.requests; + assertEquals(request?.method, "GET"); + assertEquals(request?.url, `${API_BASE_URL}/ai/models`); + assertEquals(request?.headers.get("authorization"), "Bearer vf_catalog_test"); + assertEquals(request?.headers.get("x-veryfront-project-slug"), "catalog-test"); + }); + + it("keeps the API base URL's path and query on the catalog request", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + await withMockFetch( + stub.fetch, + () => + loadVeryfrontCloudCatalog({ + ...LOAD, + apiBaseUrl: "https://api.veryfront.com/tenant/?scope=signed-value", + }), + ); + + assertEquals( + stub.requests[0]?.url, + "https://api.veryfront.com/tenant/ai/models?scope=signed-value", + ); + }); + + it("shares one request between concurrent loads for the same key", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + const [first, second] = await withMockFetch( + stub.fetch, + () => Promise.all([loadVeryfrontCloudCatalog(LOAD), loadVeryfrontCloudCatalog(LOAD)]), + ); + + assertEquals(stub.requests.length, 1); + assertEquals(first, second); + }); + + it("evicts the least recently used catalog over the cap and keeps the newest", () => { + const scopeFor = (index: number) => ({ ...LOAD, apiToken: `vf_rotating_${index}` }); + const payload = { models: [], defaultModelId: "mistral/mistral-small-2503" }; + for (let index = 0; index < VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES; index++) { + __setVeryfrontCloudCatalogForScopeForTests(scopeFor(index), payload); + } + // Reading the second entry makes it recent, so the first is evicted instead. + assertEquals(peekVeryfrontCloudCatalog(scopeFor(1)) !== undefined, true); + + __setVeryfrontCloudCatalogForScopeForTests( + scopeFor(VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES), + payload, + ); + + assertEquals( + __veryfrontCloudCatalogSizesForTests().entries, + VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES, + ); + assertEquals(peekVeryfrontCloudCatalog(scopeFor(0)), undefined); + assertEquals(peekVeryfrontCloudCatalog(scopeFor(1)) !== undefined, true); + assertEquals( + peekVeryfrontCloudCatalog(scopeFor(VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES)) !== undefined, + true, + ); + }); + + it("stays within the cap after more concurrent scopes than it holds all settle", async () => { + const count = VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES + 8; + let release: () => void = () => {}; + const gate = new Promise((resolve) => release = resolve); + // Every load is in flight at once, so none can be evicted while it lands. + const stub = recordingFetch(async () => { + await gate; + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + const loads = Array.from( + { length: count }, + (_, index) => loadVeryfrontCloudCatalog({ ...LOAD, apiToken: `vf_burst_${index}` }), + ); + release(); + await Promise.all(loads); + }); + + assertEquals(stub.requests.length, count); + assertEquals( + __veryfrontCloudCatalogSizesForTests().entries, + VERYFRONT_CLOUD_CATALOG_MAX_ENTRIES, + ); + }); + + it("forgets a failure once its retry window has passed", async () => { + const clock = useClock(); + const stub = recordingFetch(() => jsonResponse({}, 503)); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + assertEquals(__veryfrontCloudCatalogSizesForTests().failures, 1); + clock.advance(VERYFRONT_CLOUD_CATALOG_RETRY_MS); + await loadVeryfrontCloudCatalog({ ...LOAD, apiToken: "vf_other_credential" }); + }); + + // The first failure's window passed, so only the new one is kept. + assertEquals(__veryfrontCloudCatalogSizesForTests().failures, 1); + }); + + it("keeps a separate entry per credential for the same project", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + await loadVeryfrontCloudCatalog({ ...LOAD, apiToken: "vf_other_credential" }); + }); + + assertEquals( + stub.requests.map((request) => request.headers.get("authorization")), + ["Bearer vf_catalog_test", "Bearer vf_other_credential"], + ); + assertEquals(peekVeryfrontCloudCatalog({ ...LOAD, apiToken: "vf_third" }), undefined); + }); + + it("lets one caller stop waiting without failing the load for the others", async () => { + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const stub = recordingFetch(async () => { + await gate; + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + const controller = new AbortController(); + const abandoned = loadVeryfrontCloudCatalog({ ...LOAD, signal: controller.signal }); + const waiting = loadVeryfrontCloudCatalog(LOAD); + controller.abort(); + assertEquals(await abandoned, undefined); + + release?.(); + assertEquals((await waiting)?.defaultModelId, "mistral/mistral-small-2503"); + // The abandoned wait recorded no failure: the next load is a cache hit. + assertEquals( + (await loadVeryfrontCloudCatalog(LOAD))?.defaultModelId, + "mistral/mistral-small-2503", + ); + }); + assertEquals(stub.requests.length, 1); + assertEquals(stub.requests[0]?.signal.aborted, false); + }); + + it("stops waiting after maxWaitMs while the request finishes for later callers", async () => { + let release: (() => void) | undefined; + const gate = new Promise((resolve) => release = resolve); + const stub = recordingFetch(async () => { + await gate; + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + assertEquals(await loadVeryfrontCloudCatalog({ ...LOAD, maxWaitMs: 1 }), undefined); + release?.(); + await settleBackgroundRefresh(); + assertEquals(peekVeryfrontCloudCatalog(LOAD)?.defaultModelId, "mistral/mistral-small-2503"); + }); + assertEquals(stub.requests.length, 1); + }); + + it("keeps a separate entry per project", async () => { + const stub = recordingFetch(() => jsonResponse(servedCatalogPayload())); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + await loadVeryfrontCloudCatalog({ ...LOAD, projectSlug: "other-project" }); + await loadVeryfrontCloudCatalog(LOAD); + }); + + assertEquals( + stub.requests.map((request) => request.headers.get("x-veryfront-project-slug")), + ["catalog-test", "other-project"], + ); + }); + + it("serves a fresh entry from cache and refreshes a stale one in the background", async () => { + const clock = useClock(); + let defaultModelId = "mistral/mistral-small-2503"; + const stub = recordingFetch(() => jsonResponse(catalogPayload(defaultModelId))); + + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + clock.advance(VERYFRONT_CLOUD_CATALOG_TTL_MS - 1); + await loadVeryfrontCloudCatalog(LOAD); + assertEquals(stub.requests.length, 1); + + defaultModelId = "anthropic/claude-sonnet-4-6"; + clock.advance(1); + // Stale: the stale entry answers at once while one refresh runs. + const stale = await loadVeryfrontCloudCatalog(LOAD); + assertEquals(stale?.defaultModelId, "mistral/mistral-small-2503"); + + await settleBackgroundRefresh(); + assertEquals(stub.requests.length, 2); + const refreshed = await loadVeryfrontCloudCatalog(LOAD); + assertEquals(refreshed?.defaultModelId, "anthropic/claude-sonnet-4-6"); + assertEquals( + peekVeryfrontCloudCatalog(LOAD)?.defaultModelId, + "anthropic/claude-sonnet-4-6", + ); + assertEquals(stub.requests.length, 2); + }); + }); + + it("resolves to undefined when the catalog cannot be loaded, and retries later", async () => { + const clock = useClock(); + let status = 503; + const stub = recordingFetch(() => + status === 200 ? jsonResponse(servedCatalogPayload()) : jsonResponse({}, status) + ); + + await withMockFetch(stub.fetch, async () => { + assertEquals(await loadVeryfrontCloudCatalog(LOAD), undefined); + assertEquals(peekVeryfrontCloudCatalog(LOAD), undefined); + + status = 200; + clock.advance(VERYFRONT_CLOUD_CATALOG_RETRY_MS - 1); + assertEquals(await loadVeryfrontCloudCatalog(LOAD), undefined); + assertEquals(stub.requests.length, 1); + + clock.advance(1); + assertEquals( + (await loadVeryfrontCloudCatalog(LOAD))?.defaultModelId, + "mistral/mistral-small-2503", + ); + assertEquals(stub.requests.length, 2); + }); + }); + + it("treats a body without a model list as a failed load", async () => { + const stub = recordingFetch(() => jsonResponse({ unexpected: true })); + + const loaded = await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(LOAD)); + + assertEquals(loaded, undefined); + assertEquals(peekVeryfrontCloudCatalog(LOAD), undefined); + }); + + it("keeps the stale catalog when a refresh fails", async () => { + const clock = useClock(); + let fail = false; + const stub = recordingFetch(() => { + if (fail) return Promise.reject(new TypeError("network down")); + return jsonResponse(servedCatalogPayload()); + }); + + await withMockFetch(stub.fetch, async () => { + const loaded = await loadVeryfrontCloudCatalog(LOAD); + fail = true; + clock.advance(VERYFRONT_CLOUD_CATALOG_TTL_MS); + await loadVeryfrontCloudCatalog(LOAD); + await settleBackgroundRefresh(); + const afterFailure = await loadVeryfrontCloudCatalog(LOAD); + + assertEquals(stub.requests.length, 2); + assertEquals(afterFailure, loaded); + assertEquals(peekVeryfrontCloudCatalog(LOAD), loaded); + }); + }); + + it("never logs a signed query value from the API base URL or the error text", async () => { + const warnings: unknown[] = []; + const originalWarn = logger.warn; + logger.warn = (_message: string, fields?: unknown) => warnings.push(fields); + const signed = { ...LOAD, apiBaseUrl: `${API_BASE_URL}/tenant/?scope=signed-secret-value` }; + const stub = recordingFetch(() => + Promise.reject( + new TypeError( + `error sending request for url (${API_BASE_URL}/tenant/ai/models?scope=signed-secret-value)`, + ), + ) + ); + try { + await withMockFetch(stub.fetch, () => loadVeryfrontCloudCatalog(signed)); + } finally { + logger.warn = originalWarn; + } + + assertEquals(warnings.length, 1); + assertEquals(JSON.stringify(warnings).includes("signed-secret-value"), false); + assertEquals((warnings[0] as { apiBaseUrl?: string }).apiBaseUrl, `${API_BASE_URL}/tenant/`); + }); + + it("logs one warning per failing scope, naming the scope but never the credential", async () => { + const warnings: { message: string; fields: unknown }[] = []; + const originalWarn = logger.warn; + logger.warn = (message: string, fields?: unknown) => warnings.push({ message, fields }); + const other = { ...LOAD, apiToken: "vf_catalog_other", projectSlug: "catalog-other" }; + const stub = recordingFetch(() => Promise.reject(new TypeError("network down"))); + try { + await withMockFetch(stub.fetch, async () => { + await loadVeryfrontCloudCatalog(LOAD); + await loadVeryfrontCloudCatalog(other); + }); + } finally { + logger.warn = originalWarn; + } + + assertEquals( + warnings.map((warning) => (warning.fields as { projectSlug?: string }).projectSlug), + ["catalog-test", "catalog-other"], + ); + assertEquals(JSON.stringify(warnings).includes("vf_catalog_"), false); + }); + }); +}); diff --git a/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts b/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts index 997123d959..e190a9ed73 100644 --- a/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts +++ b/tests/integration/provider/veryfront-cloud-model-id-rule.test.ts @@ -1,5 +1,7 @@ import { assertEquals } from "#veryfront/testing/assert.ts"; -import { describe, it } from "#veryfront/testing/bdd.ts"; +import { afterEach, beforeEach, describe, it } from "#veryfront/testing/bdd.ts"; +import { seedServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; +import { __resetVeryfrontCloudCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.ts"; import { DEFAULT_VERYFRONT_CLOUD_MODEL_ID, findVeryfrontCloudModel, @@ -62,6 +64,8 @@ function runtimeAccepts(modelId: string): boolean { } describe("veryfront-cloud model id rule", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); for (const modelId of MODEL_IDS) { it(`agrees with the runtime about ${JSON.stringify(modelId)}`, () => { assertEquals( @@ -134,6 +138,8 @@ function representativePayload(): Record { * new way of producing one is caught here rather than found in the catalog. */ describe("veryfront-cloud catalog round trip", () => { + beforeEach(seedServedCatalogForTests); + afterEach(__resetVeryfrontCloudCatalogForTests); it("resolves every shipped entry back to itself", () => { assertEquals(VERYFRONT_CLOUD_CHAT_MODELS.length > 0, true); diff --git a/tests/integration/provider/veryfront-cloud-model-table-importers.test.ts b/tests/integration/provider/veryfront-cloud-model-table-importers.test.ts new file mode 100644 index 0000000000..3ec0c1ad2c --- /dev/null +++ b/tests/integration/provider/veryfront-cloud-model-table-importers.test.ts @@ -0,0 +1,36 @@ +import { assertEquals } from "#veryfront/testing/assert.ts"; +import { describe, it } from "#veryfront/testing/bdd.ts"; + +const SHIPPED_TABLE_MODULE = "model-catalog.data.ts"; +const SHIM_MODULE = "src/provider/veryfront-cloud/model-catalog.deprecated.ts"; + +/** Every non-test source file under `src/`, relative to the repository root. */ +async function sourceFiles(root: URL, dir = "src"): Promise { + const files: string[] = []; + for await (const entry of Deno.readDir(new URL(`${dir}/`, root))) { + const path = `${dir}/${entry.name}`; + if (entry.isDirectory) files.push(...await sourceFiles(root, path)); + else if (/\.tsx?$/.test(entry.name) && !/\.test(-helpers)?\.tsx?$/.test(entry.name)) { + files.push(path); + } + } + return files; +} + +describe("veryfront-cloud shipped model table", () => { + it("is the only source module that imports the shipped table", async () => { + const root = new URL("../../../", import.meta.url); + const importers: string[] = []; + for (const path of await sourceFiles(root)) { + if (path.endsWith(`/${SHIPPED_TABLE_MODULE}`)) continue; + const text = await Deno.readTextFile(new URL(path, root)); + if ( + new RegExp(`^import(?! type)[^;]*["'][^"']*${SHIPPED_TABLE_MODULE}["']`, "m").test(text) + ) { + importers.push(path); + } + } + + assertEquals(importers, [SHIM_MODULE]); + }); +}); diff --git a/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts b/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts index 9e8eceaadf..3e62d89fae 100644 --- a/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts +++ b/tests/integration/semantic-unit-boundary/src/provider/veryfront-cloud/issue-1834-recorded-context.test.ts @@ -1,6 +1,7 @@ import "#veryfront/schemas/_test-setup.ts"; import { assertEquals } from "#veryfront/testing/assert.ts"; import { afterEach, it } from "#veryfront/testing/bdd.ts"; +import { useServedCatalogForTests } from "#veryfront/provider/veryfront-cloud/catalog-client.test-helpers.ts"; import { installMockFetch, restoreMockFetch } from "#veryfront/testing/mock-fetch.ts"; import { deleteEnv, setEnv } from "#veryfront/compat/process.ts"; import { clearModelProviders } from "#veryfront/provider"; @@ -24,6 +25,7 @@ afterEach(() => { }); it("converts the sanitized issue 1834 context into a valid Mistral gateway ingress request", async () => { + using _catalog = useServedCatalogForTests(); setEnv("VERYFRONT_API_TOKEN", "vf_test_issue_1834"); setEnv("VERYFRONT_PROJECT_SLUG", "issue-1834-project"); let capturedUrl = "";