diff --git a/CHANGELOG.md b/CHANGELOG.md index 287c45cdde..c8d32c772d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,9 +63,9 @@ unchanged. ### Changed: Veryfront Cloud models call the vendor-neutral endpoints -`veryfront-cloud/*` models that speak the OpenAI protocol now send requests to -`/ai/v1`, and models that speak the Anthropic protocol send them to -`/ai/anthropic/v1`. The request body names the model as +`veryfront-cloud/*` models now send requests under `/ai/v1`: models that +speak the OpenAI protocol use `/chat/completions` or `/responses`, and models +that speak the Anthropic protocol use `/messages`. The request body names the model as `/`. Google models keep their existing route. Authentication, project and billing headers are unchanged, and platform refusals such as insufficient credits or a missing project are reported as before. diff --git a/docs/architecture/27-agent-message-stream-dataflow.md b/docs/architecture/27-agent-message-stream-dataflow.md index 644970a2f2..5b5d4b485d 100644 --- a/docs/architecture/27-agent-message-stream-dataflow.md +++ b/docs/architecture/27-agent-message-stream-dataflow.md @@ -235,9 +235,9 @@ and the provider runtime switch is in | Alias | Upstream model ID | Runtime path | Verified provider contract | Veryfront integration status | | ------------------ | ------------------------------------- | ------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------ | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| `opus` | `anthropic/claude-opus-4-8` | Anthropic Messages API through `ai/anthropic/v1` | Anthropic lists Opus 4.8 with 128k synchronous max output and adaptive thinking support, but no manual extended thinking support. | Catalog does not enable manual thinking for Opus 4.8, so provider options do not force temperature to 1 for this model. Runtime max output cap is 128k. | -| `sonnet` | `anthropic/claude-sonnet-4-6` | Anthropic Messages API through `ai/anthropic/v1` | Anthropic lists Sonnet 4.6 with 64k synchronous max output and manual extended thinking still functional but deprecated in favor of adaptive thinking. | Catalog enables manual thinking with 2048 budget tokens and provider options set temperature to 1. Runtime and Anthropic request caps are 64k. | -| `haiku` | `anthropic/claude-haiku-4-5-20251001` | Anthropic Messages API through `ai/anthropic/v1` | Anthropic lists Haiku 4.5 with 64k synchronous max output and extended thinking support. | Catalog enables manual thinking with 1024 budget tokens. Runtime max output cap is 64k. | +| `opus` | `anthropic/claude-opus-4-8` | Anthropic Messages API through `ai/v1/messages` | Anthropic lists Opus 4.8 with 128k synchronous max output and adaptive thinking support, but no manual extended thinking support. | Catalog does not enable manual thinking for Opus 4.8, so provider options do not force temperature to 1 for this model. Runtime max output cap is 128k. | +| `sonnet` | `anthropic/claude-sonnet-4-6` | Anthropic Messages API through `ai/v1/messages` | Anthropic lists Sonnet 4.6 with 64k synchronous max output and manual extended thinking still functional but deprecated in favor of adaptive thinking. | Catalog enables manual thinking with 2048 budget tokens and provider options set temperature to 1. Runtime and Anthropic request caps are 64k. | +| `haiku` | `anthropic/claude-haiku-4-5-20251001` | Anthropic Messages API through `ai/v1/messages` | Anthropic lists Haiku 4.5 with 64k synchronous max output and extended thinking support. | Catalog enables manual thinking with 1024 budget tokens. Runtime max output cap is 64k. | | `gpt-5.5` | `openai/gpt-5.5` | OpenAI-compatible runtime through `ai/v1` | OpenAI documents GPT-5.5 as the latest GPT-5 family target for API requests. | Catalog routes through the OpenAI runtime. Runtime max output cap is 128k. | | `gemini-3.5-flash` | `google-ai-studio/gemini-3.5-flash` | Google runtime through `ai/gateway/google/v1beta` | Google lists Gemini 3.5 Flash in the Gemini API model catalog. | Catalog aliases `google-ai-studio` to the Google provider. Runtime max output cap is 65,536. | | `kimi-k2.6` | `moonshotai/kimi-k2.6` | OpenAI-compatible runtime through `ai/v1` | Kimi documents Kimi K2.6 as the current model code for OpenAI-compatible chat completions, streaming, and tool use. | Catalog routes Moonshot through the OpenAI runtime. Runtime capability detection normalizes Kimi K2.6 requests to temperature 1 for thinking mode and 0.6 when thinking is disabled. | diff --git a/docs/guides/configuration.md b/docs/guides/configuration.md index 9cf3ae8e5d..38ff149a78 100644 --- a/docs/guides/configuration.md +++ b/docs/guides/configuration.md @@ -375,9 +375,10 @@ ignored and the host's own setting stands. ### Veryfront Cloud model routes -`veryfront-cloud/*` models call the platform's vendor-neutral endpoints. -Models that speak the OpenAI protocol use `/ai/v1`, and models that speak -the Anthropic protocol use `/ai/anthropic/v1`. The request body names the +`veryfront-cloud/*` models call the platform's vendor-neutral endpoints under +`/ai/v1`: models that speak the OpenAI protocol use `/chat/completions` or +`/responses`, and models that speak the Anthropic protocol use `/messages`. The +request body names the model as `/`, for example `anthropic/claude-sonnet-4-6`. Google models keep their existing route. diff --git a/src/eval/model-access.test.ts b/src/eval/model-access.test.ts index 4d79dcca2b..34b90c687f 100644 --- a/src/eval/model-access.test.ts +++ b/src/eval/model-access.test.ts @@ -267,6 +267,9 @@ describe("eval/model-access", () => { "/ai/v1/chat/completions", "/ai/v1/responses", "/ai/v1/embeddings", + "/ai/v1/messages", + "/ai/v1/messages/count_tokens", + // The earlier Anthropic-protocol prefix, still carried by recorded failures. "/ai/anthropic/v1/messages", ] ) { diff --git a/src/eval/model-access.ts b/src/eval/model-access.ts index 15d6ff1b06..8a27deb0ae 100644 --- a/src/eval/model-access.ts +++ b/src/eval/model-access.ts @@ -74,7 +74,8 @@ function statusDenial(status: number): EvalModelAccessDenial | undefined { /** * Model gateway routes under a Veryfront API base URL: the vendor-neutral - * OpenAI- and Anthropic-protocol routes, and the vendor-scoped route. + * route, the vendor-scoped route, and the earlier Anthropic-protocol prefix, + * which recorded failures from previous releases can still carry. */ const GATEWAY_ROUTE_PREFIXES: readonly string[] = ["/ai/v1/", "/ai/anthropic/v1/", "/ai/gateway/"]; diff --git a/src/provider/veryfront-cloud/gateway-routing.test.ts b/src/provider/veryfront-cloud/gateway-routing.test.ts index 3d155203fc..5c997e3d08 100644 --- a/src/provider/veryfront-cloud/gateway-routing.test.ts +++ b/src/provider/veryfront-cloud/gateway-routing.test.ts @@ -57,28 +57,28 @@ describe("provider/veryfront-cloud gateway routing", () => { { model: "anthropic/claude-opus-4-8", provider: "anthropic", - gatewayBaseUrl: "https://api.veryfront.com/ai/anthropic/v1", + gatewayBaseUrl: "https://api.veryfront.com/ai/v1", genAiSystem: "anthropic", toolProfile: "anthropic", }, { model: "anthropic/claude-opus-4-6", provider: "anthropic", - gatewayBaseUrl: "https://api.veryfront.com/ai/anthropic/v1", + gatewayBaseUrl: "https://api.veryfront.com/ai/v1", genAiSystem: "anthropic", toolProfile: "anthropic", }, { model: "anthropic/claude-sonnet-4-6", provider: "anthropic", - gatewayBaseUrl: "https://api.veryfront.com/ai/anthropic/v1", + gatewayBaseUrl: "https://api.veryfront.com/ai/v1", genAiSystem: "anthropic", toolProfile: "anthropic", }, { model: "anthropic/claude-haiku-4-5-20251001", provider: "anthropic", - gatewayBaseUrl: "https://api.veryfront.com/ai/anthropic/v1", + gatewayBaseUrl: "https://api.veryfront.com/ai/v1", genAiSystem: "anthropic", toolProfile: "anthropic", }, @@ -180,7 +180,7 @@ describe("provider/veryfront-cloud gateway routing", () => { return [alias, getVeryfrontCloudGatewayBaseUrl(API_BASE_URL, provider)]; }), [ - ["anthropic", "https://api.veryfront.com/ai/anthropic/v1"], + ["anthropic", "https://api.veryfront.com/ai/v1"], ["openai", "https://api.veryfront.com/ai/v1"], ["google", "https://api.veryfront.com/ai/gateway/google/v1beta"], ["google-ai-studio", "https://api.veryfront.com/ai/gateway/google/v1beta"], @@ -197,7 +197,7 @@ describe("provider/veryfront-cloud gateway routing", () => { ), [ { - baseURL: "https://api.veryfront.com/ai/anthropic/v1", + baseURL: "https://api.veryfront.com/ai/v1", wireModelProvider: "anthropic", }, { baseURL: "https://api.veryfront.com/ai/v1", wireModelProvider: "openai" }, @@ -212,7 +212,7 @@ describe("provider/veryfront-cloud gateway routing", () => { it("keeps an API base path in front of the vendor-neutral path", () => { assertEquals( getVeryfrontCloudGatewayBaseUrl("https://gateway.example/api/", "anthropic"), - "https://gateway.example/api/ai/anthropic/v1", + "https://gateway.example/api/ai/v1", ); assertEquals( getVeryfrontCloudGatewayBaseUrl("https://gateway.example/api", "openai"), diff --git a/src/provider/veryfront-cloud/provider.test.ts b/src/provider/veryfront-cloud/provider.test.ts index 1c12e697d8..2de2c0bbc0 100644 --- a/src/provider/veryfront-cloud/provider.test.ts +++ b/src/provider/veryfront-cloud/provider.test.ts @@ -1065,7 +1065,7 @@ describe("provider/veryfront-cloud", () => { await drainStream(stream); assertEquals( - capturedRequest?.url.startsWith("https://api.veryfront.com/ai/anthropic/v1"), + capturedRequest?.url.startsWith("https://api.veryfront.com/ai/v1"), true, "the anthropic runtime must be pointed at the Veryfront Cloud anthropic gateway", ); @@ -1278,7 +1278,7 @@ describe("provider/veryfront-cloud", () => { assertEquals(routes, [ [ "anthropic/claude-sonnet-4-6", - "https://api.veryfront.com/ai/anthropic/v1/messages", + "https://api.veryfront.com/ai/v1/messages", "anthropic", ], [ @@ -1513,7 +1513,7 @@ describe("provider/veryfront-cloud vendor-neutral routes", () => { const PROTOCOL_CASES: ReadonlyArray = [ [ "anthropic/claude-sonnet-4-6", - "https://api.veryfront.com/ai/anthropic/v1/messages", + "https://api.veryfront.com/ai/v1/messages", "https://api.veryfront.com/ai/gateway/anthropic/v1/messages", "claude-sonnet-4-6", ], @@ -1568,7 +1568,7 @@ describe("provider/veryfront-cloud vendor-neutral routes", () => { await withEnv({ [VERYFRONT_CLOUD_GATEWAY_ROUTES_ENV]: "neutral" }, () => { assertEquals( getVeryfrontCloudGatewayBaseUrl("https://api.veryfront.com", "anthropic"), - "https://api.veryfront.com/ai/anthropic/v1", + "https://api.veryfront.com/ai/v1", ); return Promise.resolve(); }); diff --git a/src/provider/veryfront-cloud/shared.ts b/src/provider/veryfront-cloud/shared.ts index 15a53b8366..8bcc73165a 100644 --- a/src/provider/veryfront-cloud/shared.ts +++ b/src/provider/veryfront-cloud/shared.ts @@ -369,7 +369,7 @@ export const VERYFRONT_CLOUD_GATEWAY_ROUTES_ENV = "VERYFRONT_CLOUD_GATEWAY_ROUTE */ const NEUTRAL_GATEWAY_PATHS_BY_PROTOCOL: ReadonlyMap = new Map([ ["openai", "ai/v1"], - ["anthropic", "ai/anthropic/v1"], + ["anthropic", "ai/v1"], ]); /** Where a Veryfront Cloud model's requests go, and how the body names the model. */ @@ -407,8 +407,8 @@ function getVeryfrontCloudVendorGatewayBaseUrl( } /** - * Gateway route for a provider. OpenAI-protocol providers use `/ai/v1`, - * Anthropic-protocol providers use `/ai/anthropic/v1`, and Google keeps + * Gateway route for a provider. OpenAI- and Anthropic-protocol providers both + * use `/ai/v1` (`/chat/completions`, `/responses` or `/messages`), and Google keeps * its vendor-scoped path. {@link VERYFRONT_CLOUD_GATEWAY_ROUTES_ENV} set to * `vendor` restores the vendor-scoped path for every provider. */