Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
48 changes: 48 additions & 0 deletions src/agent/runtime/model-resolution.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -10,6 +10,7 @@ import { deleteEnv, setEnv } from "#veryfront/compat/process.ts";
import { afterEach, describe, it } from "#veryfront/testing/bdd.ts";
import {
resolveVeryfrontCloudModelId,
VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES,
VERYFRONT_CLOUD_CHAT_MODELS,
} from "#veryfront/provider/veryfront-cloud/model-catalog.ts";
import {
Expand Down Expand Up @@ -394,6 +395,53 @@ describe("agent/runtime/model-resolution", () => {
);
});

it("routes every catalog provider through veryfront-cloud when only hosted bootstrap is available", () => {
setEnv("VERYFRONT_API_TOKEN", "vf_test_runtime");
setEnv("VERYFRONT_PROJECT_SLUG", "demo-project");

for (const provider of VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES) {
// Mistral model IDs are gated by the catalog, so it needs a listed one.
const modelId = provider === "mistral" ? "mistral-small-2503" : "model-x";
assertEquals(
resolveRuntimeModel(`${provider}/${modelId}`),
`veryfront-cloud/${provider}/${modelId}`,
provider,
);
}
});

it("routes every vendor the gateway catalog serves through veryfront-cloud (#1913)", () => {
// The vendors GET /ai/models lists. Qwen is served before the catalog
// snapshot in this package names it; a vendor missing here fails hosted
// runs with `Model provider "<vendor>" not registered`.
setEnv("VERYFRONT_API_TOKEN", "vf_test_runtime");
setEnv("VERYFRONT_PROJECT_SLUG", "demo-project");

assertEquals(
[
"anthropic/claude-sonnet-4-6",
"openai/gpt-5-nano",
"google/gemini-3.5-flash",
"mistral/mistral-small-2503",
"deepseek/deepseek-v4-flash",
"qwen/qwen3.8-27b",
].map((model) => resolveRuntimeModel(model)),
[
"veryfront-cloud/anthropic/claude-sonnet-4-6",
"veryfront-cloud/openai/gpt-5-nano",
"veryfront-cloud/google/gemini-3.5-flash",
"veryfront-cloud/mistral/mistral-small-2503",
"veryfront-cloud/deepseek/deepseek-v4-flash",
"veryfront-cloud/qwen/qwen3.8-27b",
],
);
});

it("keeps a gateway-only provider unrouted without hosted bootstrap", () => {
setEnv("OPENAI_API_KEY", "sk-test");
assertEquals(resolveRuntimeModel("qwen/qwen3.8-27b"), "qwen/qwen3.8-27b");
});

it("routes catalog Gemini, Mistral, and Kimi models through veryfront-cloud when only hosted bootstrap is available", () => {
setEnv("VERYFRONT_API_TOKEN", "vf_test_runtime");
setEnv("VERYFRONT_PROJECT_SLUG", "demo-project");
Expand Down
19 changes: 11 additions & 8 deletions src/agent/runtime/model-resolution.ts
Original file line number Diff line number Diff line change
Expand Up @@ -8,6 +8,7 @@ import {
createRetiredVeryfrontCloudModelError,
findVeryfrontCloudModelByModelId,
isRetiredVeryfrontCloudModelId,
VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES,
} from "#veryfront/provider/veryfront-cloud/model-catalog.ts";
import { DEFAULT_MODEL_CREDENTIAL_MISMATCH, NOT_SUPPORTED } from "#veryfront/errors";
import {
Expand All @@ -21,14 +22,16 @@ import { getModelRuntimeProvider } from "#veryfront/provider/runtime-inspection.
export const AUTO_AGENT_MODEL = "auto";
export const DEFAULT_AGENT_MODEL = "openai/gpt-5-nano";

const HOSTED_PROVIDER_NAMES = new Set([
"deepseek",
"anthropic",
"google",
"google-ai-studio",
"mistral",
"moonshotai",
"openai",
/**
* Providers the gateway serves that the catalog snapshot shipped in this
* package does not list yet. Each routes on the default surface, which the
* gateway serves at the vendor-neutral `/ai/v1`. Drop an entry once
* `deno task generate:model-catalog` adds its provider to the snapshot.
*/
const GATEWAY_PROVIDERS_AHEAD_OF_CATALOG = ["qwen"] as const;
const HOSTED_PROVIDER_NAMES: ReadonlySet<string> = new Set([
...VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES,
...GATEWAY_PROVIDERS_AHEAD_OF_CATALOG,
]);
const DIRECT_CREDENTIAL_PROVIDER_ALIASES = new Map<string, string>([
["google-ai-studio", "google"],
Expand Down
14 changes: 14 additions & 0 deletions src/provider/veryfront-cloud/model-catalog.ts
Original file line number Diff line number Diff line change
Expand Up @@ -415,6 +415,20 @@ export const VERYFRONT_CLOUD_CHAT_MODELS: readonly VeryfrontCloudChatModel[] = O
}),
);

/**
* Every provider name the catalog data routes: accepted aliases, providers
* with a routing row, and providers of a listed chat model. Runtime model
* resolution sends `<provider>/<model>` for these through the gateway when no
* direct provider credential applies.
*/
export const VERYFRONT_CLOUD_CATALOG_PROVIDER_NAMES: readonly string[] = Object.freeze([
...new Set<string>([
...VERYFRONT_CLOUD_PROVIDER_ALIASES.map(([alias]) => alias),
...VERYFRONT_CLOUD_PROVIDER_ROUTING.map(([provider]) => provider),
...VERYFRONT_CLOUD_CHAT_MODELS.map((model) => model.provider),
]),
]);

const defaultVeryfrontCloudChatModel = VERYFRONT_CLOUD_CHAT_MODELS.find(
(model) => model.id === DEFAULT_VERYFRONT_CLOUD_MODEL_ID,
);
Expand Down
38 changes: 38 additions & 0 deletions src/provider/veryfront-cloud/provider.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -1330,6 +1330,44 @@ describe("provider/veryfront-cloud", () => {
assertEquals(result.text, "Hello");
});

it("sends a hosted qwen/qwen3.8-27b run to the vendor-neutral gateway route (#1913)", async () => {
setCloudBootstrap();
const encoder = new TextEncoder();
let captured: { url: string; model: unknown } | undefined;

installMockFetch(
(async (input: URL | Request | string, init?: RequestInit) => {
const request = new Request(input, init);
captured = { url: request.url, model: JSON.parse(await request.text()).model };

return new Response(
new ReadableStream({
start(controller) {
controller.enqueue(
encoder.encode('data: {"choices":[{"delta":{"content":"Hello"}}]}\n\n'),
);
controller.enqueue(
encoder.encode('data: {"choices":[{"finish_reason":"stop"}]}\n\n'),
);
controller.enqueue(encoder.encode("data: [DONE]\n\n"));
controller.close();
},
}),
{ status: 200, headers: { "content-type": "text/event-stream" } },
);
}) as typeof fetch,
);

const assistant = agent({ model: "qwen/qwen3.8-27b", system: "You are concise." });
const result = await assistant.generate({ input: "Hi" });

assertEquals(captured, {
url: "https://api.veryfront.com/ai/v1/chat/completions",
model: "qwen/qwen3.8-27b",
});
assertEquals(result.text, "Hello");
});

it("keeps an unlisted provider on chat completions for a reasoning-style model id", async () => {
// "gpt-5.4" is a reasoning-style ID. Only the provider that implements the
// OpenAI surface natively serves /responses, so an unlisted provider must
Expand Down
Loading