From cc53db4406e5a82422ab7e7f409b39e30e3eb37a Mon Sep 17 00:00:00 2001 From: Shoubhit Dash Date: Mon, 17 Aug 2026 17:56:03 +0530 Subject: [PATCH] feat(core): discover vLLM models (#43022) --- packages/core/src/plugin/provider.ts | 2 + packages/core/src/plugin/provider/vllm.ts | 162 ++++++++++ .../core/test/plugin/provider-vllm.test.ts | 289 ++++++++++++++++++ .../www/content/docs/(Configure)/models.mdx | 37 +++ 4 files changed, 490 insertions(+) create mode 100644 packages/core/src/plugin/provider/vllm.ts create mode 100644 packages/core/test/plugin/provider-vllm.test.ts diff --git a/packages/core/src/plugin/provider.ts b/packages/core/src/plugin/provider.ts index 9097ea299cb..18a3cddfdec 100644 --- a/packages/core/src/plugin/provider.ts +++ b/packages/core/src/plugin/provider.ts @@ -29,6 +29,7 @@ import { SapAICorePlugin } from "./provider/sap-ai-core.js" import { TogetherAIPlugin } from "./provider/togetherai.js" import { VercelPlugin } from "./provider/vercel.js" import { VenicePlugin } from "./provider/venice.js" +import { VLLMPlugin } from "./provider/vllm.js" import { XAIPlugin } from "./provider/xai.js" import { ZenmuxPlugin } from "./provider/zenmux.js" import type { PluginInternal } from "./internal.js" @@ -64,6 +65,7 @@ export const ProviderPlugins: PluginInternal.InternalPlugin[] = [ TogetherAIPlugin, VercelPlugin, VenicePlugin, + VLLMPlugin, XAIPlugin, ZenmuxPlugin, DynamicProviderPlugin, diff --git a/packages/core/src/plugin/provider/vllm.ts b/packages/core/src/plugin/provider/vllm.ts new file mode 100644 index 00000000000..0950f4e0b0c --- /dev/null +++ b/packages/core/src/plugin/provider/vllm.ts @@ -0,0 +1,162 @@ +import { define } from "@opencode-ai/plugin/effect/plugin" +import { Document, type Entry } from "@opencode-ai/schema/config" +import { Duration, Effect, Schedule, Schema, Semaphore, Stream } from "effect" +import { HttpClient, HttpClientRequest, HttpClientResponse } from "effect/unstable/http" +import { Config } from "../../config.js" +import { Model } from "../../model.js" +import { Provider } from "../../provider.js" +import type { PluginInternal } from "../internal.js" + +const providerID = "vllm" + +const RemoteModel = Schema.Struct({ + id: Schema.String, + owned_by: Schema.String, + max_model_len: Schema.NullOr(Schema.Int), +}) + +const Response = Schema.Struct({ data: Schema.Array(RemoteModel) }) +const discovery = new Map() +const discoveryLock = Semaphore.makeUnsafe(1) + +export function make(origin = "http://127.0.0.1:8000", interval: Duration.Input = "30 seconds") { + return define({ + id: "opencode.provider.vllm", + effect: Effect.fn(function* (ctx) { + const http = HttpClient.filterStatusOk(yield* HttpClient.HttpClient) + const config = yield* Config.Service + const source = { current: configured(yield* config.entries(), origin) } + const loaded = { models: [] as (typeof RemoteModel.Type)[], hash: "[]" } + + yield* ctx.integration.transform((integrations) => { + if (loaded.models.length === 0) return + integrations.remove(providerID) + }) + + yield* ctx.catalog.transform((catalog) => { + if (loaded.models.length === 0) return + for (const model of catalog.provider.get(providerID)?.models.values() ?? []) { + catalog.model.remove(providerID, model.id) + } + catalog.provider.update(providerID, (provider) => { + provider.name = "vLLM" + provider.package = "@opencode-ai/ai/providers/openai-compatible" + provider.settings = { + baseURL: source.current.baseURL, + provider: providerID, + apiKey: source.current.apiKey ?? "", + } + provider.integrationID = undefined + provider.activation = "enabled" + }) + for (const item of loaded.models) { + catalog.model.update(providerID, item.id, (model) => { + model.modelID = Model.ID.make(item.id) + model.name = item.id + // Tool calling depends on vLLM server flags and parsers that model discovery does not report. + model.capabilities = { tools: false, input: ["text"], output: ["text"] } + model.limit = { context: item.max_model_len ?? 0, output: 0 } + }) + } + }) + + const discover = Effect.fn("VLLMPlugin.discover")(function* () { + const current = source.current + if (!current.healthEndpoint || !current.modelsEndpoint) return undefined + return yield* discoveryLock.withPermit( + Effect.gen(function* () { + const endpoint = `${current.healthEndpoint}\n${current.modelsEndpoint}` + const cached = discovery.get(endpoint) + if (cached && cached.apiKey === current.apiKey && Date.now() - cached.checked < Duration.toMillis(interval)) + return { source: current, models: cached.models } + discovery.set(endpoint, { + checked: Date.now(), + apiKey: current.apiKey, + models: cached && cached.apiKey === current.apiKey ? cached.models : undefined, + }) + const request = (endpoint: string) => + current.apiKey + ? HttpClientRequest.get(endpoint).pipe( + HttpClientRequest.acceptJson, + HttpClientRequest.bearerToken(current.apiKey), + ) + : HttpClientRequest.get(endpoint).pipe(HttpClientRequest.acceptJson) + yield* http.execute(request(current.healthEndpoint)).pipe(Effect.timeout("1 second")) + const response = yield* http + .execute(request(current.modelsEndpoint)) + .pipe(Effect.flatMap(HttpClientResponse.schemaBodyJson(Response)), Effect.timeout("1 second")) + const models = response.data + .filter((model) => model.owned_by === providerID && model.id.length > 0) + .toSorted((a, b) => a.id.localeCompare(b.id)) + discovery.set(endpoint, { checked: Date.now(), apiKey: current.apiKey, models }) + return { source: current, models } + }), + ) + }) + + const refresh = Effect.fn("VLLMPlugin.refresh")(function* () { + const result = yield* discover() + if (!result?.models || result.source !== source.current) return + const hash = JSON.stringify(result.models) + if (hash === loaded.hash) return + loaded.models = result.models + loaded.hash = hash + yield* ctx.integration.reload() + yield* ctx.catalog.reload() + }) + + // Keep the last successful inventory through transient outages instead of flickering model availability. + yield* refresh().pipe(Effect.ignore, Effect.repeat(Schedule.spaced(interval)), Effect.forkScoped) + const reload = Effect.fn("VLLMPlugin.reload")(function* () { + const next = configured(yield* config.entries(), origin) + if ( + next.baseURL === source.current.baseURL && + next.apiKey === source.current.apiKey && + next.healthEndpoint === source.current.healthEndpoint && + next.modelsEndpoint === source.current.modelsEndpoint + ) + return + source.current = next + loaded.models = [] + loaded.hash = "[]" + yield* ctx.integration.reload() + yield* ctx.catalog.reload() + yield* refresh().pipe(Effect.ignore) + }) + yield* ctx.event.subscribe().pipe( + Stream.filter((event) => event.type === "config.updated"), + Stream.runForEach(reload), + Effect.forkScoped({ startImmediately: true }), + ) + }), + } satisfies PluginInternal.InternalPlugin) +} + +export const VLLMPlugin = make() + +function configured(entries: readonly Entry[], origin: string) { + const settings = entries + .filter((entry): entry is Document => entry.type === "document") + .flatMap((entry) => { + const settings = entry.info.providers?.[providerID]?.settings + return settings ? [settings] : [] + }) + .reduce((result, item) => Provider.mergeOverlay(result, item), undefined) + const baseURL = ( + typeof settings?.baseURL === "string" ? settings.baseURL : `${origin.replace(/\/+$/, "")}/v1` + ).replace(/\/+$/, "") + const apiKey = typeof settings?.apiKey === "string" ? settings.apiKey : undefined + if (!URL.canParse(baseURL)) return { baseURL, apiKey } + const models = new URL(baseURL) + if (models.protocol !== "http:" && models.protocol !== "https:") return { baseURL, apiKey } + models.pathname = `${models.pathname.replace(/\/+$/, "")}/models` + models.search = "" + models.hash = "" + const health = new URL(baseURL) + const path = health.pathname.replace(/\/+$/, "") + const prefix = path.endsWith("/v1") ? path.slice(0, -3) : path + health.pathname = `${prefix}/health` + health.search = "" + health.hash = "" + return { baseURL, apiKey, healthEndpoint: health.toString(), modelsEndpoint: models.toString() } +} diff --git a/packages/core/test/plugin/provider-vllm.test.ts b/packages/core/test/plugin/provider-vllm.test.ts new file mode 100644 index 00000000000..fb70390360c --- /dev/null +++ b/packages/core/test/plugin/provider-vllm.test.ts @@ -0,0 +1,289 @@ +import { Bus } from "@opencode-ai/core/bus" +import { Catalog } from "@opencode-ai/core/catalog" +import { Config } from "@opencode-ai/core/config" +import { Integration } from "@opencode-ai/core/integration" +import { Model } from "@opencode-ai/core/model" +import { Plugin } from "@opencode-ai/core/plugin" +import { PluginHost } from "@opencode-ai/core/plugin/host" +import { ProviderPlugins } from "@opencode-ai/core/plugin/provider" +import { make, VLLMPlugin } from "@opencode-ai/core/plugin/provider/vllm" +import { Provider } from "@opencode-ai/core/provider" +import { Document, Event, Info } from "@opencode-ai/schema/config" +import { describe, expect } from "bun:test" +import { Duration, Effect, Layer, Schema } from "effect" +import { testEffect } from "../lib/effect" +import { PluginTestLayer } from "./fixture" + +const it = testEffect(Layer.merge(PluginTestLayer, Config.testLayer())) +const decode = Schema.decodeUnknownSync(Info) + +const addPlugin = Effect.fn(function* (origin: string, interval: Duration.Input = "1 hour") { + const plugin = yield* Plugin.Service + const host = yield* PluginHost.make(plugin) + yield* make(origin, interval).effect(host) +}) + +function eventually( + effect: Effect.Effect, + predicate: (value: A) => boolean, + remaining = 3000, +): Effect.Effect { + return Effect.gen(function* () { + const value = yield* effect + if (predicate(value)) return value + if (remaining === 0) return yield* Effect.fail(new Error("Timed out waiting for value")) + yield* Effect.promise(() => Bun.sleep(1)) + return yield* eventually(effect, predicate, remaining - 1) + }) +} + +const remoteModel = (id: string, max_model_len = 32_768, owned_by = "vllm") => ({ + id, + object: "model", + created: 1, + owned_by, + root: id, + parent: null, + max_model_len, + permission: [], +}) + +describe("VLLMPlugin", () => { + it.live("waits for readiness and discovers official vLLM model metadata", () => + Effect.acquireUseRelease( + Effect.sync(() => { + const state = { healthy: false, models: 0 } + return { + state, + server: Bun.serve({ + port: 0, + fetch: (request) => { + const path = new URL(request.url).pathname + if (path === "/health") return new Response(null, { status: state.healthy ? 200 : 503 }) + state.models++ + return Response.json({ + object: "list", + data: [remoteModel("Qwen/Qwen3-Coder", 65_536), remoteModel("foreign-model", 4096, "other")], + }) + }, + }), + } + }), + ({ state, server }) => + Effect.gen(function* () { + const catalog = yield* Catalog.Service + const providerID = Provider.ID.make("vllm") + expect(VLLMPlugin.id).toBe("opencode.provider.vllm") + expect(ProviderPlugins.map((item) => item.id)).toContain("opencode.provider.vllm") + yield* addPlugin(server.url.origin, "5 millis") + yield* Effect.promise(() => Bun.sleep(20)) + expect(yield* catalog.provider.get(providerID)).toBeUndefined() + expect(state.models).toBe(0) + + state.healthy = true + const model = yield* eventually( + catalog.model.get(providerID, Model.ID.make("Qwen/Qwen3-Coder")), + (item) => item !== undefined, + ) + expect(yield* catalog.provider.get(providerID)).toEqual({ + id: providerID, + name: "vLLM", + package: "@opencode-ai/ai/providers/openai-compatible", + settings: { baseURL: `${server.url.origin}/v1`, provider: "vllm", apiKey: "" }, + activation: "enabled", + }) + expect((yield* catalog.provider.available()).map((provider) => provider.id)).toContain(providerID) + expect(model).toMatchObject({ + modelID: "Qwen/Qwen3-Coder", + name: "Qwen/Qwen3-Coder", + capabilities: { tools: false, input: ["text"], output: ["text"] }, + limit: { context: 65_536, output: 0 }, + }) + expect(yield* catalog.model.get(providerID, Model.ID.make("foreign-model"))).toBeUndefined() + }), + ({ server }) => Effect.promise(() => server.stop(true)), + ), + ) + + it.live("refreshes inventory while retaining the last success through transient failures", () => + Effect.acquireUseRelease( + Effect.sync(() => { + const state = { failing: false, models: [remoteModel("first-model")] } + return { + state, + server: Bun.serve({ + port: 0, + fetch: (request) => { + if (state.failing) return new Response(null, { status: 503 }) + if (new URL(request.url).pathname === "/health") return new Response() + return Response.json({ object: "list", data: state.models }) + }, + }), + } + }), + ({ state, server }) => + Effect.gen(function* () { + const catalog = yield* Catalog.Service + const providerID = Provider.ID.make("vllm") + yield* addPlugin(server.url.origin, "5 millis") + yield* eventually(catalog.model.get(providerID, Model.ID.make("first-model")), (model) => model !== undefined) + + state.failing = true + state.models = [remoteModel("second-model")] + yield* Effect.promise(() => Bun.sleep(30)) + expect(yield* catalog.model.get(providerID, Model.ID.make("first-model"))).toBeDefined() + expect(yield* catalog.model.get(providerID, Model.ID.make("second-model"))).toBeUndefined() + + state.failing = false + yield* eventually( + catalog.model.get(providerID, Model.ID.make("second-model")), + (model) => model !== undefined, + ) + expect(yield* catalog.model.get(providerID, Model.ID.make("first-model"))).toBeUndefined() + }), + ({ server }) => Effect.promise(() => server.stop(true)), + ), + ) + + it.live("replaces and restores same-ID Models.dev entries after an empty success", () => + Effect.acquireUseRelease( + Effect.sync(() => { + const models = [remoteModel("discovered-model")] + return { + models, + server: Bun.serve({ + port: 0, + fetch: (request) => + new URL(request.url).pathname === "/health" + ? new Response() + : Response.json({ object: "list", data: models }), + }), + } + }), + ({ models, server }) => + Effect.gen(function* () { + const catalog = yield* Catalog.Service + const integrations = yield* Integration.Service + const providerID = Provider.ID.make("vllm") + yield* integrations.transform((draft) => { + draft.update(Integration.ID.make("vllm"), (integration) => { + integration.name = "vLLM" + }) + draft.method.update({ + integrationID: Integration.ID.make("vllm"), + method: { type: "env", names: ["VLLM_API_KEY"] }, + }) + }) + yield* catalog.transform((draft) => { + draft.provider.update(providerID, (provider) => { + provider.name = "vLLM" + provider.package = "aisdk:@ai-sdk/openai-compatible" + provider.integrationID = Integration.ID.make("vllm") + provider.activation = "auto" + }) + draft.model.update(providerID, Model.ID.make("static-model"), () => {}) + }) + + yield* addPlugin(server.url.origin, "5 millis") + yield* eventually( + catalog.model.get(providerID, Model.ID.make("discovered-model")), + (model) => model !== undefined, + ) + expect(yield* integrations.get(Integration.ID.make("vllm"))).toBeUndefined() + expect((yield* catalog.provider.get(providerID))?.integrationID).toBeUndefined() + expect((yield* catalog.provider.get(providerID))?.activation).toBe("enabled") + expect(yield* catalog.model.get(providerID, Model.ID.make("static-model"))).toBeUndefined() + + models.splice(0) + yield* eventually( + catalog.model.get(providerID, Model.ID.make("static-model")), + (model) => model !== undefined, + ) + expect(yield* catalog.model.get(providerID, Model.ID.make("discovered-model"))).toBeUndefined() + expect(yield* integrations.get(Integration.ID.make("vllm"))).toBeDefined() + expect((yield* catalog.provider.get(providerID))?.integrationID).toBe(Integration.ID.make("vllm")) + expect((yield* catalog.provider.get(providerID))?.activation).toBe("auto") + }), + ({ server }) => Effect.promise(() => server.stop(true)), + ), + ) + + it.live( + "reloads layered custom endpoint and bearer authentication settings", + () => + Effect.acquireUseRelease( + Effect.sync(() => { + const requests: Array<{ authorization: string | null; path: string }> = [] + return { + requests, + initial: Bun.serve({ + port: 0, + fetch: (request) => + new URL(request.url).pathname === "/health" + ? new Response() + : Response.json({ object: "list", data: [remoteModel("initial-model")] }), + }), + configured: Bun.serve({ + port: 0, + fetch: (request) => { + requests.push({ + authorization: request.headers.get("authorization"), + path: new URL(request.url).pathname, + }) + if (new URL(request.url).pathname === "/proxy/health") return new Response() + return Response.json({ object: "list", data: [remoteModel("configured-model")] }) + }, + }), + } + }), + ({ requests, initial, configured }) => + Effect.gen(function* () { + const bus = yield* Bus.Service + const catalog = yield* Catalog.Service + const config = yield* Config.Test + const providerID = Provider.ID.make("vllm") + yield* addPlugin(initial.url.origin) + yield* eventually( + catalog.model.get(providerID, Model.ID.make("initial-model")), + (model) => model !== undefined, + ) + + const baseURL = `${configured.url.origin}/proxy/v1` + yield* config.setEntries([configuration({ baseURL }), configuration({ apiKey: "secret" })]) + yield* bus.publish(Event.Updated, {}) + yield* eventually( + catalog.model.get(providerID, Model.ID.make("configured-model")), + (model) => model !== undefined, + ) + + expect(requests).toContainEqual({ authorization: "Bearer secret", path: "/proxy/health" }) + expect(requests).toContainEqual({ authorization: "Bearer secret", path: "/proxy/v1/models" }) + expect(yield* catalog.model.get(providerID, Model.ID.make("initial-model"))).toBeUndefined() + expect((yield* catalog.provider.get(providerID))?.settings).toEqual({ + baseURL, + provider: "vllm", + apiKey: "secret", + }) + + requests.splice(0) + yield* config.setEntries([configuration({ baseURL }), configuration({ apiKey: "next-secret" })]) + yield* bus.publish(Event.Updated, {}) + yield* eventually( + catalog.provider.get(providerID), + (provider) => provider?.settings?.apiKey === "next-secret", + ) + expect(requests).toContainEqual({ authorization: "Bearer next-secret", path: "/proxy/health" }) + expect(requests).toContainEqual({ authorization: "Bearer next-secret", path: "/proxy/v1/models" }) + }), + ({ initial, configured }) => Effect.promise(() => Promise.all([initial.stop(true), configured.stop(true)])), + ), + 10_000, + ) +}) + +function configuration(settings: { baseURL?: string; apiKey?: string }) { + return new Document({ + type: "document", + info: decode({ providers: { vllm: { settings } } }), + }) +} diff --git a/packages/www/content/docs/(Configure)/models.mdx b/packages/www/content/docs/(Configure)/models.mdx index 94cc0d7e0bf..1ecb2f90a39 100644 --- a/packages/www/content/docs/(Configure)/models.mdx +++ b/packages/www/content/docs/(Configure)/models.mdx @@ -221,6 +221,43 @@ For a different host or port, configure the OpenAI-compatible base URL. Models a Omit `apiKey` when LM Studio authentication is disabled. +#### vLLM + +OpenCode automatically discovers models from a vLLM server listening on its default address, `http://127.0.0.1:8000`. +Discovered models use the `vllm` provider ID and the model ID reported by vLLM: + +```jsonc title="opencode.jsonc" +{ + "$schema": "https://opencode.ai/config.json", + "model": "vllm/Qwen/Qwen3-Coder-30B-A3B-Instruct", +} +``` + +OpenCode checks vLLM's `/health` endpoint and refreshes `/v1/models` in the background. It uses the reported +`max_model_len` as the context limit and only includes model cards owned by `vllm`. Discovered vLLM models advertise +text input and output, but not vision or tools. Tool calling is conservative because vLLM enables it with server-level +flags such as `--enable-auto-tool-choice` and `--tool-call-parser`, which model discovery does not report. Disable +discovery with `"plugins": ["-opencode.provider.vllm"]`. + +For a different endpoint or an authenticated server, configure its OpenAI-compatible base URL: + +```jsonc title="opencode.jsonc" +{ + "$schema": "https://opencode.ai/config.json", + "providers": { + "vllm": { + "settings": { + "baseURL": "http://127.0.0.1:9000/v1", + "apiKey": "{env:VLLM_API_KEY}", + }, + }, + }, +} +``` + +Omit `apiKey` when authentication is disabled. Path-prefixed proxy URLs are supported; for example, +`https://example.com/vllm/v1` checks `/vllm/health` and discovers `/vllm/v1/models`. + For an OpenAI-compatible server, define a provider package, endpoint, and at least one model: ```jsonc title="opencode.jsonc"