Vendor OpenClaw source as Adolf fork baseline
Some checks failed
ClawSweeper Dispatch / dispatch (push) Has been cancelled
CodeQL / Security High (actions) (push) Has been cancelled
CodeQL / Security High (channel-runtime-boundary) (push) Has been cancelled
CodeQL / Security High (core-auth-secrets) (push) Has been cancelled
CodeQL / Security High (mcp-process-tool-boundary) (push) Has been cancelled
CodeQL / Security High (network-ssrf-boundary) (push) Has been cancelled
CodeQL / Security High (plugin-trust-boundary) (push) Has been cancelled
CodeQL / Security High (process-exec-boundary) (push) Has been cancelled
Docs Sync Publish Repo / sync-publish-repo (push) Has been cancelled
Docs / docs (push) Has been cancelled
OpenClaw Stable Main Closeout / Resolve stable release closeout inputs (push) Has been cancelled
OpenClaw Stable Main Closeout / Verify stable main closeout (push) Has been cancelled
Workflow Sanity / no-tabs (push) Has been cancelled
Workflow Sanity / actionlint (push) Has been cancelled
Workflow Sanity / generated-doc-baselines (push) Has been cancelled
CI / runner-admission (push) Has been cancelled
CI / preflight (push) Has been cancelled
CI / security-fast (push) Has been cancelled
CI / pnpm-store-warmup (push) Has been cancelled
CI / build-artifacts (push) Has been cancelled
CI / native-i18n (push) Has been cancelled
CI / ${{ matrix.check_name }} (push) Has been cancelled
CI / ${{ matrix.checkName }} (push) Has been cancelled
CI / checks-node-compat-node22 (push) Has been cancelled
CI / check-bundled-channel-config-metadata (push) Has been cancelled
CI / check-dependencies (push) Has been cancelled
CI / check-guards (push) Has been cancelled
CI / check-lint (push) Has been cancelled
CI / check-prod-types (push) Has been cancelled
CI / check-shrinkwrap (push) Has been cancelled
CI / check-test-types (push) Has been cancelled
CI / check-additional-boundaries-a (push) Has been cancelled
CI / check-additional-boundaries-bcd (push) Has been cancelled
CI / check-additional-extension-bundled (push) Has been cancelled
CI / check-additional-extension-channels (push) Has been cancelled
CI / check-additional-extension-package-boundary (push) Has been cancelled
CI / check-additional-runtime-topology-architecture (push) Has been cancelled
CI / check-session-accessor-boundary (push) Has been cancelled
CI / check-session-transcript-reader-boundary (push) Has been cancelled
CI / check-docs (push) Has been cancelled
CI / skills-python (push) Has been cancelled
CI / macos-swift (push) Has been cancelled
CI / ios-build (push) Has been cancelled
CI / ci-timings-summary (push) Has been cancelled
Native App Locale Refresh / Refresh native fa (push) Has been cancelled
Native App Locale Refresh / Refresh native fr (push) Has been cancelled
Native App Locale Refresh / Refresh native hi (push) Has been cancelled
Native App Locale Refresh / Refresh native id (push) Has been cancelled
Native App Locale Refresh / Refresh native it (push) Has been cancelled
Native App Locale Refresh / Refresh native ja-JP (push) Has been cancelled
Control UI Locale Refresh / plan (push) Has been cancelled
Control UI Locale Refresh / Refresh ${{ matrix.locale }} (push) Has been cancelled
Control UI Locale Refresh / Commit control UI locale refresh (push) Has been cancelled
Live Media Runner Image / Build live media runner image (push) Has been cancelled
Native App Locale Refresh / Refresh native ar (push) Has been cancelled
Native App Locale Refresh / Refresh native de (push) Has been cancelled
Native App Locale Refresh / Refresh native es (push) Has been cancelled
Native App Locale Refresh / Refresh native ko (push) Has been cancelled
Native App Locale Refresh / Refresh native nl (push) Has been cancelled
Native App Locale Refresh / Refresh native pl (push) Has been cancelled
Native App Locale Refresh / Refresh native pt-BR (push) Has been cancelled
Native App Locale Refresh / Refresh native ru (push) Has been cancelled
Native App Locale Refresh / Refresh native sv (push) Has been cancelled
Native App Locale Refresh / Refresh native th (push) Has been cancelled
Native App Locale Refresh / Refresh native tr (push) Has been cancelled
Native App Locale Refresh / Refresh native uk (push) Has been cancelled
Native App Locale Refresh / Refresh native vi (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-CN (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-TW (push) Has been cancelled
Native App Locale Refresh / Commit native locale refresh (push) Has been cancelled
Plugin Init Scaffold Validation / Validate provider scaffold (push) Has been cancelled
Plugin NPM Release / preview_plugins_npm (push) Has been cancelled
Plugin NPM Release / Validate release publish approval (push) Has been cancelled
Plugin NPM Release / preview_plugin_pack (push) Has been cancelled
Plugin NPM Release / publish_plugins_npm (push) Has been cancelled
Sandbox Common Smoke / sandbox-common-smoke (push) Has been cancelled
Website Installer Sync / static (push) Has been cancelled
Website Installer Sync / linux-docker (push) Has been cancelled
Website Installer Sync / macos-installer (push) Has been cancelled
Website Installer Sync / windows-installer (push) Has been cancelled
Website Installer Sync / sync-website (push) Has been cancelled
Some checks failed
ClawSweeper Dispatch / dispatch (push) Has been cancelled
CodeQL / Security High (actions) (push) Has been cancelled
CodeQL / Security High (channel-runtime-boundary) (push) Has been cancelled
CodeQL / Security High (core-auth-secrets) (push) Has been cancelled
CodeQL / Security High (mcp-process-tool-boundary) (push) Has been cancelled
CodeQL / Security High (network-ssrf-boundary) (push) Has been cancelled
CodeQL / Security High (plugin-trust-boundary) (push) Has been cancelled
CodeQL / Security High (process-exec-boundary) (push) Has been cancelled
Docs Sync Publish Repo / sync-publish-repo (push) Has been cancelled
Docs / docs (push) Has been cancelled
OpenClaw Stable Main Closeout / Resolve stable release closeout inputs (push) Has been cancelled
OpenClaw Stable Main Closeout / Verify stable main closeout (push) Has been cancelled
Workflow Sanity / no-tabs (push) Has been cancelled
Workflow Sanity / actionlint (push) Has been cancelled
Workflow Sanity / generated-doc-baselines (push) Has been cancelled
CI / runner-admission (push) Has been cancelled
CI / preflight (push) Has been cancelled
CI / security-fast (push) Has been cancelled
CI / pnpm-store-warmup (push) Has been cancelled
CI / build-artifacts (push) Has been cancelled
CI / native-i18n (push) Has been cancelled
CI / ${{ matrix.check_name }} (push) Has been cancelled
CI / ${{ matrix.checkName }} (push) Has been cancelled
CI / checks-node-compat-node22 (push) Has been cancelled
CI / check-bundled-channel-config-metadata (push) Has been cancelled
CI / check-dependencies (push) Has been cancelled
CI / check-guards (push) Has been cancelled
CI / check-lint (push) Has been cancelled
CI / check-prod-types (push) Has been cancelled
CI / check-shrinkwrap (push) Has been cancelled
CI / check-test-types (push) Has been cancelled
CI / check-additional-boundaries-a (push) Has been cancelled
CI / check-additional-boundaries-bcd (push) Has been cancelled
CI / check-additional-extension-bundled (push) Has been cancelled
CI / check-additional-extension-channels (push) Has been cancelled
CI / check-additional-extension-package-boundary (push) Has been cancelled
CI / check-additional-runtime-topology-architecture (push) Has been cancelled
CI / check-session-accessor-boundary (push) Has been cancelled
CI / check-session-transcript-reader-boundary (push) Has been cancelled
CI / check-docs (push) Has been cancelled
CI / skills-python (push) Has been cancelled
CI / macos-swift (push) Has been cancelled
CI / ios-build (push) Has been cancelled
CI / ci-timings-summary (push) Has been cancelled
Native App Locale Refresh / Refresh native fa (push) Has been cancelled
Native App Locale Refresh / Refresh native fr (push) Has been cancelled
Native App Locale Refresh / Refresh native hi (push) Has been cancelled
Native App Locale Refresh / Refresh native id (push) Has been cancelled
Native App Locale Refresh / Refresh native it (push) Has been cancelled
Native App Locale Refresh / Refresh native ja-JP (push) Has been cancelled
Control UI Locale Refresh / plan (push) Has been cancelled
Control UI Locale Refresh / Refresh ${{ matrix.locale }} (push) Has been cancelled
Control UI Locale Refresh / Commit control UI locale refresh (push) Has been cancelled
Live Media Runner Image / Build live media runner image (push) Has been cancelled
Native App Locale Refresh / Refresh native ar (push) Has been cancelled
Native App Locale Refresh / Refresh native de (push) Has been cancelled
Native App Locale Refresh / Refresh native es (push) Has been cancelled
Native App Locale Refresh / Refresh native ko (push) Has been cancelled
Native App Locale Refresh / Refresh native nl (push) Has been cancelled
Native App Locale Refresh / Refresh native pl (push) Has been cancelled
Native App Locale Refresh / Refresh native pt-BR (push) Has been cancelled
Native App Locale Refresh / Refresh native ru (push) Has been cancelled
Native App Locale Refresh / Refresh native sv (push) Has been cancelled
Native App Locale Refresh / Refresh native th (push) Has been cancelled
Native App Locale Refresh / Refresh native tr (push) Has been cancelled
Native App Locale Refresh / Refresh native uk (push) Has been cancelled
Native App Locale Refresh / Refresh native vi (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-CN (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-TW (push) Has been cancelled
Native App Locale Refresh / Commit native locale refresh (push) Has been cancelled
Plugin Init Scaffold Validation / Validate provider scaffold (push) Has been cancelled
Plugin NPM Release / preview_plugins_npm (push) Has been cancelled
Plugin NPM Release / Validate release publish approval (push) Has been cancelled
Plugin NPM Release / preview_plugin_pack (push) Has been cancelled
Plugin NPM Release / publish_plugins_npm (push) Has been cancelled
Sandbox Common Smoke / sandbox-common-smoke (push) Has been cancelled
Website Installer Sync / static (push) Has been cancelled
Website Installer Sync / linux-docker (push) Has been cancelled
Website Installer Sync / macos-installer (push) Has been cancelled
Website Installer Sync / windows-installer (push) Has been cancelled
Website Installer Sync / sync-website (push) Has been cancelled
Adolf is a fork/vendored clone of github.com/openclaw/openclaw (v2026.6.11), free to diverge. Tree copied sans upstream .git; upstream remote added for future syncs. Node pinned to 24 (.nvmrc); engines already require >=22.19. Preserves docs/ARCHITECTURE.md. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01LeqyaxJF2nbRXJtae2kNB2
This commit is contained in:
3
extensions/vllm/README.md
Normal file
3
extensions/vllm/README.md
Normal file
@@ -0,0 +1,3 @@
|
||||
# vLLM Provider
|
||||
|
||||
Bundled provider plugin for vLLM discovery and setup.
|
||||
9
extensions/vllm/api.ts
Normal file
9
extensions/vllm/api.ts
Normal file
@@ -0,0 +1,9 @@
|
||||
// Vllm API module exposes the plugin public contract.
|
||||
export {
|
||||
VLLM_DEFAULT_API_KEY_ENV_VAR,
|
||||
VLLM_DEFAULT_BASE_URL,
|
||||
VLLM_MODEL_PLACEHOLDER,
|
||||
VLLM_PROVIDER_LABEL,
|
||||
} from "./defaults.js";
|
||||
export { buildVllmProvider } from "./models.js";
|
||||
export { createVllmQwenThinkingWrapper, wrapVllmProviderStream } from "./stream.js";
|
||||
5
extensions/vllm/defaults.ts
Normal file
5
extensions/vllm/defaults.ts
Normal file
@@ -0,0 +1,5 @@
|
||||
// Vllm plugin module implements defaults behavior.
|
||||
export const VLLM_DEFAULT_BASE_URL = "http://127.0.0.1:8000/v1";
|
||||
export const VLLM_PROVIDER_LABEL = "vLLM";
|
||||
export const VLLM_DEFAULT_API_KEY_ENV_VAR = "VLLM_API_KEY";
|
||||
export const VLLM_MODEL_PLACEHOLDER = "meta-llama/Meta-Llama-3-8B-Instruct";
|
||||
99
extensions/vllm/index.ts
Normal file
99
extensions/vllm/index.ts
Normal file
@@ -0,0 +1,99 @@
|
||||
// Vllm plugin entrypoint registers its OpenClaw integration.
|
||||
import {
|
||||
definePluginEntry,
|
||||
type OpenClawPluginApi,
|
||||
type ProviderAuthMethodNonInteractiveContext,
|
||||
} from "openclaw/plugin-sdk/plugin-entry";
|
||||
import {
|
||||
buildVllmProvider,
|
||||
VLLM_DEFAULT_API_KEY_ENV_VAR,
|
||||
VLLM_DEFAULT_BASE_URL,
|
||||
VLLM_MODEL_PLACEHOLDER,
|
||||
VLLM_PROVIDER_LABEL,
|
||||
} from "./api.js";
|
||||
import { wrapVllmProviderStream } from "./stream.js";
|
||||
import { resolveThinkingProfile } from "./thinking-policy.js";
|
||||
|
||||
const PROVIDER_ID = "vllm";
|
||||
|
||||
async function loadProviderSetup() {
|
||||
return await import("openclaw/plugin-sdk/provider-setup");
|
||||
}
|
||||
|
||||
export default definePluginEntry({
|
||||
id: "vllm",
|
||||
name: "vLLM Provider",
|
||||
description: "Bundled vLLM provider plugin",
|
||||
register(api: OpenClawPluginApi) {
|
||||
api.registerProvider({
|
||||
id: PROVIDER_ID,
|
||||
label: "vLLM",
|
||||
docsPath: "/providers/vllm",
|
||||
envVars: ["VLLM_API_KEY"],
|
||||
auth: [
|
||||
{
|
||||
id: "custom",
|
||||
label: VLLM_PROVIDER_LABEL,
|
||||
hint: "Local/self-hosted OpenAI-compatible server",
|
||||
kind: "custom",
|
||||
run: async (ctx) => {
|
||||
const providerSetup = await loadProviderSetup();
|
||||
return await providerSetup.promptAndConfigureOpenAICompatibleSelfHostedProviderAuth({
|
||||
cfg: ctx.config,
|
||||
prompter: ctx.prompter,
|
||||
providerId: PROVIDER_ID,
|
||||
providerLabel: VLLM_PROVIDER_LABEL,
|
||||
defaultBaseUrl: VLLM_DEFAULT_BASE_URL,
|
||||
defaultApiKeyEnvVar: VLLM_DEFAULT_API_KEY_ENV_VAR,
|
||||
modelPlaceholder: VLLM_MODEL_PLACEHOLDER,
|
||||
});
|
||||
},
|
||||
runNonInteractive: async (ctx: ProviderAuthMethodNonInteractiveContext) => {
|
||||
const providerSetup = await loadProviderSetup();
|
||||
return await providerSetup.configureOpenAICompatibleSelfHostedProviderNonInteractive({
|
||||
ctx,
|
||||
providerId: PROVIDER_ID,
|
||||
providerLabel: VLLM_PROVIDER_LABEL,
|
||||
defaultBaseUrl: VLLM_DEFAULT_BASE_URL,
|
||||
defaultApiKeyEnvVar: VLLM_DEFAULT_API_KEY_ENV_VAR,
|
||||
modelPlaceholder: VLLM_MODEL_PLACEHOLDER,
|
||||
});
|
||||
},
|
||||
},
|
||||
],
|
||||
catalog: {
|
||||
order: "late",
|
||||
run: async (ctx) => {
|
||||
const providerSetup = await loadProviderSetup();
|
||||
return await providerSetup.discoverOpenAICompatibleSelfHostedProvider({
|
||||
ctx,
|
||||
providerId: PROVIDER_ID,
|
||||
buildProvider: buildVllmProvider,
|
||||
});
|
||||
},
|
||||
},
|
||||
wizard: {
|
||||
setup: {
|
||||
choiceId: "vllm",
|
||||
choiceLabel: "vLLM",
|
||||
choiceHint: "Local/self-hosted OpenAI-compatible server",
|
||||
groupId: "vllm",
|
||||
groupLabel: "vLLM",
|
||||
groupHint: "Local/self-hosted OpenAI-compatible",
|
||||
methodId: "custom",
|
||||
},
|
||||
modelPicker: {
|
||||
label: "vLLM (custom)",
|
||||
hint: "Enter vLLM URL + API key + model",
|
||||
methodId: "custom",
|
||||
},
|
||||
},
|
||||
buildUnknownModelHint: () =>
|
||||
"vLLM requires authentication to be registered as a provider. " +
|
||||
'Set VLLM_API_KEY (any value works) or run "openclaw configure". ' +
|
||||
"See: https://docs.openclaw.ai/providers/vllm",
|
||||
resolveThinkingProfile,
|
||||
wrapStreamFn: wrapVllmProviderStream,
|
||||
});
|
||||
},
|
||||
});
|
||||
24
extensions/vllm/models.ts
Normal file
24
extensions/vllm/models.ts
Normal file
@@ -0,0 +1,24 @@
|
||||
// Vllm plugin module implements models behavior.
|
||||
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-contracts";
|
||||
import { discoverOpenAICompatibleLocalModels } from "openclaw/plugin-sdk/provider-setup";
|
||||
import { VLLM_DEFAULT_BASE_URL, VLLM_PROVIDER_LABEL } from "./defaults.js";
|
||||
|
||||
type ModelsConfig = NonNullable<OpenClawConfig["models"]>;
|
||||
type ProviderConfig = NonNullable<ModelsConfig["providers"]>[string];
|
||||
|
||||
export async function buildVllmProvider(params?: {
|
||||
baseUrl?: string;
|
||||
apiKey?: string;
|
||||
}): Promise<ProviderConfig> {
|
||||
const baseUrl = (params?.baseUrl?.trim() || VLLM_DEFAULT_BASE_URL).replace(/\/+$/, "");
|
||||
const models = await discoverOpenAICompatibleLocalModels({
|
||||
baseUrl,
|
||||
apiKey: params?.apiKey,
|
||||
label: VLLM_PROVIDER_LABEL,
|
||||
});
|
||||
return {
|
||||
baseUrl,
|
||||
api: "openai-completions",
|
||||
models,
|
||||
};
|
||||
}
|
||||
50
extensions/vllm/openclaw.plugin.json
Normal file
50
extensions/vllm/openclaw.plugin.json
Normal file
@@ -0,0 +1,50 @@
|
||||
{
|
||||
"id": "vllm",
|
||||
"activation": {
|
||||
"onStartup": false
|
||||
},
|
||||
"enabledByDefault": true,
|
||||
"providers": ["vllm"],
|
||||
"providerRequest": {
|
||||
"providers": {
|
||||
"vllm": {
|
||||
"family": "vllm",
|
||||
"openAICompletions": {
|
||||
"supportsStreamingUsage": true
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"modelPricing": {
|
||||
"providers": {
|
||||
"vllm": {
|
||||
"external": false
|
||||
}
|
||||
}
|
||||
},
|
||||
"setup": {
|
||||
"providers": [
|
||||
{
|
||||
"id": "vllm",
|
||||
"envVars": ["VLLM_API_KEY"]
|
||||
}
|
||||
]
|
||||
},
|
||||
"providerAuthChoices": [
|
||||
{
|
||||
"provider": "vllm",
|
||||
"method": "custom",
|
||||
"choiceId": "vllm",
|
||||
"choiceLabel": "vLLM",
|
||||
"choiceHint": "Local/self-hosted OpenAI-compatible server",
|
||||
"groupId": "vllm",
|
||||
"groupLabel": "vLLM",
|
||||
"groupHint": "Local/self-hosted OpenAI-compatible"
|
||||
}
|
||||
],
|
||||
"configSchema": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"properties": {}
|
||||
}
|
||||
}
|
||||
15
extensions/vllm/package.json
Normal file
15
extensions/vllm/package.json
Normal file
@@ -0,0 +1,15 @@
|
||||
{
|
||||
"name": "@openclaw/vllm-provider",
|
||||
"version": "2026.6.11",
|
||||
"private": true,
|
||||
"description": "OpenClaw vLLM provider plugin",
|
||||
"type": "module",
|
||||
"devDependencies": {
|
||||
"@openclaw/plugin-sdk": "workspace:*"
|
||||
},
|
||||
"openclaw": {
|
||||
"extensions": [
|
||||
"./index.ts"
|
||||
]
|
||||
}
|
||||
}
|
||||
29
extensions/vllm/provider-discovery.contract.test.ts
Normal file
29
extensions/vllm/provider-discovery.contract.test.ts
Normal file
@@ -0,0 +1,29 @@
|
||||
// Vllm tests cover provider discovery.contract plugin behavior.
|
||||
import { fileURLToPath } from "node:url";
|
||||
import { registerSingleProviderPlugin } from "openclaw/plugin-sdk/plugin-test-runtime";
|
||||
import { describeVllmProviderDiscoveryContract } from "openclaw/plugin-sdk/provider-test-contracts";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import vllmPlugin from "./index.js";
|
||||
|
||||
describeVllmProviderDiscoveryContract({
|
||||
load: () => import("./index.js"),
|
||||
apiModuleId: fileURLToPath(new URL("./api.js", import.meta.url)),
|
||||
});
|
||||
|
||||
describe("vLLM provider registration", () => {
|
||||
it("exposes the binary thinking profile hook", async () => {
|
||||
const provider = await registerSingleProviderPlugin(vllmPlugin);
|
||||
|
||||
expect(
|
||||
provider.resolveThinkingProfile?.({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
compat: { thinkingFormat: "qwen-chat-template" },
|
||||
}),
|
||||
).toEqual({
|
||||
levels: [{ id: "off" }, { id: "low", label: "on" }],
|
||||
defaultLevel: "off",
|
||||
});
|
||||
});
|
||||
});
|
||||
63
extensions/vllm/provider-policy-api.test.ts
Normal file
63
extensions/vllm/provider-policy-api.test.ts
Normal file
@@ -0,0 +1,63 @@
|
||||
// Vllm tests cover provider policy api plugin behavior.
|
||||
import { describe, expect, it } from "vitest";
|
||||
import { resolveThinkingProfile } from "./provider-policy-api.js";
|
||||
|
||||
describe("vLLM provider thinking policy", () => {
|
||||
it("exposes a binary profile for configured Qwen chat-template models", () => {
|
||||
expect(
|
||||
resolveThinkingProfile({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
compat: { thinkingFormat: "qwen-chat-template" },
|
||||
}),
|
||||
).toEqual({
|
||||
levels: [{ id: "off" }, { id: "low", label: "on" }],
|
||||
defaultLevel: "off",
|
||||
});
|
||||
});
|
||||
|
||||
it("uses configured Qwen compat even when catalog reasoning metadata is absent", () => {
|
||||
expect(
|
||||
resolveThinkingProfile({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
compat: { thinkingFormat: "qwen-chat-template" },
|
||||
}),
|
||||
).toEqual({
|
||||
levels: [{ id: "off" }, { id: "low", label: "on" }],
|
||||
defaultLevel: "off",
|
||||
});
|
||||
});
|
||||
|
||||
it("exposes a binary profile for vLLM Nemotron 3 reasoning models", () => {
|
||||
expect(
|
||||
resolveThinkingProfile({
|
||||
provider: "vllm",
|
||||
modelId: "nemotron-3-super",
|
||||
reasoning: true,
|
||||
}),
|
||||
).toEqual({
|
||||
levels: [{ id: "off" }, { id: "low", label: "on" }],
|
||||
defaultLevel: "off",
|
||||
});
|
||||
});
|
||||
|
||||
it("does not flatten unconfigured or non-reasoning vLLM models", () => {
|
||||
expect(
|
||||
resolveThinkingProfile({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
}),
|
||||
).toBeNull();
|
||||
expect(
|
||||
resolveThinkingProfile({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
reasoning: false,
|
||||
compat: { thinkingFormat: "qwen-chat-template" },
|
||||
}),
|
||||
).toBeNull();
|
||||
});
|
||||
});
|
||||
2
extensions/vllm/provider-policy-api.ts
Normal file
2
extensions/vllm/provider-policy-api.ts
Normal file
@@ -0,0 +1,2 @@
|
||||
// Vllm API module exposes the plugin public contract.
|
||||
export { resolveThinkingProfile } from "./thinking-policy.js";
|
||||
8
extensions/vllm/register.runtime.ts
Normal file
8
extensions/vllm/register.runtime.ts
Normal file
@@ -0,0 +1,8 @@
|
||||
// Vllm plugin module implements register behavior.
|
||||
export {
|
||||
buildVllmProvider,
|
||||
VLLM_DEFAULT_API_KEY_ENV_VAR,
|
||||
VLLM_DEFAULT_BASE_URL,
|
||||
VLLM_MODEL_PLACEHOLDER,
|
||||
VLLM_PROVIDER_LABEL,
|
||||
} from "./api.js";
|
||||
330
extensions/vllm/stream.test.ts
Normal file
330
extensions/vllm/stream.test.ts
Normal file
@@ -0,0 +1,330 @@
|
||||
// Vllm tests cover stream plugin behavior.
|
||||
import type { StreamFn } from "openclaw/plugin-sdk/agent-core";
|
||||
import type { Context, Model } from "openclaw/plugin-sdk/llm";
|
||||
import { describe, expect, it } from "vitest";
|
||||
import {
|
||||
createVllmProviderThinkingWrapper,
|
||||
createVllmQwenThinkingWrapper,
|
||||
wrapVllmProviderStream,
|
||||
} from "./stream.js";
|
||||
|
||||
function capturePayload(params: {
|
||||
format: "chat-template" | "top-level";
|
||||
thinkingLevel?: "off" | "low" | "medium" | "high" | "xhigh" | "max";
|
||||
reasoning?: unknown;
|
||||
initialPayload?: Record<string, unknown>;
|
||||
model?: Partial<Model<"openai-completions">>;
|
||||
}): Record<string, unknown> {
|
||||
let captured: Record<string, unknown> = {};
|
||||
const baseStreamFn: StreamFn = (_model, _context, options) => {
|
||||
const payload = { ...params.initialPayload };
|
||||
options?.onPayload?.(payload, _model);
|
||||
captured = payload;
|
||||
return {} as ReturnType<StreamFn>;
|
||||
};
|
||||
|
||||
const wrapped = createVllmQwenThinkingWrapper({
|
||||
baseStreamFn,
|
||||
format: params.format,
|
||||
thinkingLevel: params.thinkingLevel ?? "high",
|
||||
});
|
||||
void wrapped(
|
||||
{
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
...params.model,
|
||||
} as Model<"openai-completions">,
|
||||
{ messages: [] } as Context,
|
||||
params.reasoning === undefined ? {} : ({ reasoning: params.reasoning } as never),
|
||||
);
|
||||
|
||||
return captured;
|
||||
}
|
||||
|
||||
describe("createVllmQwenThinkingWrapper", () => {
|
||||
it("maps Qwen chat-template thinking off to chat_template_kwargs", () => {
|
||||
const payload = capturePayload({
|
||||
format: "chat-template",
|
||||
reasoning: "none",
|
||||
initialPayload: {
|
||||
reasoning_effort: "high",
|
||||
reasoning: { effort: "high" },
|
||||
reasoningEffort: "high",
|
||||
},
|
||||
});
|
||||
|
||||
expect(payload).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: false,
|
||||
preserve_thinking: true,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("maps Qwen chat-template thinking on to chat_template_kwargs", () => {
|
||||
expect(capturePayload({ format: "chat-template", reasoning: "medium" })).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: true,
|
||||
preserve_thinking: true,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("preserves explicit chat-template kwargs while setting enable_thinking", () => {
|
||||
expect(
|
||||
capturePayload({
|
||||
format: "chat-template",
|
||||
thinkingLevel: "off",
|
||||
initialPayload: {
|
||||
chat_template_kwargs: {
|
||||
preserve_thinking: false,
|
||||
force_nonempty_content: true,
|
||||
},
|
||||
},
|
||||
}),
|
||||
).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: false,
|
||||
preserve_thinking: false,
|
||||
force_nonempty_content: true,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("maps Qwen top-level thinking format to enable_thinking", () => {
|
||||
expect(capturePayload({ format: "top-level", thinkingLevel: "off" })).toEqual({
|
||||
enable_thinking: false,
|
||||
});
|
||||
expect(capturePayload({ format: "top-level", thinkingLevel: "high" })).toEqual({
|
||||
enable_thinking: true,
|
||||
});
|
||||
});
|
||||
|
||||
it("patches configured Qwen models unless reasoning is explicitly disabled", () => {
|
||||
expect(capturePayload({ format: "chat-template", model: { reasoning: undefined } })).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: true,
|
||||
preserve_thinking: true,
|
||||
},
|
||||
});
|
||||
expect(capturePayload({ format: "chat-template", model: { reasoning: false } })).toStrictEqual(
|
||||
{},
|
||||
);
|
||||
});
|
||||
|
||||
it("skips non-completions models", () => {
|
||||
expect(
|
||||
capturePayload({ format: "chat-template", model: { api: "openai-responses" as never } }),
|
||||
).toStrictEqual({});
|
||||
});
|
||||
});
|
||||
|
||||
describe("createVllmProviderThinkingWrapper", () => {
|
||||
function captureProviderPayload(params: {
|
||||
thinkingLevel?: "off" | "low" | "medium" | "high" | "xhigh" | "max";
|
||||
initialPayload?: Record<string, unknown>;
|
||||
model?: Partial<Model<"openai-completions">>;
|
||||
}): Record<string, unknown> {
|
||||
let captured: Record<string, unknown> = {};
|
||||
const baseStreamFn: StreamFn = (_model, _context, options) => {
|
||||
const payload = { ...params.initialPayload };
|
||||
options?.onPayload?.(payload, _model);
|
||||
captured = payload;
|
||||
return {} as ReturnType<StreamFn>;
|
||||
};
|
||||
|
||||
const wrapped = createVllmProviderThinkingWrapper({
|
||||
baseStreamFn,
|
||||
thinkingLevel: params.thinkingLevel ?? "high",
|
||||
});
|
||||
void wrapped(
|
||||
{
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "nemotron-3-super",
|
||||
reasoning: true,
|
||||
...params.model,
|
||||
} as Model<"openai-completions">,
|
||||
{ messages: [] } as Context,
|
||||
{},
|
||||
);
|
||||
|
||||
return captured;
|
||||
}
|
||||
|
||||
it("injects Nemotron 3 chat-template kwargs when thinking is off", () => {
|
||||
expect(captureProviderPayload({ thinkingLevel: "off" })).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: false,
|
||||
force_nonempty_content: true,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("does not inject Nemotron 3 chat-template kwargs when thinking is enabled", () => {
|
||||
expect(captureProviderPayload({ thinkingLevel: "low" })).toStrictEqual({});
|
||||
});
|
||||
|
||||
it("preserves existing Nemotron 3 chat-template kwargs over defaults", () => {
|
||||
expect(
|
||||
captureProviderPayload({
|
||||
thinkingLevel: "off",
|
||||
initialPayload: {
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: true,
|
||||
},
|
||||
},
|
||||
}),
|
||||
).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: true,
|
||||
force_nonempty_content: true,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("skips non-Nemotron vLLM models", () => {
|
||||
expect(
|
||||
captureProviderPayload({
|
||||
thinkingLevel: "off",
|
||||
model: { id: "Qwen/Qwen3-8B" },
|
||||
}),
|
||||
).toStrictEqual({});
|
||||
});
|
||||
});
|
||||
|
||||
describe("wrapVllmProviderStream", () => {
|
||||
it("registers when vLLM Qwen thinking format compat is configured", () => {
|
||||
expect(
|
||||
wrapVllmProviderStream({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
extraParams: {},
|
||||
model: {
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
compat: { thinkingFormat: "qwen-chat-template" },
|
||||
} as Model<"openai-completions">,
|
||||
streamFn: undefined,
|
||||
} as never),
|
||||
).toBeTypeOf("function");
|
||||
});
|
||||
|
||||
it("ignores request params when Qwen thinking format compat is not configured", () => {
|
||||
expect(
|
||||
wrapVllmProviderStream({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
extraParams: { qwenThinkingFormat: "chat-template" },
|
||||
model: {
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
} as Model<"openai-completions">,
|
||||
streamFn: undefined,
|
||||
} as never),
|
||||
).toBeUndefined();
|
||||
});
|
||||
|
||||
it("uses model compat for Qwen thinking format", () => {
|
||||
let captured: Record<string, unknown> = {};
|
||||
const baseStreamFn: StreamFn = (_model, _context, options) => {
|
||||
const payload = {};
|
||||
options?.onPayload?.(payload, _model);
|
||||
captured = payload;
|
||||
return {} as ReturnType<StreamFn>;
|
||||
};
|
||||
const model = {
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "Qwen/Qwen3-8B",
|
||||
reasoning: true,
|
||||
compat: { thinkingFormat: "qwen-chat-template" },
|
||||
} as unknown as Model<"openai-completions">;
|
||||
const wrapped = wrapVllmProviderStream({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
extraParams: {},
|
||||
thinkingLevel: "off",
|
||||
model,
|
||||
streamFn: baseStreamFn,
|
||||
} as never);
|
||||
|
||||
expect(wrapped).toBeTypeOf("function");
|
||||
void wrapped?.(model, { messages: [] } as Context, {});
|
||||
|
||||
expect(captured).toEqual({
|
||||
chat_template_kwargs: {
|
||||
enable_thinking: false,
|
||||
preserve_thinking: true,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("skips unconfigured vLLM and non-vLLM providers", () => {
|
||||
expect(
|
||||
wrapVllmProviderStream({
|
||||
provider: "vllm",
|
||||
modelId: "Qwen/Qwen3-8B",
|
||||
extraParams: {},
|
||||
model: {
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "Qwen/Qwen3-8B",
|
||||
} as Model<"openai-completions">,
|
||||
streamFn: undefined,
|
||||
} as never),
|
||||
).toBeUndefined();
|
||||
|
||||
expect(
|
||||
wrapVllmProviderStream({
|
||||
provider: "openai",
|
||||
modelId: "gpt-5.4",
|
||||
extraParams: {},
|
||||
model: {
|
||||
api: "openai-completions",
|
||||
provider: "openai",
|
||||
id: "gpt-5.4",
|
||||
} as Model<"openai-completions">,
|
||||
streamFn: undefined,
|
||||
} as never),
|
||||
).toBeUndefined();
|
||||
});
|
||||
|
||||
it("registers for vLLM Nemotron when thinking is off", () => {
|
||||
expect(
|
||||
wrapVllmProviderStream({
|
||||
provider: "vllm",
|
||||
modelId: "nemotron-3-super",
|
||||
extraParams: {},
|
||||
thinkingLevel: "off",
|
||||
model: {
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "nemotron-3-super",
|
||||
} as Model<"openai-completions">,
|
||||
streamFn: undefined,
|
||||
} as never),
|
||||
).toBeTypeOf("function");
|
||||
|
||||
expect(
|
||||
wrapVllmProviderStream({
|
||||
provider: "vllm",
|
||||
modelId: "nemotron-3-super",
|
||||
extraParams: {},
|
||||
thinkingLevel: "low",
|
||||
model: {
|
||||
api: "openai-completions",
|
||||
provider: "vllm",
|
||||
id: "nemotron-3-super",
|
||||
} as Model<"openai-completions">,
|
||||
streamFn: undefined,
|
||||
} as never),
|
||||
).toBeUndefined();
|
||||
});
|
||||
});
|
||||
125
extensions/vllm/stream.ts
Normal file
125
extensions/vllm/stream.ts
Normal file
@@ -0,0 +1,125 @@
|
||||
// Vllm plugin module implements stream behavior.
|
||||
import type { StreamFn } from "openclaw/plugin-sdk/agent-core";
|
||||
import type { ProviderWrapStreamFnContext } from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { normalizeProviderId } from "openclaw/plugin-sdk/provider-model-shared";
|
||||
import {
|
||||
createPayloadPatchStreamWrapper,
|
||||
isOpenAICompatibleThinkingEnabled,
|
||||
setQwenChatTemplateThinking,
|
||||
} from "openclaw/plugin-sdk/provider-stream-shared";
|
||||
import {
|
||||
resolveVllmQwenThinkingFormatFromCompat,
|
||||
type VllmQwenThinkingFormat,
|
||||
} from "./thinking-policy.js";
|
||||
|
||||
type VllmThinkingLevel = ProviderWrapStreamFnContext["thinkingLevel"];
|
||||
|
||||
function isVllmProviderId(providerId: string): boolean {
|
||||
return normalizeProviderId(providerId) === "vllm";
|
||||
}
|
||||
|
||||
function resolveVllmQwenThinkingFormat(
|
||||
ctx: Pick<ProviderWrapStreamFnContext, "model">,
|
||||
): VllmQwenThinkingFormat | undefined {
|
||||
return resolveVllmQwenThinkingFormatFromCompat(ctx.model?.compat);
|
||||
}
|
||||
|
||||
function isVllmNemotronModel(model: { api?: unknown; provider?: unknown; id?: unknown }): boolean {
|
||||
return (
|
||||
model.api === "openai-completions" &&
|
||||
typeof model.provider === "string" &&
|
||||
normalizeProviderId(model.provider) === "vllm" &&
|
||||
typeof model.id === "string" &&
|
||||
/\bnemotron-3(?:[-_](?:nano|super|ultra))?\b/i.test(model.id)
|
||||
);
|
||||
}
|
||||
|
||||
function setNemotronThinkingOffChatTemplateKwargs(payload: Record<string, unknown>): void {
|
||||
const defaults = {
|
||||
enable_thinking: false,
|
||||
force_nonempty_content: true,
|
||||
};
|
||||
const existing = payload.chat_template_kwargs;
|
||||
payload.chat_template_kwargs =
|
||||
existing && typeof existing === "object" && !Array.isArray(existing)
|
||||
? {
|
||||
...defaults,
|
||||
...(existing as Record<string, unknown>),
|
||||
}
|
||||
: defaults;
|
||||
}
|
||||
|
||||
export function createVllmQwenThinkingWrapper(params: {
|
||||
baseStreamFn: StreamFn | undefined;
|
||||
format: VllmQwenThinkingFormat;
|
||||
thinkingLevel: VllmThinkingLevel;
|
||||
}): StreamFn {
|
||||
return createPayloadPatchStreamWrapper(
|
||||
params.baseStreamFn,
|
||||
({ payload: payloadObj, options }) => {
|
||||
const enableThinking = isOpenAICompatibleThinkingEnabled({
|
||||
thinkingLevel: params.thinkingLevel,
|
||||
options,
|
||||
});
|
||||
if (params.format === "chat-template") {
|
||||
setQwenChatTemplateThinking(payloadObj, enableThinking);
|
||||
} else {
|
||||
payloadObj.enable_thinking = enableThinking;
|
||||
}
|
||||
delete payloadObj.reasoning_effort;
|
||||
delete payloadObj.reasoningEffort;
|
||||
delete payloadObj.reasoning;
|
||||
},
|
||||
{
|
||||
shouldPatch: ({ model }) => model.api === "openai-completions" && (model.reasoning ?? true),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
export function createVllmProviderThinkingWrapper(params: {
|
||||
baseStreamFn: StreamFn | undefined;
|
||||
qwenFormat?: VllmQwenThinkingFormat;
|
||||
thinkingLevel: VllmThinkingLevel;
|
||||
}): StreamFn {
|
||||
const qwenWrapped = params.qwenFormat
|
||||
? createVllmQwenThinkingWrapper({
|
||||
baseStreamFn: params.baseStreamFn,
|
||||
format: params.qwenFormat,
|
||||
thinkingLevel: params.thinkingLevel,
|
||||
})
|
||||
: params.baseStreamFn;
|
||||
return createPayloadPatchStreamWrapper(
|
||||
qwenWrapped,
|
||||
({ payload: payloadObj }) => {
|
||||
setNemotronThinkingOffChatTemplateKwargs(payloadObj);
|
||||
},
|
||||
{
|
||||
shouldPatch: ({ model }) =>
|
||||
model.api === "openai-completions" &&
|
||||
params.thinkingLevel === "off" &&
|
||||
isVllmNemotronModel(model),
|
||||
},
|
||||
);
|
||||
}
|
||||
|
||||
export function wrapVllmProviderStream(ctx: ProviderWrapStreamFnContext): StreamFn | undefined {
|
||||
if (!isVllmProviderId(ctx.provider) || (ctx.model && ctx.model.api !== "openai-completions")) {
|
||||
return undefined;
|
||||
}
|
||||
const qwenFormat = resolveVllmQwenThinkingFormat(ctx);
|
||||
const shouldHandleNemotron =
|
||||
ctx.thinkingLevel === "off" &&
|
||||
isVllmNemotronModel({
|
||||
api: "openai-completions",
|
||||
provider: ctx.provider,
|
||||
id: ctx.modelId,
|
||||
});
|
||||
if (!qwenFormat && !shouldHandleNemotron) {
|
||||
return undefined;
|
||||
}
|
||||
return createVllmProviderThinkingWrapper({
|
||||
baseStreamFn: ctx.streamFn,
|
||||
qwenFormat,
|
||||
thinkingLevel: ctx.thinkingLevel,
|
||||
});
|
||||
}
|
||||
66
extensions/vllm/thinking-policy.ts
Normal file
66
extensions/vllm/thinking-policy.ts
Normal file
@@ -0,0 +1,66 @@
|
||||
// Vllm plugin module implements thinking policy behavior.
|
||||
import type {
|
||||
ProviderDefaultThinkingPolicyContext,
|
||||
ProviderThinkingProfile,
|
||||
} from "openclaw/plugin-sdk/plugin-entry";
|
||||
import { normalizeProviderId } from "openclaw/plugin-sdk/provider-model-shared";
|
||||
|
||||
export type VllmQwenThinkingFormat = "chat-template" | "top-level";
|
||||
|
||||
const VLLM_BINARY_THINKING_PROFILE = {
|
||||
levels: [{ id: "off" }, { id: "low", label: "on" }],
|
||||
defaultLevel: "off",
|
||||
} satisfies ProviderThinkingProfile;
|
||||
|
||||
export function normalizeVllmQwenThinkingFormat(
|
||||
value: unknown,
|
||||
): VllmQwenThinkingFormat | undefined {
|
||||
if (typeof value !== "string") {
|
||||
return undefined;
|
||||
}
|
||||
const normalized = value.trim().toLowerCase().replace(/_/g, "-");
|
||||
if (
|
||||
normalized === "chat-template" ||
|
||||
normalized === "chat-template-kwargs" ||
|
||||
normalized === "chat-template-kwarg" ||
|
||||
normalized === "chat-template-arguments" ||
|
||||
normalized === "qwen-chat-template"
|
||||
) {
|
||||
return "chat-template";
|
||||
}
|
||||
if (
|
||||
normalized === "top-level" ||
|
||||
normalized === "enable-thinking" ||
|
||||
normalized === "request-body" ||
|
||||
normalized === "qwen"
|
||||
) {
|
||||
return "top-level";
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
export function resolveVllmQwenThinkingFormatFromCompat(
|
||||
compat?: ProviderDefaultThinkingPolicyContext["compat"],
|
||||
): VllmQwenThinkingFormat | undefined {
|
||||
return normalizeVllmQwenThinkingFormat(compat?.thinkingFormat);
|
||||
}
|
||||
|
||||
function isVllmNemotronThinkingModel(modelId: string): boolean {
|
||||
return /\bnemotron-3(?:[-_](?:nano|super|ultra))?\b/i.test(modelId);
|
||||
}
|
||||
|
||||
export function resolveThinkingProfile(
|
||||
ctx: ProviderDefaultThinkingPolicyContext,
|
||||
): ProviderThinkingProfile | null {
|
||||
if (normalizeProviderId(ctx.provider) !== "vllm") {
|
||||
return null;
|
||||
}
|
||||
if (ctx.reasoning === false) {
|
||||
return null;
|
||||
}
|
||||
const qwenFormat = resolveVllmQwenThinkingFormatFromCompat(ctx.compat);
|
||||
if (qwenFormat || (ctx.reasoning === true && isVllmNemotronThinkingModel(ctx.modelId))) {
|
||||
return VLLM_BINARY_THINKING_PROFILE;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
16
extensions/vllm/tsconfig.json
Normal file
16
extensions/vllm/tsconfig.json
Normal file
@@ -0,0 +1,16 @@
|
||||
{
|
||||
"extends": "../tsconfig.package-boundary.base.json",
|
||||
"compilerOptions": {
|
||||
"rootDir": "."
|
||||
},
|
||||
"include": ["./*.ts", "./src/**/*.ts"],
|
||||
"exclude": [
|
||||
"./**/*.test.ts",
|
||||
"./dist/**",
|
||||
"./node_modules/**",
|
||||
"./src/test-support/**",
|
||||
"./src/**/*test-helpers.ts",
|
||||
"./src/**/*test-harness.ts",
|
||||
"./src/**/*test-support.ts"
|
||||
]
|
||||
}
|
||||
Reference in New Issue
Block a user