Vendor OpenClaw source as Adolf fork baseline
Some checks failed
ClawSweeper Dispatch / dispatch (push) Has been cancelled
CodeQL / Security High (actions) (push) Has been cancelled
CodeQL / Security High (channel-runtime-boundary) (push) Has been cancelled
CodeQL / Security High (core-auth-secrets) (push) Has been cancelled
CodeQL / Security High (mcp-process-tool-boundary) (push) Has been cancelled
CodeQL / Security High (network-ssrf-boundary) (push) Has been cancelled
CodeQL / Security High (plugin-trust-boundary) (push) Has been cancelled
CodeQL / Security High (process-exec-boundary) (push) Has been cancelled
Docs Sync Publish Repo / sync-publish-repo (push) Has been cancelled
Docs / docs (push) Has been cancelled
OpenClaw Stable Main Closeout / Resolve stable release closeout inputs (push) Has been cancelled
OpenClaw Stable Main Closeout / Verify stable main closeout (push) Has been cancelled
Workflow Sanity / no-tabs (push) Has been cancelled
Workflow Sanity / actionlint (push) Has been cancelled
Workflow Sanity / generated-doc-baselines (push) Has been cancelled
CI / runner-admission (push) Has been cancelled
CI / preflight (push) Has been cancelled
CI / security-fast (push) Has been cancelled
CI / pnpm-store-warmup (push) Has been cancelled
CI / build-artifacts (push) Has been cancelled
CI / native-i18n (push) Has been cancelled
CI / ${{ matrix.check_name }} (push) Has been cancelled
CI / ${{ matrix.checkName }} (push) Has been cancelled
CI / checks-node-compat-node22 (push) Has been cancelled
CI / check-bundled-channel-config-metadata (push) Has been cancelled
CI / check-dependencies (push) Has been cancelled
CI / check-guards (push) Has been cancelled
CI / check-lint (push) Has been cancelled
CI / check-prod-types (push) Has been cancelled
CI / check-shrinkwrap (push) Has been cancelled
CI / check-test-types (push) Has been cancelled
CI / check-additional-boundaries-a (push) Has been cancelled
CI / check-additional-boundaries-bcd (push) Has been cancelled
CI / check-additional-extension-bundled (push) Has been cancelled
CI / check-additional-extension-channels (push) Has been cancelled
CI / check-additional-extension-package-boundary (push) Has been cancelled
CI / check-additional-runtime-topology-architecture (push) Has been cancelled
CI / check-session-accessor-boundary (push) Has been cancelled
CI / check-session-transcript-reader-boundary (push) Has been cancelled
CI / check-docs (push) Has been cancelled
CI / skills-python (push) Has been cancelled
CI / macos-swift (push) Has been cancelled
CI / ios-build (push) Has been cancelled
CI / ci-timings-summary (push) Has been cancelled
Native App Locale Refresh / Refresh native fa (push) Has been cancelled
Native App Locale Refresh / Refresh native fr (push) Has been cancelled
Native App Locale Refresh / Refresh native hi (push) Has been cancelled
Native App Locale Refresh / Refresh native id (push) Has been cancelled
Native App Locale Refresh / Refresh native it (push) Has been cancelled
Native App Locale Refresh / Refresh native ja-JP (push) Has been cancelled
Control UI Locale Refresh / plan (push) Has been cancelled
Control UI Locale Refresh / Refresh ${{ matrix.locale }} (push) Has been cancelled
Control UI Locale Refresh / Commit control UI locale refresh (push) Has been cancelled
Live Media Runner Image / Build live media runner image (push) Has been cancelled
Native App Locale Refresh / Refresh native ar (push) Has been cancelled
Native App Locale Refresh / Refresh native de (push) Has been cancelled
Native App Locale Refresh / Refresh native es (push) Has been cancelled
Native App Locale Refresh / Refresh native ko (push) Has been cancelled
Native App Locale Refresh / Refresh native nl (push) Has been cancelled
Native App Locale Refresh / Refresh native pl (push) Has been cancelled
Native App Locale Refresh / Refresh native pt-BR (push) Has been cancelled
Native App Locale Refresh / Refresh native ru (push) Has been cancelled
Native App Locale Refresh / Refresh native sv (push) Has been cancelled
Native App Locale Refresh / Refresh native th (push) Has been cancelled
Native App Locale Refresh / Refresh native tr (push) Has been cancelled
Native App Locale Refresh / Refresh native uk (push) Has been cancelled
Native App Locale Refresh / Refresh native vi (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-CN (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-TW (push) Has been cancelled
Native App Locale Refresh / Commit native locale refresh (push) Has been cancelled
Plugin Init Scaffold Validation / Validate provider scaffold (push) Has been cancelled
Plugin NPM Release / preview_plugins_npm (push) Has been cancelled
Plugin NPM Release / Validate release publish approval (push) Has been cancelled
Plugin NPM Release / preview_plugin_pack (push) Has been cancelled
Plugin NPM Release / publish_plugins_npm (push) Has been cancelled
Sandbox Common Smoke / sandbox-common-smoke (push) Has been cancelled
Website Installer Sync / static (push) Has been cancelled
Website Installer Sync / linux-docker (push) Has been cancelled
Website Installer Sync / macos-installer (push) Has been cancelled
Website Installer Sync / windows-installer (push) Has been cancelled
Website Installer Sync / sync-website (push) Has been cancelled

Adolf is a fork/vendored clone of github.com/openclaw/openclaw (v2026.6.11),
free to diverge. Tree copied sans upstream .git; upstream remote added for
future syncs. Node pinned to 24 (.nvmrc); engines already require >=22.19.
Preserves docs/ARCHITECTURE.md.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LeqyaxJF2nbRXJtae2kNB2
This commit is contained in:
2026-07-05 09:36:54 +00:00
parent 3216769225
commit bedb527145
21108 changed files with 6010766 additions and 0 deletions

View File

@@ -0,0 +1,3 @@
# vLLM Provider
Bundled provider plugin for vLLM discovery and setup.

9
extensions/vllm/api.ts Normal file
View File

@@ -0,0 +1,9 @@
// Vllm API module exposes the plugin public contract.
export {
VLLM_DEFAULT_API_KEY_ENV_VAR,
VLLM_DEFAULT_BASE_URL,
VLLM_MODEL_PLACEHOLDER,
VLLM_PROVIDER_LABEL,
} from "./defaults.js";
export { buildVllmProvider } from "./models.js";
export { createVllmQwenThinkingWrapper, wrapVllmProviderStream } from "./stream.js";

View File

@@ -0,0 +1,5 @@
// Vllm plugin module implements defaults behavior.
export const VLLM_DEFAULT_BASE_URL = "http://127.0.0.1:8000/v1";
export const VLLM_PROVIDER_LABEL = "vLLM";
export const VLLM_DEFAULT_API_KEY_ENV_VAR = "VLLM_API_KEY";
export const VLLM_MODEL_PLACEHOLDER = "meta-llama/Meta-Llama-3-8B-Instruct";

99
extensions/vllm/index.ts Normal file
View File

@@ -0,0 +1,99 @@
// Vllm plugin entrypoint registers its OpenClaw integration.
import {
definePluginEntry,
type OpenClawPluginApi,
type ProviderAuthMethodNonInteractiveContext,
} from "openclaw/plugin-sdk/plugin-entry";
import {
buildVllmProvider,
VLLM_DEFAULT_API_KEY_ENV_VAR,
VLLM_DEFAULT_BASE_URL,
VLLM_MODEL_PLACEHOLDER,
VLLM_PROVIDER_LABEL,
} from "./api.js";
import { wrapVllmProviderStream } from "./stream.js";
import { resolveThinkingProfile } from "./thinking-policy.js";
const PROVIDER_ID = "vllm";
async function loadProviderSetup() {
return await import("openclaw/plugin-sdk/provider-setup");
}
export default definePluginEntry({
id: "vllm",
name: "vLLM Provider",
description: "Bundled vLLM provider plugin",
register(api: OpenClawPluginApi) {
api.registerProvider({
id: PROVIDER_ID,
label: "vLLM",
docsPath: "/providers/vllm",
envVars: ["VLLM_API_KEY"],
auth: [
{
id: "custom",
label: VLLM_PROVIDER_LABEL,
hint: "Local/self-hosted OpenAI-compatible server",
kind: "custom",
run: async (ctx) => {
const providerSetup = await loadProviderSetup();
return await providerSetup.promptAndConfigureOpenAICompatibleSelfHostedProviderAuth({
cfg: ctx.config,
prompter: ctx.prompter,
providerId: PROVIDER_ID,
providerLabel: VLLM_PROVIDER_LABEL,
defaultBaseUrl: VLLM_DEFAULT_BASE_URL,
defaultApiKeyEnvVar: VLLM_DEFAULT_API_KEY_ENV_VAR,
modelPlaceholder: VLLM_MODEL_PLACEHOLDER,
});
},
runNonInteractive: async (ctx: ProviderAuthMethodNonInteractiveContext) => {
const providerSetup = await loadProviderSetup();
return await providerSetup.configureOpenAICompatibleSelfHostedProviderNonInteractive({
ctx,
providerId: PROVIDER_ID,
providerLabel: VLLM_PROVIDER_LABEL,
defaultBaseUrl: VLLM_DEFAULT_BASE_URL,
defaultApiKeyEnvVar: VLLM_DEFAULT_API_KEY_ENV_VAR,
modelPlaceholder: VLLM_MODEL_PLACEHOLDER,
});
},
},
],
catalog: {
order: "late",
run: async (ctx) => {
const providerSetup = await loadProviderSetup();
return await providerSetup.discoverOpenAICompatibleSelfHostedProvider({
ctx,
providerId: PROVIDER_ID,
buildProvider: buildVllmProvider,
});
},
},
wizard: {
setup: {
choiceId: "vllm",
choiceLabel: "vLLM",
choiceHint: "Local/self-hosted OpenAI-compatible server",
groupId: "vllm",
groupLabel: "vLLM",
groupHint: "Local/self-hosted OpenAI-compatible",
methodId: "custom",
},
modelPicker: {
label: "vLLM (custom)",
hint: "Enter vLLM URL + API key + model",
methodId: "custom",
},
},
buildUnknownModelHint: () =>
"vLLM requires authentication to be registered as a provider. " +
'Set VLLM_API_KEY (any value works) or run "openclaw configure". ' +
"See: https://docs.openclaw.ai/providers/vllm",
resolveThinkingProfile,
wrapStreamFn: wrapVllmProviderStream,
});
},
});

24
extensions/vllm/models.ts Normal file
View File

@@ -0,0 +1,24 @@
// Vllm plugin module implements models behavior.
import type { OpenClawConfig } from "openclaw/plugin-sdk/config-contracts";
import { discoverOpenAICompatibleLocalModels } from "openclaw/plugin-sdk/provider-setup";
import { VLLM_DEFAULT_BASE_URL, VLLM_PROVIDER_LABEL } from "./defaults.js";
type ModelsConfig = NonNullable<OpenClawConfig["models"]>;
type ProviderConfig = NonNullable<ModelsConfig["providers"]>[string];
export async function buildVllmProvider(params?: {
baseUrl?: string;
apiKey?: string;
}): Promise<ProviderConfig> {
const baseUrl = (params?.baseUrl?.trim() || VLLM_DEFAULT_BASE_URL).replace(/\/+$/, "");
const models = await discoverOpenAICompatibleLocalModels({
baseUrl,
apiKey: params?.apiKey,
label: VLLM_PROVIDER_LABEL,
});
return {
baseUrl,
api: "openai-completions",
models,
};
}

View File

@@ -0,0 +1,50 @@
{
"id": "vllm",
"activation": {
"onStartup": false
},
"enabledByDefault": true,
"providers": ["vllm"],
"providerRequest": {
"providers": {
"vllm": {
"family": "vllm",
"openAICompletions": {
"supportsStreamingUsage": true
}
}
}
},
"modelPricing": {
"providers": {
"vllm": {
"external": false
}
}
},
"setup": {
"providers": [
{
"id": "vllm",
"envVars": ["VLLM_API_KEY"]
}
]
},
"providerAuthChoices": [
{
"provider": "vllm",
"method": "custom",
"choiceId": "vllm",
"choiceLabel": "vLLM",
"choiceHint": "Local/self-hosted OpenAI-compatible server",
"groupId": "vllm",
"groupLabel": "vLLM",
"groupHint": "Local/self-hosted OpenAI-compatible"
}
],
"configSchema": {
"type": "object",
"additionalProperties": false,
"properties": {}
}
}

View File

@@ -0,0 +1,15 @@
{
"name": "@openclaw/vllm-provider",
"version": "2026.6.11",
"private": true,
"description": "OpenClaw vLLM provider plugin",
"type": "module",
"devDependencies": {
"@openclaw/plugin-sdk": "workspace:*"
},
"openclaw": {
"extensions": [
"./index.ts"
]
}
}

View File

@@ -0,0 +1,29 @@
// Vllm tests cover provider discovery.contract plugin behavior.
import { fileURLToPath } from "node:url";
import { registerSingleProviderPlugin } from "openclaw/plugin-sdk/plugin-test-runtime";
import { describeVllmProviderDiscoveryContract } from "openclaw/plugin-sdk/provider-test-contracts";
import { describe, expect, it } from "vitest";
import vllmPlugin from "./index.js";
describeVllmProviderDiscoveryContract({
load: () => import("./index.js"),
apiModuleId: fileURLToPath(new URL("./api.js", import.meta.url)),
});
describe("vLLM provider registration", () => {
it("exposes the binary thinking profile hook", async () => {
const provider = await registerSingleProviderPlugin(vllmPlugin);
expect(
provider.resolveThinkingProfile?.({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
reasoning: true,
compat: { thinkingFormat: "qwen-chat-template" },
}),
).toEqual({
levels: [{ id: "off" }, { id: "low", label: "on" }],
defaultLevel: "off",
});
});
});

View File

@@ -0,0 +1,63 @@
// Vllm tests cover provider policy api plugin behavior.
import { describe, expect, it } from "vitest";
import { resolveThinkingProfile } from "./provider-policy-api.js";
describe("vLLM provider thinking policy", () => {
it("exposes a binary profile for configured Qwen chat-template models", () => {
expect(
resolveThinkingProfile({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
reasoning: true,
compat: { thinkingFormat: "qwen-chat-template" },
}),
).toEqual({
levels: [{ id: "off" }, { id: "low", label: "on" }],
defaultLevel: "off",
});
});
it("uses configured Qwen compat even when catalog reasoning metadata is absent", () => {
expect(
resolveThinkingProfile({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
compat: { thinkingFormat: "qwen-chat-template" },
}),
).toEqual({
levels: [{ id: "off" }, { id: "low", label: "on" }],
defaultLevel: "off",
});
});
it("exposes a binary profile for vLLM Nemotron 3 reasoning models", () => {
expect(
resolveThinkingProfile({
provider: "vllm",
modelId: "nemotron-3-super",
reasoning: true,
}),
).toEqual({
levels: [{ id: "off" }, { id: "low", label: "on" }],
defaultLevel: "off",
});
});
it("does not flatten unconfigured or non-reasoning vLLM models", () => {
expect(
resolveThinkingProfile({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
reasoning: true,
}),
).toBeNull();
expect(
resolveThinkingProfile({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
reasoning: false,
compat: { thinkingFormat: "qwen-chat-template" },
}),
).toBeNull();
});
});

View File

@@ -0,0 +1,2 @@
// Vllm API module exposes the plugin public contract.
export { resolveThinkingProfile } from "./thinking-policy.js";

View File

@@ -0,0 +1,8 @@
// Vllm plugin module implements register behavior.
export {
buildVllmProvider,
VLLM_DEFAULT_API_KEY_ENV_VAR,
VLLM_DEFAULT_BASE_URL,
VLLM_MODEL_PLACEHOLDER,
VLLM_PROVIDER_LABEL,
} from "./api.js";

View File

@@ -0,0 +1,330 @@
// Vllm tests cover stream plugin behavior.
import type { StreamFn } from "openclaw/plugin-sdk/agent-core";
import type { Context, Model } from "openclaw/plugin-sdk/llm";
import { describe, expect, it } from "vitest";
import {
createVllmProviderThinkingWrapper,
createVllmQwenThinkingWrapper,
wrapVllmProviderStream,
} from "./stream.js";
function capturePayload(params: {
format: "chat-template" | "top-level";
thinkingLevel?: "off" | "low" | "medium" | "high" | "xhigh" | "max";
reasoning?: unknown;
initialPayload?: Record<string, unknown>;
model?: Partial<Model<"openai-completions">>;
}): Record<string, unknown> {
let captured: Record<string, unknown> = {};
const baseStreamFn: StreamFn = (_model, _context, options) => {
const payload = { ...params.initialPayload };
options?.onPayload?.(payload, _model);
captured = payload;
return {} as ReturnType<StreamFn>;
};
const wrapped = createVllmQwenThinkingWrapper({
baseStreamFn,
format: params.format,
thinkingLevel: params.thinkingLevel ?? "high",
});
void wrapped(
{
api: "openai-completions",
provider: "vllm",
id: "Qwen/Qwen3-8B",
reasoning: true,
...params.model,
} as Model<"openai-completions">,
{ messages: [] } as Context,
params.reasoning === undefined ? {} : ({ reasoning: params.reasoning } as never),
);
return captured;
}
describe("createVllmQwenThinkingWrapper", () => {
it("maps Qwen chat-template thinking off to chat_template_kwargs", () => {
const payload = capturePayload({
format: "chat-template",
reasoning: "none",
initialPayload: {
reasoning_effort: "high",
reasoning: { effort: "high" },
reasoningEffort: "high",
},
});
expect(payload).toEqual({
chat_template_kwargs: {
enable_thinking: false,
preserve_thinking: true,
},
});
});
it("maps Qwen chat-template thinking on to chat_template_kwargs", () => {
expect(capturePayload({ format: "chat-template", reasoning: "medium" })).toEqual({
chat_template_kwargs: {
enable_thinking: true,
preserve_thinking: true,
},
});
});
it("preserves explicit chat-template kwargs while setting enable_thinking", () => {
expect(
capturePayload({
format: "chat-template",
thinkingLevel: "off",
initialPayload: {
chat_template_kwargs: {
preserve_thinking: false,
force_nonempty_content: true,
},
},
}),
).toEqual({
chat_template_kwargs: {
enable_thinking: false,
preserve_thinking: false,
force_nonempty_content: true,
},
});
});
it("maps Qwen top-level thinking format to enable_thinking", () => {
expect(capturePayload({ format: "top-level", thinkingLevel: "off" })).toEqual({
enable_thinking: false,
});
expect(capturePayload({ format: "top-level", thinkingLevel: "high" })).toEqual({
enable_thinking: true,
});
});
it("patches configured Qwen models unless reasoning is explicitly disabled", () => {
expect(capturePayload({ format: "chat-template", model: { reasoning: undefined } })).toEqual({
chat_template_kwargs: {
enable_thinking: true,
preserve_thinking: true,
},
});
expect(capturePayload({ format: "chat-template", model: { reasoning: false } })).toStrictEqual(
{},
);
});
it("skips non-completions models", () => {
expect(
capturePayload({ format: "chat-template", model: { api: "openai-responses" as never } }),
).toStrictEqual({});
});
});
describe("createVllmProviderThinkingWrapper", () => {
function captureProviderPayload(params: {
thinkingLevel?: "off" | "low" | "medium" | "high" | "xhigh" | "max";
initialPayload?: Record<string, unknown>;
model?: Partial<Model<"openai-completions">>;
}): Record<string, unknown> {
let captured: Record<string, unknown> = {};
const baseStreamFn: StreamFn = (_model, _context, options) => {
const payload = { ...params.initialPayload };
options?.onPayload?.(payload, _model);
captured = payload;
return {} as ReturnType<StreamFn>;
};
const wrapped = createVllmProviderThinkingWrapper({
baseStreamFn,
thinkingLevel: params.thinkingLevel ?? "high",
});
void wrapped(
{
api: "openai-completions",
provider: "vllm",
id: "nemotron-3-super",
reasoning: true,
...params.model,
} as Model<"openai-completions">,
{ messages: [] } as Context,
{},
);
return captured;
}
it("injects Nemotron 3 chat-template kwargs when thinking is off", () => {
expect(captureProviderPayload({ thinkingLevel: "off" })).toEqual({
chat_template_kwargs: {
enable_thinking: false,
force_nonempty_content: true,
},
});
});
it("does not inject Nemotron 3 chat-template kwargs when thinking is enabled", () => {
expect(captureProviderPayload({ thinkingLevel: "low" })).toStrictEqual({});
});
it("preserves existing Nemotron 3 chat-template kwargs over defaults", () => {
expect(
captureProviderPayload({
thinkingLevel: "off",
initialPayload: {
chat_template_kwargs: {
enable_thinking: true,
},
},
}),
).toEqual({
chat_template_kwargs: {
enable_thinking: true,
force_nonempty_content: true,
},
});
});
it("skips non-Nemotron vLLM models", () => {
expect(
captureProviderPayload({
thinkingLevel: "off",
model: { id: "Qwen/Qwen3-8B" },
}),
).toStrictEqual({});
});
});
describe("wrapVllmProviderStream", () => {
it("registers when vLLM Qwen thinking format compat is configured", () => {
expect(
wrapVllmProviderStream({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
extraParams: {},
model: {
api: "openai-completions",
provider: "vllm",
id: "Qwen/Qwen3-8B",
reasoning: true,
compat: { thinkingFormat: "qwen-chat-template" },
} as Model<"openai-completions">,
streamFn: undefined,
} as never),
).toBeTypeOf("function");
});
it("ignores request params when Qwen thinking format compat is not configured", () => {
expect(
wrapVllmProviderStream({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
extraParams: { qwenThinkingFormat: "chat-template" },
model: {
api: "openai-completions",
provider: "vllm",
id: "Qwen/Qwen3-8B",
reasoning: true,
} as Model<"openai-completions">,
streamFn: undefined,
} as never),
).toBeUndefined();
});
it("uses model compat for Qwen thinking format", () => {
let captured: Record<string, unknown> = {};
const baseStreamFn: StreamFn = (_model, _context, options) => {
const payload = {};
options?.onPayload?.(payload, _model);
captured = payload;
return {} as ReturnType<StreamFn>;
};
const model = {
api: "openai-completions",
provider: "vllm",
id: "Qwen/Qwen3-8B",
reasoning: true,
compat: { thinkingFormat: "qwen-chat-template" },
} as unknown as Model<"openai-completions">;
const wrapped = wrapVllmProviderStream({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
extraParams: {},
thinkingLevel: "off",
model,
streamFn: baseStreamFn,
} as never);
expect(wrapped).toBeTypeOf("function");
void wrapped?.(model, { messages: [] } as Context, {});
expect(captured).toEqual({
chat_template_kwargs: {
enable_thinking: false,
preserve_thinking: true,
},
});
});
it("skips unconfigured vLLM and non-vLLM providers", () => {
expect(
wrapVllmProviderStream({
provider: "vllm",
modelId: "Qwen/Qwen3-8B",
extraParams: {},
model: {
api: "openai-completions",
provider: "vllm",
id: "Qwen/Qwen3-8B",
} as Model<"openai-completions">,
streamFn: undefined,
} as never),
).toBeUndefined();
expect(
wrapVllmProviderStream({
provider: "openai",
modelId: "gpt-5.4",
extraParams: {},
model: {
api: "openai-completions",
provider: "openai",
id: "gpt-5.4",
} as Model<"openai-completions">,
streamFn: undefined,
} as never),
).toBeUndefined();
});
it("registers for vLLM Nemotron when thinking is off", () => {
expect(
wrapVllmProviderStream({
provider: "vllm",
modelId: "nemotron-3-super",
extraParams: {},
thinkingLevel: "off",
model: {
api: "openai-completions",
provider: "vllm",
id: "nemotron-3-super",
} as Model<"openai-completions">,
streamFn: undefined,
} as never),
).toBeTypeOf("function");
expect(
wrapVllmProviderStream({
provider: "vllm",
modelId: "nemotron-3-super",
extraParams: {},
thinkingLevel: "low",
model: {
api: "openai-completions",
provider: "vllm",
id: "nemotron-3-super",
} as Model<"openai-completions">,
streamFn: undefined,
} as never),
).toBeUndefined();
});
});

125
extensions/vllm/stream.ts Normal file
View File

@@ -0,0 +1,125 @@
// Vllm plugin module implements stream behavior.
import type { StreamFn } from "openclaw/plugin-sdk/agent-core";
import type { ProviderWrapStreamFnContext } from "openclaw/plugin-sdk/plugin-entry";
import { normalizeProviderId } from "openclaw/plugin-sdk/provider-model-shared";
import {
createPayloadPatchStreamWrapper,
isOpenAICompatibleThinkingEnabled,
setQwenChatTemplateThinking,
} from "openclaw/plugin-sdk/provider-stream-shared";
import {
resolveVllmQwenThinkingFormatFromCompat,
type VllmQwenThinkingFormat,
} from "./thinking-policy.js";
type VllmThinkingLevel = ProviderWrapStreamFnContext["thinkingLevel"];
function isVllmProviderId(providerId: string): boolean {
return normalizeProviderId(providerId) === "vllm";
}
function resolveVllmQwenThinkingFormat(
ctx: Pick<ProviderWrapStreamFnContext, "model">,
): VllmQwenThinkingFormat | undefined {
return resolveVllmQwenThinkingFormatFromCompat(ctx.model?.compat);
}
function isVllmNemotronModel(model: { api?: unknown; provider?: unknown; id?: unknown }): boolean {
return (
model.api === "openai-completions" &&
typeof model.provider === "string" &&
normalizeProviderId(model.provider) === "vllm" &&
typeof model.id === "string" &&
/\bnemotron-3(?:[-_](?:nano|super|ultra))?\b/i.test(model.id)
);
}
function setNemotronThinkingOffChatTemplateKwargs(payload: Record<string, unknown>): void {
const defaults = {
enable_thinking: false,
force_nonempty_content: true,
};
const existing = payload.chat_template_kwargs;
payload.chat_template_kwargs =
existing && typeof existing === "object" && !Array.isArray(existing)
? {
...defaults,
...(existing as Record<string, unknown>),
}
: defaults;
}
export function createVllmQwenThinkingWrapper(params: {
baseStreamFn: StreamFn | undefined;
format: VllmQwenThinkingFormat;
thinkingLevel: VllmThinkingLevel;
}): StreamFn {
return createPayloadPatchStreamWrapper(
params.baseStreamFn,
({ payload: payloadObj, options }) => {
const enableThinking = isOpenAICompatibleThinkingEnabled({
thinkingLevel: params.thinkingLevel,
options,
});
if (params.format === "chat-template") {
setQwenChatTemplateThinking(payloadObj, enableThinking);
} else {
payloadObj.enable_thinking = enableThinking;
}
delete payloadObj.reasoning_effort;
delete payloadObj.reasoningEffort;
delete payloadObj.reasoning;
},
{
shouldPatch: ({ model }) => model.api === "openai-completions" && (model.reasoning ?? true),
},
);
}
export function createVllmProviderThinkingWrapper(params: {
baseStreamFn: StreamFn | undefined;
qwenFormat?: VllmQwenThinkingFormat;
thinkingLevel: VllmThinkingLevel;
}): StreamFn {
const qwenWrapped = params.qwenFormat
? createVllmQwenThinkingWrapper({
baseStreamFn: params.baseStreamFn,
format: params.qwenFormat,
thinkingLevel: params.thinkingLevel,
})
: params.baseStreamFn;
return createPayloadPatchStreamWrapper(
qwenWrapped,
({ payload: payloadObj }) => {
setNemotronThinkingOffChatTemplateKwargs(payloadObj);
},
{
shouldPatch: ({ model }) =>
model.api === "openai-completions" &&
params.thinkingLevel === "off" &&
isVllmNemotronModel(model),
},
);
}
export function wrapVllmProviderStream(ctx: ProviderWrapStreamFnContext): StreamFn | undefined {
if (!isVllmProviderId(ctx.provider) || (ctx.model && ctx.model.api !== "openai-completions")) {
return undefined;
}
const qwenFormat = resolveVllmQwenThinkingFormat(ctx);
const shouldHandleNemotron =
ctx.thinkingLevel === "off" &&
isVllmNemotronModel({
api: "openai-completions",
provider: ctx.provider,
id: ctx.modelId,
});
if (!qwenFormat && !shouldHandleNemotron) {
return undefined;
}
return createVllmProviderThinkingWrapper({
baseStreamFn: ctx.streamFn,
qwenFormat,
thinkingLevel: ctx.thinkingLevel,
});
}

View File

@@ -0,0 +1,66 @@
// Vllm plugin module implements thinking policy behavior.
import type {
ProviderDefaultThinkingPolicyContext,
ProviderThinkingProfile,
} from "openclaw/plugin-sdk/plugin-entry";
import { normalizeProviderId } from "openclaw/plugin-sdk/provider-model-shared";
export type VllmQwenThinkingFormat = "chat-template" | "top-level";
const VLLM_BINARY_THINKING_PROFILE = {
levels: [{ id: "off" }, { id: "low", label: "on" }],
defaultLevel: "off",
} satisfies ProviderThinkingProfile;
export function normalizeVllmQwenThinkingFormat(
value: unknown,
): VllmQwenThinkingFormat | undefined {
if (typeof value !== "string") {
return undefined;
}
const normalized = value.trim().toLowerCase().replace(/_/g, "-");
if (
normalized === "chat-template" ||
normalized === "chat-template-kwargs" ||
normalized === "chat-template-kwarg" ||
normalized === "chat-template-arguments" ||
normalized === "qwen-chat-template"
) {
return "chat-template";
}
if (
normalized === "top-level" ||
normalized === "enable-thinking" ||
normalized === "request-body" ||
normalized === "qwen"
) {
return "top-level";
}
return undefined;
}
export function resolveVllmQwenThinkingFormatFromCompat(
compat?: ProviderDefaultThinkingPolicyContext["compat"],
): VllmQwenThinkingFormat | undefined {
return normalizeVllmQwenThinkingFormat(compat?.thinkingFormat);
}
function isVllmNemotronThinkingModel(modelId: string): boolean {
return /\bnemotron-3(?:[-_](?:nano|super|ultra))?\b/i.test(modelId);
}
export function resolveThinkingProfile(
ctx: ProviderDefaultThinkingPolicyContext,
): ProviderThinkingProfile | null {
if (normalizeProviderId(ctx.provider) !== "vllm") {
return null;
}
if (ctx.reasoning === false) {
return null;
}
const qwenFormat = resolveVllmQwenThinkingFormatFromCompat(ctx.compat);
if (qwenFormat || (ctx.reasoning === true && isVllmNemotronThinkingModel(ctx.modelId))) {
return VLLM_BINARY_THINKING_PROFILE;
}
return null;
}

View File

@@ -0,0 +1,16 @@
{
"extends": "../tsconfig.package-boundary.base.json",
"compilerOptions": {
"rootDir": "."
},
"include": ["./*.ts", "./src/**/*.ts"],
"exclude": [
"./**/*.test.ts",
"./dist/**",
"./node_modules/**",
"./src/test-support/**",
"./src/**/*test-helpers.ts",
"./src/**/*test-harness.ts",
"./src/**/*test-support.ts"
]
}