Some checks failed
ClawSweeper Dispatch / dispatch (push) Has been cancelled
CodeQL / Security High (actions) (push) Has been cancelled
CodeQL / Security High (channel-runtime-boundary) (push) Has been cancelled
CodeQL / Security High (core-auth-secrets) (push) Has been cancelled
CodeQL / Security High (mcp-process-tool-boundary) (push) Has been cancelled
CodeQL / Security High (network-ssrf-boundary) (push) Has been cancelled
CodeQL / Security High (plugin-trust-boundary) (push) Has been cancelled
CodeQL / Security High (process-exec-boundary) (push) Has been cancelled
Docs Sync Publish Repo / sync-publish-repo (push) Has been cancelled
Docs / docs (push) Has been cancelled
OpenClaw Stable Main Closeout / Resolve stable release closeout inputs (push) Has been cancelled
OpenClaw Stable Main Closeout / Verify stable main closeout (push) Has been cancelled
Workflow Sanity / no-tabs (push) Has been cancelled
Workflow Sanity / actionlint (push) Has been cancelled
Workflow Sanity / generated-doc-baselines (push) Has been cancelled
CI / runner-admission (push) Has been cancelled
CI / preflight (push) Has been cancelled
CI / security-fast (push) Has been cancelled
CI / pnpm-store-warmup (push) Has been cancelled
CI / build-artifacts (push) Has been cancelled
CI / native-i18n (push) Has been cancelled
CI / ${{ matrix.check_name }} (push) Has been cancelled
CI / ${{ matrix.checkName }} (push) Has been cancelled
CI / checks-node-compat-node22 (push) Has been cancelled
CI / check-bundled-channel-config-metadata (push) Has been cancelled
CI / check-dependencies (push) Has been cancelled
CI / check-guards (push) Has been cancelled
CI / check-lint (push) Has been cancelled
CI / check-prod-types (push) Has been cancelled
CI / check-shrinkwrap (push) Has been cancelled
CI / check-test-types (push) Has been cancelled
CI / check-additional-boundaries-a (push) Has been cancelled
CI / check-additional-boundaries-bcd (push) Has been cancelled
CI / check-additional-extension-bundled (push) Has been cancelled
CI / check-additional-extension-channels (push) Has been cancelled
CI / check-additional-extension-package-boundary (push) Has been cancelled
CI / check-additional-runtime-topology-architecture (push) Has been cancelled
CI / check-session-accessor-boundary (push) Has been cancelled
CI / check-session-transcript-reader-boundary (push) Has been cancelled
CI / check-docs (push) Has been cancelled
CI / skills-python (push) Has been cancelled
CI / macos-swift (push) Has been cancelled
CI / ios-build (push) Has been cancelled
CI / ci-timings-summary (push) Has been cancelled
Native App Locale Refresh / Refresh native fa (push) Has been cancelled
Native App Locale Refresh / Refresh native fr (push) Has been cancelled
Native App Locale Refresh / Refresh native hi (push) Has been cancelled
Native App Locale Refresh / Refresh native id (push) Has been cancelled
Native App Locale Refresh / Refresh native it (push) Has been cancelled
Native App Locale Refresh / Refresh native ja-JP (push) Has been cancelled
Control UI Locale Refresh / plan (push) Has been cancelled
Control UI Locale Refresh / Refresh ${{ matrix.locale }} (push) Has been cancelled
Control UI Locale Refresh / Commit control UI locale refresh (push) Has been cancelled
Live Media Runner Image / Build live media runner image (push) Has been cancelled
Native App Locale Refresh / Refresh native ar (push) Has been cancelled
Native App Locale Refresh / Refresh native de (push) Has been cancelled
Native App Locale Refresh / Refresh native es (push) Has been cancelled
Native App Locale Refresh / Refresh native ko (push) Has been cancelled
Native App Locale Refresh / Refresh native nl (push) Has been cancelled
Native App Locale Refresh / Refresh native pl (push) Has been cancelled
Native App Locale Refresh / Refresh native pt-BR (push) Has been cancelled
Native App Locale Refresh / Refresh native ru (push) Has been cancelled
Native App Locale Refresh / Refresh native sv (push) Has been cancelled
Native App Locale Refresh / Refresh native th (push) Has been cancelled
Native App Locale Refresh / Refresh native tr (push) Has been cancelled
Native App Locale Refresh / Refresh native uk (push) Has been cancelled
Native App Locale Refresh / Refresh native vi (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-CN (push) Has been cancelled
Native App Locale Refresh / Refresh native zh-TW (push) Has been cancelled
Native App Locale Refresh / Commit native locale refresh (push) Has been cancelled
Plugin Init Scaffold Validation / Validate provider scaffold (push) Has been cancelled
Plugin NPM Release / preview_plugins_npm (push) Has been cancelled
Plugin NPM Release / Validate release publish approval (push) Has been cancelled
Plugin NPM Release / preview_plugin_pack (push) Has been cancelled
Plugin NPM Release / publish_plugins_npm (push) Has been cancelled
Sandbox Common Smoke / sandbox-common-smoke (push) Has been cancelled
Website Installer Sync / static (push) Has been cancelled
Website Installer Sync / linux-docker (push) Has been cancelled
Website Installer Sync / macos-installer (push) Has been cancelled
Website Installer Sync / windows-installer (push) Has been cancelled
Website Installer Sync / sync-website (push) Has been cancelled
Adolf is a fork/vendored clone of github.com/openclaw/openclaw (v2026.6.11), free to diverge. Tree copied sans upstream .git; upstream remote added for future syncs. Node pinned to 24 (.nvmrc); engines already require >=22.19. Preserves docs/ARCHITECTURE.md. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01LeqyaxJF2nbRXJtae2kNB2
347 lines
10 KiB
TypeScript
347 lines
10 KiB
TypeScript
// Llm Task tests cover llm task tool plugin behavior.
|
|
import { afterAll, beforeEach, describe, expect, it, vi } from "vitest";
|
|
|
|
vi.mock("../api.js", async () => {
|
|
const actual = await vi.importActual<typeof import("../api.js")>("../api.js");
|
|
return {
|
|
...actual,
|
|
resolvePreferredOpenClawTmpDir: () => "/tmp",
|
|
};
|
|
});
|
|
|
|
afterAll(() => {
|
|
vi.doUnmock("../api.js");
|
|
vi.resetModules();
|
|
});
|
|
|
|
import { createLlmTaskTool } from "./llm-task-tool.js";
|
|
|
|
const runEmbeddedAgent = vi.fn(async () => ({
|
|
meta: { startedAt: Date.now() },
|
|
payloads: [{ text: "{}" }],
|
|
}));
|
|
|
|
const resolveThinkingPolicy = vi.fn(() => ({
|
|
levels: [
|
|
{ id: "off", label: "off" },
|
|
{ id: "minimal", label: "minimal" },
|
|
{ id: "low", label: "low" },
|
|
{ id: "medium", label: "medium" },
|
|
{ id: "high", label: "high" },
|
|
],
|
|
}));
|
|
|
|
const normalizeThinkingLevel = vi.fn((raw?: string | null) => {
|
|
const value = raw?.trim().toLowerCase();
|
|
if (!value) {
|
|
return undefined;
|
|
}
|
|
if (value === "on") {
|
|
return "low";
|
|
}
|
|
if (["off", "minimal", "low", "medium", "high", "xhigh", "adaptive", "max"].includes(value)) {
|
|
return value;
|
|
}
|
|
return undefined;
|
|
});
|
|
|
|
function fakeApi(overrides: any = {}) {
|
|
return {
|
|
id: "llm-task",
|
|
name: "llm-task",
|
|
source: "test",
|
|
config: {
|
|
agents: { defaults: { workspace: "/tmp", model: { primary: "openai/gpt-5.5" } } },
|
|
},
|
|
pluginConfig: {},
|
|
runtime: {
|
|
version: "test",
|
|
agent: {
|
|
defaults: { provider: "openai", model: "gpt-5.5" },
|
|
runEmbeddedAgent,
|
|
resolveThinkingPolicy,
|
|
normalizeThinkingLevel,
|
|
},
|
|
},
|
|
logger: { debug() {}, info() {}, warn() {}, error() {} },
|
|
registerTool() {},
|
|
...overrides,
|
|
};
|
|
}
|
|
|
|
function mockEmbeddedRunJson(payload: unknown) {
|
|
(runEmbeddedAgent as any).mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: JSON.stringify(payload) }],
|
|
});
|
|
}
|
|
|
|
function resetRunnerMocks() {
|
|
runEmbeddedAgent.mockReset();
|
|
runEmbeddedAgent.mockImplementation(async () => ({
|
|
meta: { startedAt: Date.now() },
|
|
payloads: [{ text: "{}" }],
|
|
}));
|
|
resolveThinkingPolicy.mockClear();
|
|
normalizeThinkingLevel.mockClear();
|
|
}
|
|
|
|
async function executeEmbeddedRun(input: Record<string, unknown>) {
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
await tool.execute("id", input);
|
|
return (runEmbeddedAgent as any).mock.calls[0]?.[0];
|
|
}
|
|
|
|
describe("llm-task tool (json-only)", () => {
|
|
beforeEach(() => {
|
|
resetRunnerMocks();
|
|
});
|
|
|
|
it("returns parsed json", async () => {
|
|
(runEmbeddedAgent as any).mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: JSON.stringify({ foo: "bar" }) }],
|
|
});
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
const res = await tool.execute("id", { prompt: "return foo" });
|
|
expect((res as any).details.json).toEqual({ foo: "bar" });
|
|
});
|
|
|
|
it("strips fenced json", async () => {
|
|
(runEmbeddedAgent as any).mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: '```json\n{"ok":true}\n```' }],
|
|
});
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
const res = await tool.execute("id", { prompt: "return ok" });
|
|
expect((res as any).details.json).toEqual({ ok: true });
|
|
});
|
|
|
|
it("validates schema", async () => {
|
|
(runEmbeddedAgent as any).mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: JSON.stringify({ foo: "bar" }) }],
|
|
});
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
const schema = {
|
|
type: "object",
|
|
properties: { foo: { type: "string" } },
|
|
required: ["foo"],
|
|
additionalProperties: false,
|
|
};
|
|
const res = await tool.execute("id", { prompt: "return foo", schema });
|
|
expect((res as any).details.json).toEqual({ foo: "bar" });
|
|
});
|
|
|
|
it("validates caller schemas with repeated $id independently across calls", async () => {
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
(runEmbeddedAgent as any)
|
|
.mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: JSON.stringify({ foo: "bar" }) }],
|
|
})
|
|
.mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: JSON.stringify({ count: 1 }) }],
|
|
});
|
|
|
|
await expect(
|
|
tool.execute("id", {
|
|
prompt: "return foo",
|
|
schema: {
|
|
$id: "https://example.test/llm-task-result",
|
|
type: "object",
|
|
properties: { foo: { type: "string" } },
|
|
required: ["foo"],
|
|
additionalProperties: false,
|
|
},
|
|
}),
|
|
).resolves.toEqual({
|
|
content: [{ type: "text", text: '{\n "foo": "bar"\n}' }],
|
|
details: { json: { foo: "bar" }, provider: "openai", model: "gpt-5.5" },
|
|
});
|
|
|
|
await expect(
|
|
tool.execute("id", {
|
|
prompt: "return count",
|
|
schema: {
|
|
$id: "https://example.test/llm-task-result",
|
|
type: "object",
|
|
properties: { count: { type: "number" } },
|
|
required: ["count"],
|
|
additionalProperties: false,
|
|
},
|
|
}),
|
|
).resolves.toEqual({
|
|
content: [{ type: "text", text: '{\n "count": 1\n}' }],
|
|
details: { json: { count: 1 }, provider: "openai", model: "gpt-5.5" },
|
|
});
|
|
});
|
|
|
|
it("throws on invalid json", async () => {
|
|
(runEmbeddedAgent as any).mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: "not-json" }],
|
|
});
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
await expect(tool.execute("id", { prompt: "x" })).rejects.toThrow(/invalid json/i);
|
|
});
|
|
|
|
it("throws on schema mismatch", async () => {
|
|
(runEmbeddedAgent as any).mockResolvedValueOnce({
|
|
meta: {},
|
|
payloads: [{ text: JSON.stringify({ foo: 1 }) }],
|
|
});
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
const schema = { type: "object", properties: { foo: { type: "string" } }, required: ["foo"] };
|
|
await expect(tool.execute("id", { prompt: "x", schema })).rejects.toThrow(/match schema/i);
|
|
});
|
|
|
|
it("passes provider/model overrides to embedded runner", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({
|
|
prompt: "x",
|
|
provider: "anthropic",
|
|
model: "claude-4-sonnet",
|
|
});
|
|
expect(call.provider).toBe("anthropic");
|
|
expect(call.model).toBe("claude-4-sonnet");
|
|
});
|
|
|
|
it("accepts model overrides that already include the selected provider prefix", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({
|
|
prompt: "x",
|
|
provider: "anthropic",
|
|
model: "anthropic/claude-4-sonnet",
|
|
});
|
|
expect(call.provider).toBe("anthropic");
|
|
expect(call.model).toBe("claude-4-sonnet");
|
|
});
|
|
|
|
it("resolves configured model aliases before dispatching the embedded run", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const tool = createLlmTaskTool(
|
|
fakeApi({
|
|
config: {
|
|
agents: {
|
|
defaults: {
|
|
workspace: "/tmp",
|
|
model: { primary: "anthropic/claude-sonnet-4-6" },
|
|
models: {
|
|
"google/gemini-3-flash-preview": { alias: "gemini-flash" },
|
|
},
|
|
},
|
|
},
|
|
},
|
|
}),
|
|
);
|
|
|
|
await tool.execute("id", { prompt: "x", model: "gemini-flash" });
|
|
|
|
const call = (runEmbeddedAgent as any).mock.calls[0]?.[0];
|
|
expect(call.provider).toBe("google");
|
|
expect(call.model).toBe("gemini-3-flash-preview");
|
|
});
|
|
|
|
it("passes thinking override to embedded runner", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({ prompt: "x", thinking: "high" });
|
|
expect(call.thinkLevel).toBe("high");
|
|
expect(resolveThinkingPolicy).toHaveBeenCalledWith({
|
|
provider: "openai",
|
|
model: "gpt-5.5",
|
|
});
|
|
});
|
|
|
|
it("normalizes thinking aliases", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({ prompt: "x", thinking: "on" });
|
|
expect(call.thinkLevel).toBe("low");
|
|
});
|
|
|
|
it("throws on invalid thinking level", async () => {
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
await expect(tool.execute("id", { prompt: "x", thinking: "banana" })).rejects.toThrow(
|
|
/invalid thinking level/i,
|
|
);
|
|
expect(runEmbeddedAgent).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it("throws on unsupported xhigh thinking level", async () => {
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
await expect(tool.execute("id", { prompt: "x", thinking: "xhigh" })).rejects.toThrow(
|
|
/not supported/i,
|
|
);
|
|
});
|
|
|
|
it("does not pass thinkLevel when thinking is omitted", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({ prompt: "x" });
|
|
expect(call.thinkLevel).toBeUndefined();
|
|
});
|
|
|
|
it("enforces allowedModels", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const tool = createLlmTaskTool(
|
|
fakeApi({ pluginConfig: { allowedModels: ["openai/gpt-5.5"] } }),
|
|
);
|
|
await expect(
|
|
tool.execute("id", { prompt: "x", provider: "anthropic", model: "claude-4-sonnet" }),
|
|
).rejects.toThrow(/not allowed/i);
|
|
});
|
|
|
|
it("disables tools for embedded run", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({ prompt: "x" });
|
|
expect(call.disableTools).toBe(true);
|
|
});
|
|
|
|
it("rejects malformed numeric run options before dispatch", async () => {
|
|
const tool = createLlmTaskTool(fakeApi());
|
|
|
|
await expect(tool.execute("id", { prompt: "x", temperature: Number.NaN })).rejects.toThrow(
|
|
"temperature must be a finite number",
|
|
);
|
|
await expect(tool.execute("id", { prompt: "x", maxTokens: 0 })).rejects.toThrow(
|
|
"maxTokens must be a positive integer",
|
|
);
|
|
await expect(tool.execute("id", { prompt: "x", timeoutMs: "4096.5" })).rejects.toThrow(
|
|
"timeoutMs must be a positive integer",
|
|
);
|
|
expect(runEmbeddedAgent).not.toHaveBeenCalled();
|
|
});
|
|
|
|
it("passes valid numeric run options before dispatch", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({
|
|
prompt: "x",
|
|
temperature: 0.2,
|
|
maxTokens: 512,
|
|
timeoutMs: 10_000,
|
|
});
|
|
|
|
expect(call.timeoutMs).toBe(10_000);
|
|
expect(call.streamParams).toEqual({
|
|
temperature: 0.2,
|
|
maxTokens: 512,
|
|
});
|
|
});
|
|
|
|
it("normalizes numeric string run options before dispatch", async () => {
|
|
mockEmbeddedRunJson({ ok: true });
|
|
const call = await executeEmbeddedRun({
|
|
prompt: "x",
|
|
temperature: "0.2",
|
|
maxTokens: "512",
|
|
timeoutMs: "10000",
|
|
});
|
|
|
|
expect(call.timeoutMs).toBe(10_000);
|
|
expect(call.streamParams).toEqual({
|
|
temperature: 0.2,
|
|
maxTokens: 512,
|
|
});
|
|
});
|
|
});
|