Two related changes to the /usage quota route, committed together because they are entangled in the same code path. 1. Stop refreshing the Kimi OAuth token from this route (kb#87). Was already present as uncommitted working-tree WIP, not authored in this commit's session. Kimi rotates the refresh_token on every refresh (single-use), so an independent refresh here invalidated the copy the CLI's creds file holds -> the CLI's next refresh failed invalid_grant and wiped the whole login (the recurring Adolf logout, incl. the 2026-07-17 06:15 wipe / task #86). Removes KIMI_OAUTH_HOST, KIMI_CLIENT_ID, refreshKimiToken() and the kimiMemToken cache; the CLI is now the sole refresher and this route only ever READS. 2. Serve the last good reading when the token is stale, instead of erroring. Measured 2026-07-22: the access token's expires_in is 900s, so it is only valid for 15 minutes after the CLI last refreshed it -- i.e. only within 15 minutes of an actual Adolf turn. Adolf is idle most of the day, so bare reads failed far more often than they succeeded and quota gating was effectively blind. /usage now caches every success and, on a stale token, returns that payload with stale/as_of/age_s/stale_reason so callers can judge whether it is fresh enough. Cache is mirrored to the workspace volume so it survives restarts, and writes are best-effort so an unwritable volume cannot break the route. Auth behaviour is unchanged by this half. Payload shape is additive only -- existing kimi-usage -q filters keep working. Verified live after rebuild: fresh read returns weekly 16% / 5h 5% with stale:false; cache file written to /workspace/.adolf-llm/usage-cache.json; kimi-usage -q '.window_5h.pct' returns 5. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
778 lines
30 KiB
JavaScript
778 lines
30 KiB
JavaScript
const http = require('http');
|
|
const fs = require('fs');
|
|
const path = require('path');
|
|
const crypto = require('crypto');
|
|
const { spawn } = require('child_process');
|
|
|
|
const PORT = 8010;
|
|
const MODEL_ID = 'adolf';
|
|
const TIMEOUT_MS = 15 * 60 * 1000;
|
|
|
|
const WORKSPACE = '/workspace';
|
|
const CONV_ROOT = path.join(WORKSPACE, 'conversations');
|
|
const STATE_DIR = path.join(WORKSPACE, '.adolf-llm');
|
|
const MAP_FILE = path.join(STATE_DIR, 'sessions.json');
|
|
const MAX_ENTRIES = 1000; // prune oldest beyond this
|
|
|
|
fs.mkdirSync(CONV_ROOT, { recursive: true });
|
|
fs.mkdirSync(STATE_DIR, { recursive: true });
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Shared MCP layer (Gate 1). Kimi Code CLI has NO `--mcp-config-file` flag and
|
|
// no `kimi mcp` subcommand; it auto-discovers a project-root `.mcp.json` by
|
|
// walking up from its cwd to the nearest `.git` (falling back to cwd itself
|
|
// when none is found). So we drop a `.mcp.json` into each session's working
|
|
// directory before spawning kimi.
|
|
//
|
|
// Single source of truth: `/shared-mcp.json` (mounted read-only from the repo
|
|
// root's `shared-mcp.json`, the same file P6 wires into OpenClaw's own
|
|
// `mcp.servers` registry). Adding a server is then a one-file change — no
|
|
// server list is hardcoded here anymore.
|
|
//
|
|
// Gate-1 transport finding (P5, verified by decompiling the installed
|
|
// @moonshot-ai/kimi-code package, packages/agent-core/src/config/schema.ts's
|
|
// McpServerConfigSchema): Kimi's own field name for remote MCP servers is
|
|
// `transport` (literal "stdio" | "http" | "sse"), not `type`. When `transport`
|
|
// is omitted, Kimi's config preprocessor infers it from shape: `command` ->
|
|
// "stdio", `url` -> "http" (never "sse" — sse requires an explicit
|
|
// `transport: "sse"`). It does NOT recognize a `type` key at all; unknown keys
|
|
// are silently stripped by the (non-strict) zod schema.
|
|
// OpenClaw's own canonical `mcp.servers` schema (docs/gateway/
|
|
// configuration-reference.md) uses different literals for the same
|
|
// transport: `transport: "streamable-http"` or `"sse"`, with `type: "http"`
|
|
// documented as a *CLI-native alias* that `openclaw mcp set` / `openclaw
|
|
// doctor --fix` normalize into canonical `transport: "streamable-http"`.
|
|
// So the two consumers disagree on the literal value for HTTP streaming
|
|
// ("http" vs "streamable-http") under the same field name `transport` --
|
|
// writing `transport` explicitly in shared-mcp.json would satisfy at most one
|
|
// side. `type: "http"` is the one shape both sides tolerate today: Kimi
|
|
// ignores the unrecognized `type` key and correctly infers transport "http"
|
|
// from the `url` field alone; OpenClaw recognizes `type` as its documented
|
|
// alias and normalizes it on its own terms (P6 concern, not touched here).
|
|
// Hence shared-mcp.json intentionally keeps `"type": "http"` for both cognee
|
|
// and openclaw-tools rather than switching to `transport`.
|
|
let SHARED_MCP_SERVERS = {};
|
|
try {
|
|
const raw = fs.readFileSync('/shared-mcp.json', 'utf8');
|
|
SHARED_MCP_SERVERS = JSON.parse(raw).mcpServers || {};
|
|
} catch (err) {
|
|
console.error(`shared-mcp.json not loaded (${err.message}); sessions will get no shared MCP servers`);
|
|
}
|
|
|
|
function writeMcpConfig(dir) {
|
|
const cfg = { mcpServers: SHARED_MCP_SERVERS };
|
|
fs.writeFileSync(path.join(dir, '.mcp.json'), JSON.stringify(cfg, null, 2));
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Memory lives at the OpenClaw layer, not here (P8). The Adolf gateway loads
|
|
// the `cognee-memory` OpenClaw plugin, which owns all memory touchpoints:
|
|
// - before_prompt_build => LLM-free cognee graph recall, injected into the
|
|
// prompt this wrapper then receives from OpenClaw.
|
|
// - agent_end => raw `add` of the turn to cognee.
|
|
// - background sweep => async `cognify` on cognee-llm (Kimi).
|
|
// - cognee_recall tool + cognee-mcp `recall` for on-demand deep queries.
|
|
// This wrapper is therefore a dumb model endpoint again: it must never call
|
|
// cognee itself. The former cogneeSearch/cogneeAdd stubs (and their call sites)
|
|
// were deleted when the plugin took over (P8).
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Persistent conversation -> Kimi session map.
|
|
// Primary key: `chat:<chat_id>` parsed from OpenClaw's "Conversation info" block
|
|
// (Gate 2). Fallback key: `hist:<sha256(prior history)>` when no chat_id is
|
|
// present (e.g. webchat surface / future OpenClaw layout change).
|
|
// value = { convId, sessionId, dir, ts }
|
|
let sessionMap = {};
|
|
try {
|
|
sessionMap = JSON.parse(fs.readFileSync(MAP_FILE, 'utf8'));
|
|
} catch {
|
|
sessionMap = {};
|
|
}
|
|
|
|
let writeQueue = Promise.resolve();
|
|
function persistMap() {
|
|
// prune to the MAX_ENTRIES most-recently-used before writing
|
|
const keys = Object.keys(sessionMap);
|
|
if (keys.length > MAX_ENTRIES) {
|
|
keys.sort((a, b) => (sessionMap[a].ts || 0) - (sessionMap[b].ts || 0));
|
|
for (const k of keys.slice(0, keys.length - MAX_ENTRIES)) delete sessionMap[k];
|
|
}
|
|
const snapshot = JSON.stringify(sessionMap);
|
|
writeQueue = writeQueue.then(
|
|
() => fs.promises.writeFile(MAP_FILE, snapshot),
|
|
() => fs.promises.writeFile(MAP_FILE, snapshot),
|
|
);
|
|
return writeQueue;
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Message helpers.
|
|
function textOf(msg) {
|
|
const c = msg.content;
|
|
if (Array.isArray(c)) return c.map(p => (typeof p === 'string' ? p : p.text || '')).join('\n');
|
|
return c == null ? '' : String(c);
|
|
}
|
|
|
|
// only user/assistant turns define conversation identity (system is constant)
|
|
function convTurns(messages) {
|
|
return messages.filter(m => m.role === 'user' || m.role === 'assistant');
|
|
}
|
|
|
|
function historyKey(turns) {
|
|
const norm = turns.map(m => ({ role: m.role, text: textOf(m).trim() }));
|
|
return crypto.createHash('sha256').update(JSON.stringify(norm)).digest('hex');
|
|
}
|
|
|
|
function renderTranscript(turns) {
|
|
return turns
|
|
.map(m => `${m.role === 'assistant' ? 'Assistant' : 'User'}: ${textOf(m)}`)
|
|
.join('\n\n');
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Gate 2 — parse the stable chat_id out of OpenClaw's untrusted-metadata block.
|
|
// OpenClaw injects, into the *user-role* content, a block that looks like:
|
|
// Conversation info (untrusted metadata):
|
|
// ```json
|
|
// { "chat_id": "matrix:!room:server", "message_id": "...", ... }
|
|
// ```
|
|
// We grep on the label string (never a fixed line offset) then pull the first
|
|
// balanced JSON object after it and read chat_id. Robust to edited/truncated
|
|
// history, which is exactly why it beats a history hash for the common case.
|
|
const CONV_INFO_LABEL = 'Conversation info (untrusted metadata):';
|
|
|
|
function extractBalancedJson(str, from) {
|
|
const start = str.indexOf('{', from);
|
|
if (start === -1) return null;
|
|
let depth = 0;
|
|
let inStr = false;
|
|
let esc = false;
|
|
for (let i = start; i < str.length; i++) {
|
|
const ch = str[i];
|
|
if (inStr) {
|
|
if (esc) esc = false;
|
|
else if (ch === '\\') esc = true;
|
|
else if (ch === '"') inStr = false;
|
|
continue;
|
|
}
|
|
if (ch === '"') inStr = true;
|
|
else if (ch === '{') depth++;
|
|
else if (ch === '}') {
|
|
depth--;
|
|
if (depth === 0) return str.slice(start, i + 1);
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
function extractChatId(userMsg) {
|
|
const text = textOf(userMsg);
|
|
const at = text.indexOf(CONV_INFO_LABEL);
|
|
if (at === -1) return null;
|
|
const jsonStr = extractBalancedJson(text, at + CONV_INFO_LABEL.length);
|
|
if (!jsonStr) return null;
|
|
try {
|
|
const obj = JSON.parse(jsonStr);
|
|
const id = obj.chat_id;
|
|
return typeof id === 'string' && id ? id : null;
|
|
} catch {
|
|
return null;
|
|
}
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Gate 3 — media. Persist inbound image parts into the session dir and return
|
|
// relative path references; the CLI autonomously calls its built-in
|
|
// ReadMediaFile tool on referenced paths (no flag/placeholder syntax needed).
|
|
const MIME_EXT = {
|
|
'image/png': 'png',
|
|
'image/jpeg': 'jpg',
|
|
'image/jpg': 'jpg',
|
|
'image/webp': 'webp',
|
|
'image/gif': 'gif',
|
|
'image/bmp': 'bmp',
|
|
'image/heic': 'heic',
|
|
'image/heif': 'heif',
|
|
};
|
|
|
|
function extFromMime(mime) {
|
|
return MIME_EXT[(mime || '').toLowerCase()] || 'img';
|
|
}
|
|
|
|
// Persist one image_url part; returns "./img_N.ext" or null if it couldn't.
|
|
async function persistImage(url, dir, n) {
|
|
if (typeof url !== 'string' || !url) return null;
|
|
if (url.startsWith('data:')) {
|
|
const m = /^data:([^;,]+)?(;base64)?,(.*)$/s.exec(url);
|
|
if (!m) return null;
|
|
const mime = m[1] || 'application/octet-stream';
|
|
const isB64 = !!m[2];
|
|
const ext = extFromMime(mime);
|
|
const name = `img_${n}.${ext}`;
|
|
const buf = isB64
|
|
? Buffer.from(m[3], 'base64')
|
|
: Buffer.from(decodeURIComponent(m[3]), 'utf8');
|
|
fs.writeFileSync(path.join(dir, name), buf);
|
|
return `./${name}`;
|
|
}
|
|
// Remote URL: best-effort fetch so the CLI gets a local path to ReadMediaFile.
|
|
// On any failure fall back to handing the raw URL to the model as text.
|
|
if (/^https?:\/\//i.test(url)) {
|
|
try {
|
|
const resp = await fetch(url);
|
|
if (!resp.ok) return url;
|
|
const mime = resp.headers.get('content-type') || '';
|
|
const ext = extFromMime(mime.split(';')[0].trim());
|
|
const name = `img_${n}.${ext}`;
|
|
const buf = Buffer.from(await resp.arrayBuffer());
|
|
fs.writeFileSync(path.join(dir, name), buf);
|
|
return `./${name}`;
|
|
} catch {
|
|
return url;
|
|
}
|
|
}
|
|
return null;
|
|
}
|
|
|
|
// Build the prompt for the current user turn: join text parts, persist any
|
|
// image parts, append path references.
|
|
async function buildPrompt(userMsg, dir) {
|
|
const content = userMsg.content;
|
|
const textParts = [];
|
|
const imageRefs = [];
|
|
|
|
if (Array.isArray(content)) {
|
|
let n = 0;
|
|
for (const part of content) {
|
|
if (typeof part === 'string') {
|
|
textParts.push(part);
|
|
} else if (part && (part.type === 'text' || typeof part.text === 'string')) {
|
|
textParts.push(part.text || '');
|
|
} else if (part && part.type === 'image_url' && part.image_url && part.image_url.url) {
|
|
n++;
|
|
const ref = await persistImage(part.image_url.url, dir, n);
|
|
if (ref) imageRefs.push(ref);
|
|
} else if (part && part.type === 'input_image' && (part.image_url || part.url)) {
|
|
n++;
|
|
const u = typeof part.image_url === 'string' ? part.image_url : part.url;
|
|
const ref = await persistImage(u, dir, n);
|
|
if (ref) imageRefs.push(ref);
|
|
}
|
|
}
|
|
} else {
|
|
textParts.push(content == null ? '' : String(content));
|
|
}
|
|
|
|
let prompt = textParts.join('\n');
|
|
if (imageRefs.length) {
|
|
prompt += '\n\n' + imageRefs.map(r => `See attached image: ${r}`).join('\n');
|
|
}
|
|
return prompt;
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Kimi invocation with REAL streaming. Parses `--output-format stream-json`
|
|
// incrementally: each complete stdout line is one JSON object.
|
|
// {"role":"assistant","content":"..."} -> emit as a delta
|
|
// {"type":"session.resume_hint","session_id":"..."} -> capture session id
|
|
// onDelta(chunk) is called per assistant content fragment as it arrives.
|
|
// Resolves { text, sessionId } once the process closes.
|
|
function runKimi({ prompt, cwd, resumeId, onDelta, signal }) {
|
|
return new Promise((resolve, reject) => {
|
|
if (signal?.aborted) { reject(new Error('aborted before start')); return; }
|
|
const args = [];
|
|
if (resumeId) args.push('-r', resumeId);
|
|
args.push('-p', prompt, '--output-format', 'stream-json');
|
|
|
|
const child = spawn('kimi', args, { cwd, timeout: TIMEOUT_MS });
|
|
|
|
let buf = '';
|
|
let stderr = '';
|
|
const parts = [];
|
|
let sessionId = null;
|
|
let settled = false;
|
|
let aborted = false;
|
|
|
|
// If the caller aborts (the gateway/client disconnected — e.g. its idle
|
|
// watchdog gave up), kill the child so it doesn't keep grinding an
|
|
// orphaned agent turn to completion, wasting Kimi quota and streaming into
|
|
// a dead socket. SIGTERM first, hard SIGKILL if it lingers.
|
|
const onAbort = () => {
|
|
aborted = true;
|
|
try { child.kill('SIGTERM'); } catch {}
|
|
setTimeout(() => { try { child.kill('SIGKILL'); } catch {} }, 3000).unref();
|
|
};
|
|
if (signal) signal.addEventListener('abort', onAbort, { once: true });
|
|
|
|
function handleLine(line) {
|
|
const t = line.trim();
|
|
if (!t) return;
|
|
let obj;
|
|
try { obj = JSON.parse(t); } catch { return; }
|
|
if (obj.role === 'assistant' && typeof obj.content === 'string' && obj.content) {
|
|
parts.push(obj.content);
|
|
if (onDelta) onDelta(obj.content);
|
|
}
|
|
if (obj.type === 'session.resume_hint' && obj.session_id) sessionId = obj.session_id;
|
|
}
|
|
|
|
child.stdout.on('data', d => {
|
|
buf += d;
|
|
let nl;
|
|
while ((nl = buf.indexOf('\n')) !== -1) {
|
|
const line = buf.slice(0, nl);
|
|
buf = buf.slice(nl + 1);
|
|
handleLine(line);
|
|
}
|
|
});
|
|
child.stderr.on('data', d => { stderr += d; });
|
|
|
|
child.on('error', err => {
|
|
if (settled) return;
|
|
settled = true;
|
|
if (signal) signal.removeEventListener('abort', onAbort);
|
|
reject(err);
|
|
});
|
|
child.on('close', code => {
|
|
if (settled) return;
|
|
settled = true;
|
|
if (signal) signal.removeEventListener('abort', onAbort);
|
|
if (buf) handleLine(buf); // flush any trailing partial line
|
|
const text = parts.join('').trim();
|
|
if (aborted) {
|
|
reject(new Error('aborted: client disconnected'));
|
|
} else if (!text && code !== 0) {
|
|
reject(new Error(`kimi exited ${code}: ${stderr.slice(0, 2000)}`));
|
|
} else {
|
|
resolve({ text, sessionId });
|
|
}
|
|
});
|
|
});
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// One turn: resolve session (chat_id primary, history-hash fallback), persist
|
|
// media + .mcp.json, run kimi (streaming through onDelta), record the mapping,
|
|
// and fire the async cognee ingest. Returns { text }.
|
|
async function handleTurn(messages, onDelta, signal) {
|
|
const turns = convTurns(messages);
|
|
let lastUserIdx = -1;
|
|
for (let i = turns.length - 1; i >= 0; i--) {
|
|
if (turns[i].role === 'user') { lastUserIdx = i; break; }
|
|
}
|
|
if (lastUserIdx === -1) throw new Error('no user message found');
|
|
|
|
const userMsg = turns[lastUserIdx];
|
|
const prior = turns.slice(0, lastUserIdx);
|
|
const chatId = extractChatId(userMsg);
|
|
|
|
// Resolve the session key + working dir + resume id.
|
|
let key;
|
|
let convId;
|
|
let dir;
|
|
let resumeId = null;
|
|
let reseed = false; // when true, prompt with the full transcript to rebuild continuity
|
|
|
|
if (chatId) {
|
|
key = `chat:${chatId}`;
|
|
const entry = sessionMap[key];
|
|
if (entry) {
|
|
convId = entry.convId;
|
|
dir = entry.dir;
|
|
resumeId = entry.sessionId;
|
|
} else {
|
|
convId = crypto.randomUUID();
|
|
dir = path.join(CONV_ROOT, convId);
|
|
// First time we see this chat_id but history exists (server restart / lost
|
|
// map): reseed the fresh session with the transcript so context survives.
|
|
reseed = prior.length > 0;
|
|
}
|
|
} else {
|
|
// Fallback: no chat_id -> forward history-hash mapping (kimi-agent style).
|
|
if (prior.length === 0) {
|
|
convId = crypto.randomUUID();
|
|
dir = path.join(CONV_ROOT, convId);
|
|
} else {
|
|
const entry = sessionMap[`hist:${historyKey(prior)}`];
|
|
if (entry) {
|
|
convId = entry.convId;
|
|
dir = entry.dir;
|
|
resumeId = entry.sessionId;
|
|
} else {
|
|
convId = crypto.randomUUID();
|
|
dir = path.join(CONV_ROOT, convId);
|
|
reseed = true;
|
|
}
|
|
}
|
|
}
|
|
|
|
fs.mkdirSync(dir, { recursive: true });
|
|
writeMcpConfig(dir); // Gate 1: shared MCP via project-root .mcp.json
|
|
|
|
let prompt;
|
|
if (reseed) {
|
|
// Rebuild the whole conversation for a fresh session, plus current media.
|
|
const base = renderTranscript(turns.slice(0, lastUserIdx + 1));
|
|
const media = await buildPrompt(userMsg, dir);
|
|
// buildPrompt already includes the current user text; for reseed we want the
|
|
// transcript to carry it, so only append image refs.
|
|
prompt = base;
|
|
const extra = media.replace(textOf(userMsg), '').trim();
|
|
if (extra) prompt += `\n\n${extra}`;
|
|
} else {
|
|
prompt = await buildPrompt(userMsg, dir);
|
|
}
|
|
|
|
const { text, sessionId } = await runKimi({ prompt, cwd: dir, resumeId, onDelta, signal });
|
|
|
|
// Record the forward mapping.
|
|
const entry = { convId, sessionId: sessionId || resumeId, dir, ts: Date.now() };
|
|
if (chatId) {
|
|
sessionMap[`chat:${chatId}`] = entry;
|
|
} else {
|
|
const forward = turns.slice(0, lastUserIdx + 1).concat([{ role: 'assistant', content: text }]);
|
|
sessionMap[`hist:${historyKey(forward)}`] = entry;
|
|
}
|
|
persistMap();
|
|
|
|
return { text };
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// Kimi quota readout (kb #62). GET /usage — the claude-usage analog for
|
|
// Adolf. LLM-free: hits Kimi's own managed-usage endpoint directly, never
|
|
// spawns `kimi`. Mirrors the parsing logic of the installed
|
|
// @moonshot-ai/kimi-code CLI itself (decompiled from dist/main.mjs's
|
|
// parseManagedUsagePayload/toUsageRow/limitLabel/resetHintFrom — same
|
|
// endpoint, same response shape) so bucket labels/derivations stay in sync
|
|
// with what `kimi` would show via its own /usage-equivalent.
|
|
//
|
|
// Token source: the CLI's own OAuth creds file, kept fresh by the running
|
|
// `kimi` process (adolf-llm-home volume). We ONLY read the file's live
|
|
// access_token and never refresh here. Kimi's OAuth rotates the refresh_token
|
|
// on every refresh (single-use), so an independent refresh from this route
|
|
// invalidates the refresh_token the CLI's file still holds -> the CLI's next
|
|
// refresh fails `invalid_grant` and wipes the whole login (kb#87: this was the
|
|
// recurring Adolf logout, incl. the 2026-07-17 06:15 wipe / task #86). Making
|
|
// the CLI the sole refresher removes that race.
|
|
//
|
|
// Cost of that trade, measured 2026-07-22: the access token's `expires_in` is
|
|
// 900s, so it is only valid for 15 minutes after the CLI last refreshed it —
|
|
// i.e. only within 15 minutes of an actual Adolf turn. Adolf is idle most of
|
|
// the day, so a bare read failed far more often than it succeeded, which made
|
|
// quota gating effectively blind. Rather than refresh here (see above: that
|
|
// wipes the login), /usage now falls back to the LAST GOOD reading, clearly
|
|
// labelled `stale` with `as_of` + `age_s` so callers can decide whether it is
|
|
// fresh enough. The cache is written on every success and persisted to the
|
|
// workspace volume so it survives a container restart. Auth is untouched:
|
|
// this route still only ever READS the creds file.
|
|
const KIMI_CREDS_PATH = '/root/.kimi-code/credentials/kimi-code.json';
|
|
const KIMI_USAGES_URL = 'https://api.kimi.com/coding/v1/usages';
|
|
const KIMI_USAGE_CACHE_PATH = '/workspace/.adolf-llm/usage-cache.json';
|
|
|
|
// Last successful /usage payload, kept in memory and mirrored to disk.
|
|
let kimiUsageCache = null;
|
|
|
|
function readKimiUsageCache() {
|
|
if (kimiUsageCache) return kimiUsageCache;
|
|
try {
|
|
const parsed = JSON.parse(fs.readFileSync(KIMI_USAGE_CACHE_PATH, 'utf8'));
|
|
if (parsed && parsed.payload && parsed.cached_at) kimiUsageCache = parsed;
|
|
} catch { /* no cache yet, or unreadable — treated as "no cache" */ }
|
|
return kimiUsageCache;
|
|
}
|
|
|
|
function writeKimiUsageCache(payload) {
|
|
kimiUsageCache = { payload, cached_at: new Date().toISOString() };
|
|
try {
|
|
fs.mkdirSync(path.dirname(KIMI_USAGE_CACHE_PATH), { recursive: true });
|
|
fs.writeFileSync(KIMI_USAGE_CACHE_PATH, JSON.stringify(kimiUsageCache));
|
|
} catch { /* cache is best-effort; an unwritable volume must not break /usage */ }
|
|
}
|
|
|
|
async function loadKimiCreds() {
|
|
const raw = await fs.promises.readFile(KIMI_CREDS_PATH, 'utf8');
|
|
return JSON.parse(raw);
|
|
}
|
|
|
|
// Read the live access_token from the CLI's creds file. We deliberately do NOT
|
|
// refresh here (see the note above): the Kimi CLI is the sole refresher, so
|
|
// this route can never rotate the single-use refresh_token out from under it.
|
|
// A stale file token surfaces as an error -> /usage 502 -> "quota unavailable".
|
|
async function getKimiAccessToken() {
|
|
const creds = await loadKimiCreds();
|
|
const now = Math.floor(Date.now() / 1000);
|
|
if (creds.access_token && creds.expires_at && now < creds.expires_at - 30) {
|
|
return creds.access_token;
|
|
}
|
|
throw new Error('kimi access token stale (CLI refreshes on next use); quota temporarily unavailable');
|
|
}
|
|
|
|
async function fetchKimiUsagesRaw() {
|
|
const token = await getKimiAccessToken();
|
|
const res = await fetch(KIMI_USAGES_URL, { headers: { Authorization: `Bearer ${token}`, Accept: 'application/json' } });
|
|
if (!res.ok) {
|
|
const text = await res.text().catch(() => '');
|
|
throw new Error(`kimi /usages HTTP ${res.status}: ${text.slice(0, 500)}`);
|
|
}
|
|
return res.json();
|
|
}
|
|
|
|
function isRecord(v) { return v !== null && typeof v === 'object' && !Array.isArray(v); }
|
|
|
|
function toInt(v) {
|
|
if (typeof v === 'number') return Number.isFinite(v) ? Math.trunc(v) : null;
|
|
if (typeof v === 'string') { const n = Number(v); return Number.isFinite(n) ? Math.trunc(n) : null; }
|
|
return null;
|
|
}
|
|
|
|
// Port of the CLI's limitLabel(): prefer an explicit name/title/scope field,
|
|
// else derive "<N>h limit" / "<N>m limit" / "<N>d limit" from the window's
|
|
// duration+timeUnit.
|
|
function kimiLimitLabel(item, detail, window, idx) {
|
|
for (const key of ['name', 'title', 'scope']) {
|
|
const v = item[key] ?? detail[key];
|
|
if (typeof v === 'string' && v) return v;
|
|
}
|
|
const duration = toInt(window.duration ?? item.duration ?? detail.duration);
|
|
const rawUnit = window.timeUnit ?? item.timeUnit ?? detail.timeUnit;
|
|
const timeUnit = typeof rawUnit === 'string' ? rawUnit : '';
|
|
if (duration !== null) {
|
|
if (timeUnit.includes('MINUTE')) {
|
|
if (duration >= 60 && duration % 60 === 0) return `${duration / 60}h limit`;
|
|
return `${duration}m limit`;
|
|
}
|
|
if (timeUnit.includes('HOUR')) return `${duration}h limit`;
|
|
if (timeUnit.includes('DAY')) return `${duration}d limit`;
|
|
return `${duration}s limit`;
|
|
}
|
|
return `Limit #${idx + 1}`;
|
|
}
|
|
|
|
function kimiResetIso(raw) {
|
|
for (const key of ['reset_at', 'resetAt', 'reset_time', 'resetTime']) {
|
|
const v = raw[key];
|
|
if (typeof v === 'string' && v) return v;
|
|
}
|
|
return null;
|
|
}
|
|
|
|
// Port of the CLI's toUsageRow(): used = raw.used, or limit-remaining when
|
|
// used is absent.
|
|
function kimiUsageRow(raw, defaultLabel) {
|
|
if (!isRecord(raw)) return null;
|
|
const limit = toInt(raw.limit);
|
|
let used = toInt(raw.used);
|
|
const remaining = toInt(raw.remaining);
|
|
if (used === null && remaining !== null && limit !== null) used = limit - remaining;
|
|
if (used === null && limit === null) return null;
|
|
const name = typeof raw.name === 'string' ? raw.name : (typeof raw.title === 'string' ? raw.title : defaultLabel);
|
|
return {
|
|
label: name,
|
|
used: used ?? 0,
|
|
limit: limit ?? 0,
|
|
remaining: remaining !== null ? remaining : (limit !== null && used !== null ? limit - used : null),
|
|
resets: kimiResetIso(raw),
|
|
};
|
|
}
|
|
|
|
function kimiRowOut(row) {
|
|
if (!row) return null;
|
|
const pct = row.limit > 0 ? Math.round((row.used / row.limit) * 100) : null;
|
|
return { pct, used: row.used, limit: row.limit, remaining: row.remaining, resets: row.resets };
|
|
}
|
|
|
|
// Normalize Kimi's /usages payload ({ usage, limits: [...] }) into the
|
|
// claude-usage-analog shape: weekly / window_5h / window_7d, each
|
|
// pct/used/limit/remaining/resets, plus a raw `limits` passthrough so no
|
|
// bucket is lost if label text ever drifts from what we match on below.
|
|
function normalizeKimiUsage(payload) {
|
|
const rec = isRecord(payload) ? payload : {};
|
|
const summaryRow = kimiUsageRow(rec.usage, 'Weekly limit');
|
|
const limitRows = [];
|
|
const rawLimits = Array.isArray(rec.limits) ? rec.limits : [];
|
|
rawLimits.forEach((item, idx) => {
|
|
if (!isRecord(item)) return;
|
|
const detail = isRecord(item.detail) ? item.detail : item;
|
|
const window = isRecord(item.window) ? item.window : {};
|
|
const label = kimiLimitLabel(item, detail, window, idx);
|
|
const row = kimiUsageRow(detail, label);
|
|
if (row) limitRows.push(row);
|
|
});
|
|
|
|
const findByLabel = re => limitRows.find(r => re.test(r.label));
|
|
const weekly = summaryRow || findByLabel(/week/i) || null;
|
|
const window5h = findByLabel(/^5\s*h(our)?\b|5h limit/i) || null;
|
|
const window7d = findByLabel(/^7\s*d(ay)?\b|7d limit/i) || null;
|
|
|
|
return {
|
|
timestamp: new Date().toISOString(),
|
|
weekly: kimiRowOut(weekly),
|
|
window_5h: kimiRowOut(window5h),
|
|
window_7d: kimiRowOut(window7d),
|
|
limits: limitRows.map(r => ({ label: r.label, ...kimiRowOut(r) })),
|
|
};
|
|
}
|
|
|
|
// ---------------------------------------------------------------------------
|
|
// OpenAI-compatible HTTP surface.
|
|
function completionBody(text) {
|
|
return {
|
|
id: `chatcmpl-${Date.now()}`,
|
|
object: 'chat.completion',
|
|
created: Math.floor(Date.now() / 1000),
|
|
model: MODEL_ID,
|
|
choices: [{
|
|
index: 0,
|
|
message: { role: 'assistant', content: text },
|
|
finish_reason: 'stop',
|
|
}],
|
|
};
|
|
}
|
|
|
|
function sseChunk(id, created, delta, finishReason) {
|
|
return `data: ${JSON.stringify({
|
|
id, object: 'chat.completion.chunk', created, model: MODEL_ID,
|
|
choices: [{ index: 0, delta, finish_reason: finishReason }],
|
|
})}\n\n`;
|
|
}
|
|
|
|
const server = http.createServer((req, res) => {
|
|
if (req.method === 'GET' && req.url === '/v1/models') {
|
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({
|
|
object: 'list',
|
|
data: [{ id: MODEL_ID, object: 'model', owned_by: 'moonshot' }],
|
|
}));
|
|
return;
|
|
}
|
|
|
|
if (req.method === 'GET' && req.url === '/usage') {
|
|
(async () => {
|
|
try {
|
|
const raw = await fetchKimiUsagesRaw();
|
|
const out = normalizeKimiUsage(raw);
|
|
writeKimiUsageCache(out);
|
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({ ...out, stale: false }));
|
|
} catch (err) {
|
|
// Token stale (the common case when Adolf has been idle >15min) or Kimi
|
|
// unreachable. Serve the last good reading rather than nothing, labelled
|
|
// so a caller can reject it if it is too old to gate on.
|
|
const cached = readKimiUsageCache();
|
|
if (cached) {
|
|
const ageS = Math.max(0, Math.round((Date.now() - Date.parse(cached.cached_at)) / 1000));
|
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({
|
|
...cached.payload,
|
|
stale: true,
|
|
as_of: cached.cached_at,
|
|
age_s: ageS,
|
|
stale_reason: String(err.message || err),
|
|
}));
|
|
return;
|
|
}
|
|
res.writeHead(502, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({ error: String(err.message || err) }));
|
|
}
|
|
})();
|
|
return;
|
|
}
|
|
|
|
if (req.method === 'POST' && req.url === '/v1/chat/completions') {
|
|
let body = '';
|
|
req.on('data', d => { body += d; });
|
|
req.on('end', async () => {
|
|
let parsed;
|
|
try {
|
|
parsed = JSON.parse(body);
|
|
} catch {
|
|
res.writeHead(400, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({ error: 'invalid JSON body' }));
|
|
return;
|
|
}
|
|
|
|
const messages = parsed.messages || [];
|
|
|
|
if (parsed.stream) {
|
|
// Real streaming: open SSE, emit role chunk, then forward kimi deltas.
|
|
res.writeHead(200, {
|
|
'Content-Type': 'text/event-stream',
|
|
'Cache-Control': 'no-cache',
|
|
Connection: 'keep-alive',
|
|
});
|
|
const id = `chatcmpl-${Date.now()}`;
|
|
const created = Math.floor(Date.now() / 1000);
|
|
|
|
// Heartbeat keepalive: OpenClaw's LLM idle watchdog aborts a turn on
|
|
// any >120s gap between SSE stream events (default timeoutSeconds),
|
|
// not on total run length. During long thinking/tool/MCP phases Kimi
|
|
// emits stream-json events we don't forward, so the SSE stream can go
|
|
// silent well past that window. Every write resets `lastWrite`; a 5s
|
|
// ticker emits an empty-content delta once 25s of silence elapses —
|
|
// still a stream event (resets the watchdog) but appends nothing
|
|
// visible to the rendered reply or cognee-persisted text.
|
|
let lastWrite = Date.now();
|
|
const write = (delta, finish) => {
|
|
if (res.writableEnded || res.destroyed) return;
|
|
res.write(sseChunk(id, created, delta, finish));
|
|
lastWrite = Date.now();
|
|
};
|
|
write({ role: 'assistant' }, null);
|
|
const hb = setInterval(() => {
|
|
if (Date.now() - lastWrite >= 25_000) write({ content: '' }, null);
|
|
}, 5_000);
|
|
|
|
// Propagate a client/gateway disconnect down to the kimi child so an
|
|
// abandoned turn (e.g. OpenClaw's idle watchdog gave up) is killed
|
|
// instead of finishing invisibly and burning quota. `done` guards
|
|
// against the normal res.end() 'close' also aborting.
|
|
const ac = new AbortController();
|
|
let done = false;
|
|
res.on('close', () => { if (!done) ac.abort(); });
|
|
|
|
try {
|
|
await handleTurn(messages, delta => write({ content: delta }, null), ac.signal);
|
|
done = true;
|
|
write({}, 'stop');
|
|
res.write('data: [DONE]\n\n');
|
|
} catch (err) {
|
|
done = true;
|
|
// Headers already sent — surface the error inside the stream (unless
|
|
// the socket is already gone, in which case there is nowhere to write).
|
|
if (!res.writableEnded && !res.destroyed) {
|
|
write({ content: `\n[error: ${String(err.message || err)}]` }, 'stop');
|
|
res.write('data: [DONE]\n\n');
|
|
}
|
|
} finally {
|
|
clearInterval(hb);
|
|
if (!res.writableEnded) res.end();
|
|
}
|
|
return;
|
|
}
|
|
|
|
const ac = new AbortController();
|
|
let done = false;
|
|
res.on('close', () => { if (!done) ac.abort(); });
|
|
try {
|
|
const { text } = await handleTurn(messages, null, ac.signal);
|
|
done = true;
|
|
res.writeHead(200, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify(completionBody(text)));
|
|
} catch (err) {
|
|
done = true;
|
|
if (!res.writableEnded && !res.destroyed) {
|
|
res.writeHead(500, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({ error: String(err.message || err) }));
|
|
}
|
|
}
|
|
});
|
|
return;
|
|
}
|
|
|
|
res.writeHead(404, { 'Content-Type': 'application/json' });
|
|
res.end(JSON.stringify({ error: 'not found' }));
|
|
});
|
|
|
|
server.listen(PORT, () => console.log(`adolf-llm wrapper listening on :${PORT}`));
|