ai: migrate LLM backbone from Kimi CLI to Codex CLI

Retires the Moonshot/Kimi subscription in favour of the already-paid ChatGPT
plan. Both CLI wrappers now run `codex exec`; the kimi-agent container is gone.

adolf-llm + hindsight-llm:
- runKimi -> runCodex (`codex exec --json --skip-git-repo-check`), resume via
  `codex exec resume <thread_id>`.
- MCP moves from a per-session .mcp.json (a workaround for Kimi having no
  --mcp-config-file flag) to a $CODEX_HOME/config.toml generated once at
  startup from shared-mcp.json. Field translation is load-bearing:
  bearerTokenEnvVar -> bearer_token_env_var, enabledTools -> enabled_tools.
- approval_policy="never" + sandbox_mode required, or unattended turns block
  on an approval prompt nobody can answer.

kimi-agent removed. It was the ONLY large-tier deployment behind LiteLLM, so
deleting it outright would have silently degraded every large-tier request to
the local 4B model via the existing fallbacks. tier-large, the auto_router
complex-reasoning route and their fallbacks now point at the codex-backed
adolf-llm wrapper (model_name: codex-agent).

Three environment blockers fixed along the way:
- OpenAI geo-blocks this host (403 unsupported_country_region_territory).
  Both containers now egress via the host xray proxy, with NO_PROXY keeping
  MCP and *.alogins.net traffic off the tunnel.
- node:22-slim ships no system CA store; the Rust codex binary validates TLS
  against it, so every HTTPS call failed with a generic transport error while
  Node's own fetch worked. ca-certificates added to both images.
- `codex exec resume` rejects -C/--cd (plain `codex exec` accepts it), which
  broke follow-up turns while first turns succeeded.

Known regression: Kimi's managed-usage API has no Codex equivalent, so the
/usage route returns 501 and there is no quota probe for the codex model.
The two quota plugins degrade quietly to no output.

Also: stop tracking cognee.env (live LLM + JWT secrets) and gitignore it.
The secrets remain in earlier history and should be rotated.

Verified live: plain turn, SSE streaming, session resume, MCP tool call,
bearer-token MCP call, and completions through both LiteLLM routes.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014Y5QPagv4iun1ghpwM96Ff
This commit is contained in:
2026-08-01 06:13:27 +00:00
parent a27bae828a
commit 9094d71e2f
66 changed files with 653 additions and 851 deletions

15
ai/silero-tts/Dockerfile Normal file
View File

@@ -0,0 +1,15 @@
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends ffmpeg && rm -rf /var/lib/apt/lists/*
# CPU-only torch keeps image ~500MB vs ~2GB for CUDA
RUN pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cpu
RUN pip install --no-cache-dir fastapi uvicorn scipy numpy pydub omegaconf
WORKDIR /app
COPY server.py .
ENV TORCH_HOME=/cache/torch
EXPOSE 8881
CMD ["uvicorn", "server:app", "--host", "0.0.0.0", "--port", "8881"]

153
ai/silero-tts/server.py Normal file
View File

@@ -0,0 +1,153 @@
import io
import re
import logging
import numpy as np
import torch
from fastapi import FastAPI, HTTPException
from fastapi.responses import Response
from pydantic import BaseModel
import scipy.io.wavfile as wavfile
from pydub import AudioSegment
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
app = FastAPI(title="Silero TTS")
# ── Config ────────────────────────────────────────────────────────────────────
DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
SAMPLE_RATE = 24000
MAX_CHUNK = 800 # chars per Silero call
# Model identifiers (passed as `speaker` to torch.hub.load — selects model file)
MODEL_ID = {"ru": "v3_1_ru", "en": "v3_en"}
# Silero speakers (passed to apply_tts)
RU_SPEAKERS = ["aidar", "baya", "kseniya", "xenia", "eugene"]
EN_SPEAKERS = [f"en_{i}" for i in range(10)]
# OpenAI voice → Silero speaker
VOICE_MAP = {
"ru": {"alloy": "eugene", "echo": "aidar", "fable": "baya",
"onyx": "eugene", "nova": "kseniya", "shimmer": "xenia"},
"en": {"alloy": "en_3", "echo": "en_1", "fable": "en_2",
"onyx": "en_3", "nova": "en_4", "shimmer": "en_5"},
}
# ── Model cache ───────────────────────────────────────────────────────────────
_models: dict[str, object] = {}
def _get_model(language: str):
if language not in _models:
logger.info(f"Loading Silero model {MODEL_ID[language]} lang={language} device={DEVICE}")
model, _ = torch.hub.load(
repo_or_dir="snakers4/silero-models",
model="silero_tts",
language=language,
speaker=MODEL_ID[language],
trust_repo=True,
)
model.to(DEVICE)
_models[language] = model
logger.info(f"Model ready: lang={language}")
return _models[language]
@app.on_event("startup")
async def preload():
"""Preload both language models to avoid cold-start on first request."""
for lang in ("ru", "en"):
try:
_get_model(lang)
except Exception as e:
logger.warning(f"Preload failed for lang={lang}: {e}")
# ── Helpers ───────────────────────────────────────────────────────────────────
def _is_russian(text: str) -> bool:
return bool(re.search(r"[а-яёА-ЯЁ]", text))
def _split_sentences(text: str) -> list[str]:
"""Split on sentence boundaries, keeping chunks under MAX_CHUNK chars."""
if len(text) <= MAX_CHUNK:
return [text]
parts = re.split(r"(?<=[.!?;])\s+", text.strip())
chunks, cur = [], ""
for part in parts:
if len(cur) + len(part) + 1 <= MAX_CHUNK:
cur = f"{cur} {part}" if cur else part
else:
if cur:
chunks.append(cur)
# If single part is too long, split mid-word as last resort
cur = part[:MAX_CHUNK] if len(part) > MAX_CHUNK else part
if cur:
chunks.append(cur)
return chunks or [text[:MAX_CHUNK]]
def _to_bytes(audio: torch.Tensor, fmt: str) -> bytes:
pcm = (audio.cpu().numpy() * 32767).astype(np.int16)
if fmt == "pcm":
return pcm.tobytes()
buf = io.BytesIO()
wavfile.write(buf, SAMPLE_RATE, pcm)
if fmt == "wav":
return buf.getvalue()
seg = AudioSegment.from_wav(io.BytesIO(buf.getvalue()))
out = io.BytesIO()
seg.export(out, format="mp3")
return out.getvalue()
# ── API ───────────────────────────────────────────────────────────────────────
class SpeechRequest(BaseModel):
model: str = "silero"
input: str
voice: str = "alloy"
response_format: str = "mp3"
speed: float = 1.0
@app.get("/health")
async def health():
return {"status": "ok", "device": DEVICE}
@app.get("/v1/models")
async def list_models():
return {
"object": "list",
"data": [{"id": "silero", "object": "model", "owned_by": "silero"}],
}
@app.post("/v1/audio/speech")
async def speech(req: SpeechRequest):
text = req.input.strip()
if not text:
raise HTTPException(status_code=400, detail="input is empty")
language = "ru" if _is_russian(text) else "en"
vm = VOICE_MAP[language]
speaker = vm.get(req.voice, vm["alloy"])
try:
model = _get_model(language)
chunks = _split_sentences(text)
parts = [
model.apply_tts(text=chunk, speaker=speaker, sample_rate=SAMPLE_RATE)
for chunk in chunks
]
audio = parts[0] if len(parts) == 1 else torch.cat(parts)
except Exception as e:
logger.error(f"TTS error: {e}")
raise HTTPException(status_code=500, detail=str(e))
fmt = req.response_format.lower()
audio_bytes = _to_bytes(audio, fmt)
media_types = {"wav": "audio/wav", "pcm": "audio/pcm", "mp3": "audio/mpeg"}
media_type = media_types.get(fmt, "audio/mpeg")
return Response(content=audio_bytes, media_type=media_type)