ai: migrate LLM backbone from Kimi CLI to Codex CLI

Retires the Moonshot/Kimi subscription in favour of the already-paid ChatGPT
plan. Both CLI wrappers now run `codex exec`; the kimi-agent container is gone.

adolf-llm + hindsight-llm:
- runKimi -> runCodex (`codex exec --json --skip-git-repo-check`), resume via
  `codex exec resume <thread_id>`.
- MCP moves from a per-session .mcp.json (a workaround for Kimi having no
  --mcp-config-file flag) to a $CODEX_HOME/config.toml generated once at
  startup from shared-mcp.json. Field translation is load-bearing:
  bearerTokenEnvVar -> bearer_token_env_var, enabledTools -> enabled_tools.
- approval_policy="never" + sandbox_mode required, or unattended turns block
  on an approval prompt nobody can answer.

kimi-agent removed. It was the ONLY large-tier deployment behind LiteLLM, so
deleting it outright would have silently degraded every large-tier request to
the local 4B model via the existing fallbacks. tier-large, the auto_router
complex-reasoning route and their fallbacks now point at the codex-backed
adolf-llm wrapper (model_name: codex-agent).

Three environment blockers fixed along the way:
- OpenAI geo-blocks this host (403 unsupported_country_region_territory).
  Both containers now egress via the host xray proxy, with NO_PROXY keeping
  MCP and *.alogins.net traffic off the tunnel.
- node:22-slim ships no system CA store; the Rust codex binary validates TLS
  against it, so every HTTPS call failed with a generic transport error while
  Node's own fetch worked. ca-certificates added to both images.
- `codex exec resume` rejects -C/--cd (plain `codex exec` accepts it), which
  broke follow-up turns while first turns succeeded.

Known regression: Kimi's managed-usage API has no Codex equivalent, so the
/usage route returns 501 and there is no quota probe for the codex model.
The two quota plugins degrade quietly to no output.

Also: stop tracking cognee.env (live LLM + JWT secrets) and gitignore it.
The secrets remain in earlier history and should be rotated.

Verified live: plain turn, SSE streaming, session resume, MCP tool call,
bearer-token MCP call, and completions through both LiteLLM routes.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_014Y5QPagv4iun1ghpwM96Ff
This commit is contained in:
2026-08-01 06:13:27 +00:00
parent a27bae828a
commit 9094d71e2f
66 changed files with 653 additions and 851 deletions

18
ai/pipecat/Dockerfile Normal file
View File

@@ -0,0 +1,18 @@
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends gcc g++ && rm -rf /var/lib/apt/lists/*
# CPU torch first — prevents silero-vad from pulling in the CUDA variant
RUN pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cpu
RUN pip install --no-cache-dir \
"pipecat-ai[openai,livekit,silero]" \
"livekit-api" \
fastapi \
"uvicorn[standard]"
WORKDIR /app
COPY . .
EXPOSE 8882
CMD ["uvicorn", "bot:app", "--host", "0.0.0.0", "--port", "8882"]

228
ai/pipecat/bot.py Normal file
View File

@@ -0,0 +1,228 @@
import asyncio
import os
import re
import uuid
import logging
from fastapi import FastAPI
from fastapi.responses import HTMLResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel
from livekit import api as lkapi
from pipecat.audio.vad.silero import SileroVADAnalyzer
from pipecat.audio.vad.vad_analyzer import VADParams
from pipecat.frames.frames import TextFrame
from pipecat.pipeline.pipeline import Pipeline
from pipecat.pipeline.runner import PipelineRunner
from pipecat.pipeline.task import PipelineParams, PipelineTask
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
from pipecat.processors.frame_processor import FrameProcessor, FrameDirection
from pipecat.services.openai.llm import OpenAILLMService
from pipecat.services.openai.stt import OpenAISTTService
from pipecat.services.openai.tts import OpenAITTSService
from pipecat.transports.livekit.transport import LiveKitTransport, LiveKitParams
# ── TTS text normalizer ──────────────────────────────────────────────────────
# Replaces symbols and abbreviations with spoken Russian words so Silero TTS
# doesn't truncate on unknown characters.
_NORM_RULES: list[tuple[re.Pattern, str]] = [
# Temperature: +12°C / -5°С / 12 °C → плюс двенадцать градусов цельсия
(re.compile(r"([+-]?\d+)\s*°\s*[CСcс]", re.IGNORECASE), r"\1 градусов цельсия"),
# Bare degree sign: 90° → 90 градусов
(re.compile(r"(\d+)\s*°"), r"\1 градусов"),
# Percent
(re.compile(r"(\d+)\s*%"), r"\1 процентов"),
# Speed: m/s, м/с, km/h, км/ч
(re.compile(r"\bm/s\b", re.IGNORECASE), "метров в секунду"),
(re.compile(r"\bм/с\b"), "метров в секунду"),
(re.compile(r"\bkm/h\b", re.IGNORECASE), "километров в час"),
(re.compile(r"\bкм/ч\b"), "километров в час"),
# Currency
(re.compile(r"\$\s*(\d+)"), r"\1 долларов"),
(re.compile(r"(\d+)\s*\$"), r"\1 долларов"),
(re.compile(r"\s*(\d+)"), r"\1 евро"),
(re.compile(r"(\d+)\s*€"), r"\1 евро"),
(re.compile(r"(\d+)\s*₽"), r"\1 рублей"),
# Plus/minus signs before numbers
(re.compile(r"\+(\d)"), r"плюс \1"),
(re.compile(r"-(\d)"), r"минус \1"),
# Common abbreviations
(re.compile(r"\г\b"), "килограмм"),
(re.compile(r"\bг\b(?=\s|$)"), "грамм"),
(re.compile(r"\bмм\b"), "миллиметров"),
(re.compile(r"\bсм\b"), "сантиметров"),
(re.compile(r"\bкм\b"), "километров"),
# Strip remaining special chars that TTS can't handle
(re.compile(r"[°•·†‡§¶©®™«»<>{}[\]|\\~^`]"), ""),
]
def normalize_for_tts(text: str) -> str:
"""Replace symbols with spoken Russian equivalents."""
for pattern, replacement in _NORM_RULES:
text = pattern.sub(replacement, text)
return text
class TTSTextNormalizer(FrameProcessor):
"""Intercepts TextFrames between LLM and TTS, normalizing symbols to words."""
async def process_frame(self, frame, direction: FrameDirection = FrameDirection.DOWNSTREAM):
await super().process_frame(frame, direction)
if isinstance(frame, TextFrame):
original = frame.text
normalized = normalize_for_tts(original)
if normalized != original:
logger.debug(f"TTSTextNormalizer: {original!r}{normalized!r}")
await self.push_frame(TextFrame(text=normalized), direction)
else:
await self.push_frame(frame, direction)
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# ── Config ────────────────────────────────────────────────────────────────────
LIVEKIT_URL = os.getenv("LIVEKIT_URL", "ws://host.docker.internal:7880") # bot connects here
LIVEKIT_PUBLIC_URL = os.getenv("LIVEKIT_PUBLIC_URL", "wss://lk.alogins.net") # browser connects here
LIVEKIT_API_KEY = os.getenv("LIVEKIT_API_KEY", "devkey")
LIVEKIT_SECRET = os.getenv("LIVEKIT_SECRET", "")
ADOLF_URL = os.getenv("ADOLF_URL", "http://host.docker.internal:8000/v1")
STT_URL = os.getenv("STT_URL", "http://host.docker.internal:8880/v1")
TTS_URL = os.getenv("TTS_URL", "http://host.docker.internal:8881/v1")
STT_MODEL = os.getenv("STT_MODEL", "deepdml/faster-whisper-large-v3-turbo-ct2")
TTS_VOICE = os.getenv("TTS_VOICE", "onyx")
SYSTEM_PROMPT = "You are Adolf, a helpful voice assistant. Keep replies concise — 1-3 sentences. No markdown."
app = FastAPI(title="Pipecat Voice Bot")
app.mount("/static", StaticFiles(directory="static"), name="static")
# ── LiveKit helpers ───────────────────────────────────────────────────────────
def _lk_token(room: str, identity: str, is_bot: bool = False) -> str:
grants = lkapi.VideoGrants(
room_join=True,
room=room,
can_publish=True,
can_subscribe=True,
can_publish_data=True,
)
token = (
lkapi.AccessToken(LIVEKIT_API_KEY, LIVEKIT_SECRET)
.with_identity(identity)
.with_name("Adolf Bot" if is_bot else identity)
.with_grants(grants)
)
return token.to_jwt()
async def _create_room(room_name: str) -> None:
lk = lkapi.LiveKitAPI(LIVEKIT_URL, LIVEKIT_API_KEY, LIVEKIT_SECRET)
try:
await lk.room.create_room(
lkapi.CreateRoomRequest(name=room_name, empty_timeout=300, max_participants=5)
)
finally:
await lk.aclose()
# ── Pipecat pipeline ──────────────────────────────────────────────────────────
async def _run_bot(room_name: str) -> None:
bot_token = _lk_token(room_name, "pipecat-bot", is_bot=True)
transport = LiveKitTransport(
url=LIVEKIT_URL,
token=bot_token,
room_name=room_name,
params=LiveKitParams(
audio_in_enabled=True,
audio_out_enabled=True,
vad_enabled=True,
vad_analyzer=SileroVADAnalyzer(params=VADParams(
stop_secs=0.8, # wait 0.8s of silence before end-of-speech
start_secs=0.2, # start speech detection after 0.2s
confidence=0.85, # high confidence to avoid triggering on ambient noise
)),
),
)
stt = OpenAISTTService(
api_key="dummy",
base_url=STT_URL,
model=STT_MODEL,
language="ru",
)
llm = OpenAILLMService(
api_key="dummy",
base_url=ADOLF_URL,
model="adolf-light",
)
tts = OpenAITTSService(
api_key="dummy",
base_url=TTS_URL,
model="silero",
voice=TTS_VOICE,
)
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
context = OpenAILLMContext(messages)
context_aggregator = llm.create_context_aggregator(context)
normalizer = TTSTextNormalizer()
pipeline = Pipeline([
transport.input(),
stt,
context_aggregator.user(),
llm,
normalizer,
tts,
transport.output(),
context_aggregator.assistant(),
])
task = PipelineTask(pipeline, params=PipelineParams(allow_interruptions=False))
@transport.event_handler("on_participant_disconnected")
async def on_disconnect(transport, participant):
identity = participant if isinstance(participant, str) else getattr(participant, "identity", str(participant))
logger.info(f"Participant {identity} left — stopping bot")
await task.cancel()
runner = PipelineRunner()
logger.info(f"Bot starting in room={room_name}")
await runner.run(task)
logger.info(f"Bot done in room={room_name}")
# ── API ───────────────────────────────────────────────────────────────────────
class ConnectResponse(BaseModel):
room: str
token: str
url: str
@app.post("/connect", response_model=ConnectResponse)
async def connect():
room_name = f"voice-{uuid.uuid4().hex[:6]}"
await _create_room(room_name)
user_token = _lk_token(room_name, "user")
asyncio.create_task(_run_bot(room_name))
return ConnectResponse(room=room_name, token=user_token, url=LIVEKIT_PUBLIC_URL)
@app.get("/health")
async def health():
return {"status": "ok"}
@app.get("/", response_class=HTMLResponse)
async def index():
with open("static/index.html") as f:
return f.read()

View File

@@ -0,0 +1,241 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Adolf Voice</title>
<style>
* { box-sizing: border-box; margin: 0; padding: 0; }
body {
font-family: system-ui, sans-serif;
background: #0f0f0f;
color: #e0e0e0;
display: flex;
align-items: center;
justify-content: center;
min-height: 100vh;
}
.card {
background: #1a1a1a;
border: 1px solid #2a2a2a;
border-radius: 16px;
padding: 40px;
text-align: center;
width: 360px;
}
h1 { font-size: 1.4rem; font-weight: 600; margin-bottom: 8px; }
.subtitle { color: #666; font-size: 0.85rem; margin-bottom: 32px; }
#orb {
width: 100px;
height: 100px;
border-radius: 50%;
background: radial-gradient(circle, #3a3a3a 0%, #1a1a1a 100%);
border: 2px solid #333;
margin: 0 auto 24px;
cursor: pointer;
transition: all 0.3s ease;
display: flex;
align-items: center;
justify-content: center;
font-size: 2rem;
user-select: none;
}
#orb.listening {
background: radial-gradient(circle, #1e3a5f 0%, #0d1f33 100%);
border-color: #3b82f6;
box-shadow: 0 0 20px #3b82f640;
animation: pulse-blue 1.5s ease-in-out infinite;
}
#orb.speaking {
background: radial-gradient(circle, #1e4034 0%, #0d2018 100%);
border-color: #22c55e;
box-shadow: 0 0 20px #22c55e40;
animation: pulse-green 0.8s ease-in-out infinite;
}
#orb.thinking {
background: radial-gradient(circle, #3a2e1e 0%, #1a160d 100%);
border-color: #f59e0b;
box-shadow: 0 0 20px #f59e0b40;
animation: pulse-amber 1s ease-in-out infinite;
}
#orb.user-speaking {
background: radial-gradient(circle, #3a1e3a 0%, #1a0d1a 100%);
border-color: #a855f7;
box-shadow: 0 0 20px #a855f740;
animation: pulse-purple 0.6s ease-in-out infinite;
}
@keyframes pulse-blue { 0%,100%{box-shadow:0 0 20px #3b82f640} 50%{box-shadow:0 0 35px #3b82f680} }
@keyframes pulse-green { 0%,100%{box-shadow:0 0 20px #22c55e40} 50%{box-shadow:0 0 35px #22c55e80} }
@keyframes pulse-amber { 0%,100%{box-shadow:0 0 20px #f59e0b40} 50%{box-shadow:0 0 35px #f59e0b80} }
@keyframes pulse-purple { 0%,100%{box-shadow:0 0 20px #a855f740} 50%{box-shadow:0 0 35px #a855f780} }
#status {
font-size: 0.9rem;
color: #888;
margin-bottom: 16px;
min-height: 1.2em;
}
#transcript {
font-size: 0.8rem;
color: #555;
margin-bottom: 20px;
min-height: 2.4em;
line-height: 1.4;
font-style: italic;
word-break: break-word;
}
#transcript .user-text { color: #7ba8d4; font-style: normal; }
#transcript .bot-text { color: #6ab88a; font-style: normal; }
#btn {
background: #2a2a2a;
border: 1px solid #3a3a3a;
color: #e0e0e0;
padding: 10px 28px;
border-radius: 8px;
font-size: 0.9rem;
cursor: pointer;
transition: background 0.2s;
}
#btn:hover { background: #333; }
#btn:disabled { opacity: 0.4; cursor: default; }
#btn.active { border-color: #ef4444; color: #ef4444; }
</style>
</head>
<body>
<div class="card">
<h1>Adolf</h1>
<p class="subtitle">Voice assistant</p>
<div id="orb" onclick="toggle()">🎙️</div>
<div id="status">Press to connect</div>
<div id="transcript"></div>
<button id="btn" onclick="toggle()">Connect</button>
</div>
<script src="https://cdn.jsdelivr.net/npm/livekit-client/dist/livekit-client.umd.min.js"></script>
<script>
let room = null;
let audioCtx = null;
// Unlock browser autoplay — must happen on first user gesture
function unlockAudio() {
if (!audioCtx) {
audioCtx = new (window.AudioContext || window.webkitAudioContext)();
if (audioCtx.state === 'suspended') audioCtx.resume();
}
}
function setUI(state, msg) {
const orb = document.getElementById('orb');
const status = document.getElementById('status');
const btn = document.getElementById('btn');
orb.className = state || '';
status.textContent = msg;
if (state === null) {
btn.textContent = 'Connect';
btn.classList.remove('active');
orb.textContent = '🎙️';
} else {
btn.textContent = 'Disconnect';
btn.classList.add('active');
orb.textContent = state === 'thinking' ? '💭' :
state === 'speaking' ? '🔊' :
state === 'user-speaking' ? '🗣️' : '🎙️';
}
}
function addTranscript(role, text) {
const div = document.getElementById('transcript');
const cls = role === 'user' ? 'user-text' : 'bot-text';
const prefix = role === 'user' ? 'You: ' : 'Adolf: ';
div.innerHTML = `<span class="${cls}">${prefix}${text}</span>`;
}
async function toggle() {
unlockAudio();
if (room) {
room.disconnect();
return;
}
document.getElementById('btn').disabled = true;
setUI('thinking', 'Connecting…');
try {
const res = await fetch('/connect', { method: 'POST' });
const { token, url } = await res.json();
room = new LivekitClient.Room({ adaptiveStream: true, dynacast: true });
room.on(LivekitClient.RoomEvent.Connected, () => {
setUI('listening', 'Listening…');
document.getElementById('btn').disabled = false;
});
room.on(LivekitClient.RoomEvent.Disconnected, () => {
setUI(null, 'Press to connect');
document.getElementById('btn').disabled = false;
document.getElementById('transcript').innerHTML = '';
room = null;
});
room.on(LivekitClient.RoomEvent.ActiveSpeakersChanged, (speakers) => {
if (!room) return;
const botSpeaking = speakers.some(s => s.identity === 'pipecat-bot');
const userSpeaking = speakers.some(s => s.identity === 'user');
if (botSpeaking) {
setUI('speaking', 'Adolf is speaking…');
} else if (userSpeaking) {
setUI('user-speaking', 'Listening to you…');
} else {
setUI('listening', 'Listening…');
}
});
// Attach remote audio so browser plays it
room.on(LivekitClient.RoomEvent.TrackSubscribed, (track, pub, participant) => {
if (track.kind === 'audio') {
// Remove old element if any
const old = document.getElementById(`audio-${participant.identity}`);
if (old) old.remove();
const el = track.attach();
el.id = `audio-${participant.identity}`;
el.autoplay = true;
// Resume audio context on attach to beat autoplay restrictions
if (audioCtx && audioCtx.state === 'suspended') audioCtx.resume();
document.body.appendChild(el);
el.play().catch(() => {});
}
});
room.on(LivekitClient.RoomEvent.TrackUnsubscribed, (track) => {
track.detach().forEach(el => el.remove());
});
room.on(LivekitClient.RoomEvent.ParticipantConnected, (p) => {
if (p.identity === 'pipecat-bot') {
setUI('listening', 'Listening…');
}
});
// Data messages from bot (transcripts/events if pipecat sends them)
room.on(LivekitClient.RoomEvent.DataReceived, (data, participant) => {
try {
const msg = JSON.parse(new TextDecoder().decode(data));
if (msg.type === 'transcript' && msg.role === 'user') addTranscript('user', msg.text);
if (msg.type === 'transcript' && msg.role === 'bot') addTranscript('bot', msg.text);
} catch {}
});
const wsUrl = url.replace(/^http/, 'ws');
await room.connect(wsUrl, token);
await room.localParticipant.setMicrophoneEnabled(true);
} catch (err) {
console.error(err);
setUI(null, 'Error: ' + err.message);
document.getElementById('btn').disabled = false;
room = null;
}
}
</script>
</body>
</html>

122
ai/pipecat/test_pipeline.py Normal file
View File

@@ -0,0 +1,122 @@
"""
End-to-end pipeline test:
1. Call /connect to get a room + token
2. Join the LiveKit room as a Python client
3. Publish TTS audio (pre-generated) as microphone input
4. Capture bot's audio response and save to file
"""
import asyncio
import wave
import struct
import httpx
import numpy as np
from livekit import rtc
PIPECAT_URL = "http://localhost:8882"
TTS_URL = "http://host.docker.internal:8881"
OUTPUT_FILE = "/tmp/bot_response.wav"
SAMPLE_RATE = 48000
NUM_CHANNELS = 1
async def generate_tts_pcm(text: str) -> bytes:
"""Get WAV audio from Silero TTS, return raw PCM int16 bytes."""
async with httpx.AsyncClient(timeout=30) as c:
r = await c.post(f"{TTS_URL}/v1/audio/speech", json={
"input": text, "voice": "onyx", "response_format": "wav"
})
r.raise_for_status()
# Skip WAV header (44 bytes) to get raw PCM
return r.content[44:]
async def main():
# Step 1 — create room
print("[test] Creating room...")
async with httpx.AsyncClient() as c:
r = await c.post(f"{PIPECAT_URL}/connect")
r.raise_for_status()
creds = r.json()
print(f"[test] Room: {creds['room']} URL: {creds['url']}")
# Step 2 — generate test audio
test_phrase = "Привет! Как тебя зовут?"
print(f"[test] Generating TTS for: {test_phrase!r}")
pcm_bytes = await generate_tts_pcm(test_phrase)
print(f"[test] TTS PCM: {len(pcm_bytes)} bytes (~{len(pcm_bytes)//(SAMPLE_RATE*2):.1f}s)")
# Step 3 — join room
room = rtc.Room()
received_frames: list[bytes] = []
@room.on("track_subscribed")
def on_track(track, pub, participant):
if track.kind == rtc.TrackKind.KIND_AUDIO and participant.identity == "pipecat-bot":
print(f"[test] Subscribed to bot audio track")
audio_stream = rtc.AudioStream(track, sample_rate=SAMPLE_RATE, num_channels=NUM_CHANNELS)
asyncio.ensure_future(_collect_audio(audio_stream, received_frames))
ws_url = creds["url"].replace("https://", "wss://").replace("http://", "ws://")
# Connect internally via host.docker.internal
internal_url = "ws://host.docker.internal:7880"
print(f"[test] Connecting to LiveKit at {internal_url}...")
await room.connect(internal_url, creds["token"])
print(f"[test] Connected. Waiting for bot to join...")
# Wait for bot participant
for _ in range(20):
if any(p.identity == "pipecat-bot" for p in room.remote_participants.values()):
break
await asyncio.sleep(0.5)
print(f"[test] Participants: {[p.identity for p in room.remote_participants.values()]}")
# Step 4 — publish audio as microphone
print("[test] Publishing audio track...")
source = rtc.AudioSource(SAMPLE_RATE, NUM_CHANNELS)
local_track = rtc.LocalAudioTrack.create_audio_track("microphone", source)
opts = rtc.TrackPublishOptions(source=rtc.TrackSource.SOURCE_MICROPHONE)
await room.local_participant.publish_track(local_track, opts)
# Send PCM in 20ms chunks
chunk_samples = SAMPLE_RATE * 20 // 1000 # 960 samples per chunk
chunk_bytes = chunk_samples * 2 # int16
print(f"[test] Sending {len(pcm_bytes) // chunk_bytes} audio chunks...")
for i in range(0, len(pcm_bytes), chunk_bytes):
chunk = pcm_bytes[i:i + chunk_bytes]
if len(chunk) < chunk_bytes:
chunk = chunk + b'\x00' * (chunk_bytes - len(chunk))
samples = np.frombuffer(chunk, dtype=np.int16)
frame = rtc.AudioFrame(
data=samples.tobytes(),
sample_rate=SAMPLE_RATE,
num_channels=NUM_CHANNELS,
samples_per_channel=chunk_samples,
)
await source.capture_frame(frame)
await asyncio.sleep(0.02)
print("[test] Audio sent. Waiting for bot response (up to 30s)...")
await asyncio.sleep(30)
await room.disconnect()
# Step 5 — save response
if received_frames:
total = b"".join(received_frames)
print(f"[test] Received {len(total)} bytes of bot audio ({len(total)//(SAMPLE_RATE*2):.1f}s)")
with wave.open(OUTPUT_FILE, "wb") as wf:
wf.setnchannels(NUM_CHANNELS)
wf.setsampwidth(2)
wf.setframerate(SAMPLE_RATE)
wf.writeframes(total)
print(f"[test] Saved to {OUTPUT_FILE}")
else:
print("[test] No audio received from bot!")
async def _collect_audio(stream: rtc.AudioStream, buf: list):
async for event in stream:
buf.append(bytes(event.frame.data))
asyncio.run(main())