ai: migrate LLM backbone from Kimi CLI to Codex CLI
Retires the Moonshot/Kimi subscription in favour of the already-paid ChatGPT plan. Both CLI wrappers now run `codex exec`; the kimi-agent container is gone. adolf-llm + hindsight-llm: - runKimi -> runCodex (`codex exec --json --skip-git-repo-check`), resume via `codex exec resume <thread_id>`. - MCP moves from a per-session .mcp.json (a workaround for Kimi having no --mcp-config-file flag) to a $CODEX_HOME/config.toml generated once at startup from shared-mcp.json. Field translation is load-bearing: bearerTokenEnvVar -> bearer_token_env_var, enabledTools -> enabled_tools. - approval_policy="never" + sandbox_mode required, or unattended turns block on an approval prompt nobody can answer. kimi-agent removed. It was the ONLY large-tier deployment behind LiteLLM, so deleting it outright would have silently degraded every large-tier request to the local 4B model via the existing fallbacks. tier-large, the auto_router complex-reasoning route and their fallbacks now point at the codex-backed adolf-llm wrapper (model_name: codex-agent). Three environment blockers fixed along the way: - OpenAI geo-blocks this host (403 unsupported_country_region_territory). Both containers now egress via the host xray proxy, with NO_PROXY keeping MCP and *.alogins.net traffic off the tunnel. - node:22-slim ships no system CA store; the Rust codex binary validates TLS against it, so every HTTPS call failed with a generic transport error while Node's own fetch worked. ca-certificates added to both images. - `codex exec resume` rejects -C/--cd (plain `codex exec` accepts it), which broke follow-up turns while first turns succeeded. Known regression: Kimi's managed-usage API has no Codex equivalent, so the /usage route returns 501 and there is no quota probe for the codex model. The two quota plugins degrade quietly to no output. Also: stop tracking cognee.env (live LLM + JWT secrets) and gitignore it. The secrets remain in earlier history and should be rotated. Verified live: plain turn, SSE streaming, session resume, MCP tool call, bearer-token MCP call, and completions through both LiteLLM routes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014Y5QPagv4iun1ghpwM96Ff
This commit is contained in:
18
ai/pipecat/Dockerfile
Normal file
18
ai/pipecat/Dockerfile
Normal file
@@ -0,0 +1,18 @@
|
||||
FROM python:3.11-slim
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends gcc g++ && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# CPU torch first — prevents silero-vad from pulling in the CUDA variant
|
||||
RUN pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cpu
|
||||
|
||||
RUN pip install --no-cache-dir \
|
||||
"pipecat-ai[openai,livekit,silero]" \
|
||||
"livekit-api" \
|
||||
fastapi \
|
||||
"uvicorn[standard]"
|
||||
|
||||
WORKDIR /app
|
||||
COPY . .
|
||||
|
||||
EXPOSE 8882
|
||||
CMD ["uvicorn", "bot:app", "--host", "0.0.0.0", "--port", "8882"]
|
||||
228
ai/pipecat/bot.py
Normal file
228
ai/pipecat/bot.py
Normal file
@@ -0,0 +1,228 @@
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import uuid
|
||||
import logging
|
||||
|
||||
from fastapi import FastAPI
|
||||
from fastapi.responses import HTMLResponse
|
||||
from fastapi.staticfiles import StaticFiles
|
||||
from pydantic import BaseModel
|
||||
|
||||
from livekit import api as lkapi
|
||||
|
||||
from pipecat.audio.vad.silero import SileroVADAnalyzer
|
||||
from pipecat.audio.vad.vad_analyzer import VADParams
|
||||
from pipecat.frames.frames import TextFrame
|
||||
from pipecat.pipeline.pipeline import Pipeline
|
||||
from pipecat.pipeline.runner import PipelineRunner
|
||||
from pipecat.pipeline.task import PipelineParams, PipelineTask
|
||||
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
|
||||
from pipecat.processors.frame_processor import FrameProcessor, FrameDirection
|
||||
from pipecat.services.openai.llm import OpenAILLMService
|
||||
from pipecat.services.openai.stt import OpenAISTTService
|
||||
from pipecat.services.openai.tts import OpenAITTSService
|
||||
from pipecat.transports.livekit.transport import LiveKitTransport, LiveKitParams
|
||||
|
||||
|
||||
# ── TTS text normalizer ──────────────────────────────────────────────────────
|
||||
# Replaces symbols and abbreviations with spoken Russian words so Silero TTS
|
||||
# doesn't truncate on unknown characters.
|
||||
|
||||
_NORM_RULES: list[tuple[re.Pattern, str]] = [
|
||||
# Temperature: +12°C / -5°С / 12 °C → плюс двенадцать градусов цельсия
|
||||
(re.compile(r"([+-]?\d+)\s*°\s*[CСcс]", re.IGNORECASE), r"\1 градусов цельсия"),
|
||||
# Bare degree sign: 90° → 90 градусов
|
||||
(re.compile(r"(\d+)\s*°"), r"\1 градусов"),
|
||||
# Percent
|
||||
(re.compile(r"(\d+)\s*%"), r"\1 процентов"),
|
||||
# Speed: m/s, м/с, km/h, км/ч
|
||||
(re.compile(r"\bm/s\b", re.IGNORECASE), "метров в секунду"),
|
||||
(re.compile(r"\bм/с\b"), "метров в секунду"),
|
||||
(re.compile(r"\bkm/h\b", re.IGNORECASE), "километров в час"),
|
||||
(re.compile(r"\bкм/ч\b"), "километров в час"),
|
||||
# Currency
|
||||
(re.compile(r"\$\s*(\d+)"), r"\1 долларов"),
|
||||
(re.compile(r"(\d+)\s*\$"), r"\1 долларов"),
|
||||
(re.compile(r"€\s*(\d+)"), r"\1 евро"),
|
||||
(re.compile(r"(\d+)\s*€"), r"\1 евро"),
|
||||
(re.compile(r"(\d+)\s*₽"), r"\1 рублей"),
|
||||
# Plus/minus signs before numbers
|
||||
(re.compile(r"\+(\d)"), r"плюс \1"),
|
||||
(re.compile(r"-(\d)"), r"минус \1"),
|
||||
# Common abbreviations
|
||||
(re.compile(r"\bкг\b"), "килограмм"),
|
||||
(re.compile(r"\bг\b(?=\s|$)"), "грамм"),
|
||||
(re.compile(r"\bмм\b"), "миллиметров"),
|
||||
(re.compile(r"\bсм\b"), "сантиметров"),
|
||||
(re.compile(r"\bкм\b"), "километров"),
|
||||
# Strip remaining special chars that TTS can't handle
|
||||
(re.compile(r"[°•·†‡§¶©®™«»<>{}[\]|\\~^`]"), ""),
|
||||
]
|
||||
|
||||
|
||||
def normalize_for_tts(text: str) -> str:
|
||||
"""Replace symbols with spoken Russian equivalents."""
|
||||
for pattern, replacement in _NORM_RULES:
|
||||
text = pattern.sub(replacement, text)
|
||||
return text
|
||||
|
||||
|
||||
class TTSTextNormalizer(FrameProcessor):
|
||||
"""Intercepts TextFrames between LLM and TTS, normalizing symbols to words."""
|
||||
|
||||
async def process_frame(self, frame, direction: FrameDirection = FrameDirection.DOWNSTREAM):
|
||||
await super().process_frame(frame, direction)
|
||||
if isinstance(frame, TextFrame):
|
||||
original = frame.text
|
||||
normalized = normalize_for_tts(original)
|
||||
if normalized != original:
|
||||
logger.debug(f"TTSTextNormalizer: {original!r} → {normalized!r}")
|
||||
await self.push_frame(TextFrame(text=normalized), direction)
|
||||
else:
|
||||
await self.push_frame(frame, direction)
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# ── Config ────────────────────────────────────────────────────────────────────
|
||||
LIVEKIT_URL = os.getenv("LIVEKIT_URL", "ws://host.docker.internal:7880") # bot connects here
|
||||
LIVEKIT_PUBLIC_URL = os.getenv("LIVEKIT_PUBLIC_URL", "wss://lk.alogins.net") # browser connects here
|
||||
LIVEKIT_API_KEY = os.getenv("LIVEKIT_API_KEY", "devkey")
|
||||
LIVEKIT_SECRET = os.getenv("LIVEKIT_SECRET", "")
|
||||
ADOLF_URL = os.getenv("ADOLF_URL", "http://host.docker.internal:8000/v1")
|
||||
STT_URL = os.getenv("STT_URL", "http://host.docker.internal:8880/v1")
|
||||
TTS_URL = os.getenv("TTS_URL", "http://host.docker.internal:8881/v1")
|
||||
STT_MODEL = os.getenv("STT_MODEL", "deepdml/faster-whisper-large-v3-turbo-ct2")
|
||||
TTS_VOICE = os.getenv("TTS_VOICE", "onyx")
|
||||
|
||||
SYSTEM_PROMPT = "You are Adolf, a helpful voice assistant. Keep replies concise — 1-3 sentences. No markdown."
|
||||
|
||||
app = FastAPI(title="Pipecat Voice Bot")
|
||||
app.mount("/static", StaticFiles(directory="static"), name="static")
|
||||
|
||||
|
||||
# ── LiveKit helpers ───────────────────────────────────────────────────────────
|
||||
def _lk_token(room: str, identity: str, is_bot: bool = False) -> str:
|
||||
grants = lkapi.VideoGrants(
|
||||
room_join=True,
|
||||
room=room,
|
||||
can_publish=True,
|
||||
can_subscribe=True,
|
||||
can_publish_data=True,
|
||||
)
|
||||
token = (
|
||||
lkapi.AccessToken(LIVEKIT_API_KEY, LIVEKIT_SECRET)
|
||||
.with_identity(identity)
|
||||
.with_name("Adolf Bot" if is_bot else identity)
|
||||
.with_grants(grants)
|
||||
)
|
||||
return token.to_jwt()
|
||||
|
||||
|
||||
async def _create_room(room_name: str) -> None:
|
||||
lk = lkapi.LiveKitAPI(LIVEKIT_URL, LIVEKIT_API_KEY, LIVEKIT_SECRET)
|
||||
try:
|
||||
await lk.room.create_room(
|
||||
lkapi.CreateRoomRequest(name=room_name, empty_timeout=300, max_participants=5)
|
||||
)
|
||||
finally:
|
||||
await lk.aclose()
|
||||
|
||||
|
||||
# ── Pipecat pipeline ──────────────────────────────────────────────────────────
|
||||
async def _run_bot(room_name: str) -> None:
|
||||
bot_token = _lk_token(room_name, "pipecat-bot", is_bot=True)
|
||||
|
||||
transport = LiveKitTransport(
|
||||
url=LIVEKIT_URL,
|
||||
token=bot_token,
|
||||
room_name=room_name,
|
||||
params=LiveKitParams(
|
||||
audio_in_enabled=True,
|
||||
audio_out_enabled=True,
|
||||
vad_enabled=True,
|
||||
vad_analyzer=SileroVADAnalyzer(params=VADParams(
|
||||
stop_secs=0.8, # wait 0.8s of silence before end-of-speech
|
||||
start_secs=0.2, # start speech detection after 0.2s
|
||||
confidence=0.85, # high confidence to avoid triggering on ambient noise
|
||||
)),
|
||||
),
|
||||
)
|
||||
|
||||
stt = OpenAISTTService(
|
||||
api_key="dummy",
|
||||
base_url=STT_URL,
|
||||
model=STT_MODEL,
|
||||
language="ru",
|
||||
)
|
||||
|
||||
llm = OpenAILLMService(
|
||||
api_key="dummy",
|
||||
base_url=ADOLF_URL,
|
||||
model="adolf-light",
|
||||
)
|
||||
|
||||
tts = OpenAITTSService(
|
||||
api_key="dummy",
|
||||
base_url=TTS_URL,
|
||||
model="silero",
|
||||
voice=TTS_VOICE,
|
||||
)
|
||||
|
||||
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
|
||||
context = OpenAILLMContext(messages)
|
||||
context_aggregator = llm.create_context_aggregator(context)
|
||||
|
||||
normalizer = TTSTextNormalizer()
|
||||
|
||||
pipeline = Pipeline([
|
||||
transport.input(),
|
||||
stt,
|
||||
context_aggregator.user(),
|
||||
llm,
|
||||
normalizer,
|
||||
tts,
|
||||
transport.output(),
|
||||
context_aggregator.assistant(),
|
||||
])
|
||||
|
||||
task = PipelineTask(pipeline, params=PipelineParams(allow_interruptions=False))
|
||||
|
||||
@transport.event_handler("on_participant_disconnected")
|
||||
async def on_disconnect(transport, participant):
|
||||
identity = participant if isinstance(participant, str) else getattr(participant, "identity", str(participant))
|
||||
logger.info(f"Participant {identity} left — stopping bot")
|
||||
await task.cancel()
|
||||
|
||||
runner = PipelineRunner()
|
||||
logger.info(f"Bot starting in room={room_name}")
|
||||
await runner.run(task)
|
||||
logger.info(f"Bot done in room={room_name}")
|
||||
|
||||
|
||||
# ── API ───────────────────────────────────────────────────────────────────────
|
||||
class ConnectResponse(BaseModel):
|
||||
room: str
|
||||
token: str
|
||||
url: str
|
||||
|
||||
|
||||
@app.post("/connect", response_model=ConnectResponse)
|
||||
async def connect():
|
||||
room_name = f"voice-{uuid.uuid4().hex[:6]}"
|
||||
await _create_room(room_name)
|
||||
user_token = _lk_token(room_name, "user")
|
||||
asyncio.create_task(_run_bot(room_name))
|
||||
return ConnectResponse(room=room_name, token=user_token, url=LIVEKIT_PUBLIC_URL)
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
async def health():
|
||||
return {"status": "ok"}
|
||||
|
||||
|
||||
@app.get("/", response_class=HTMLResponse)
|
||||
async def index():
|
||||
with open("static/index.html") as f:
|
||||
return f.read()
|
||||
241
ai/pipecat/static/index.html
Normal file
241
ai/pipecat/static/index.html
Normal file
@@ -0,0 +1,241 @@
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1">
|
||||
<title>Adolf Voice</title>
|
||||
<style>
|
||||
* { box-sizing: border-box; margin: 0; padding: 0; }
|
||||
body {
|
||||
font-family: system-ui, sans-serif;
|
||||
background: #0f0f0f;
|
||||
color: #e0e0e0;
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
min-height: 100vh;
|
||||
}
|
||||
.card {
|
||||
background: #1a1a1a;
|
||||
border: 1px solid #2a2a2a;
|
||||
border-radius: 16px;
|
||||
padding: 40px;
|
||||
text-align: center;
|
||||
width: 360px;
|
||||
}
|
||||
h1 { font-size: 1.4rem; font-weight: 600; margin-bottom: 8px; }
|
||||
.subtitle { color: #666; font-size: 0.85rem; margin-bottom: 32px; }
|
||||
#orb {
|
||||
width: 100px;
|
||||
height: 100px;
|
||||
border-radius: 50%;
|
||||
background: radial-gradient(circle, #3a3a3a 0%, #1a1a1a 100%);
|
||||
border: 2px solid #333;
|
||||
margin: 0 auto 24px;
|
||||
cursor: pointer;
|
||||
transition: all 0.3s ease;
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
font-size: 2rem;
|
||||
user-select: none;
|
||||
}
|
||||
#orb.listening {
|
||||
background: radial-gradient(circle, #1e3a5f 0%, #0d1f33 100%);
|
||||
border-color: #3b82f6;
|
||||
box-shadow: 0 0 20px #3b82f640;
|
||||
animation: pulse-blue 1.5s ease-in-out infinite;
|
||||
}
|
||||
#orb.speaking {
|
||||
background: radial-gradient(circle, #1e4034 0%, #0d2018 100%);
|
||||
border-color: #22c55e;
|
||||
box-shadow: 0 0 20px #22c55e40;
|
||||
animation: pulse-green 0.8s ease-in-out infinite;
|
||||
}
|
||||
#orb.thinking {
|
||||
background: radial-gradient(circle, #3a2e1e 0%, #1a160d 100%);
|
||||
border-color: #f59e0b;
|
||||
box-shadow: 0 0 20px #f59e0b40;
|
||||
animation: pulse-amber 1s ease-in-out infinite;
|
||||
}
|
||||
#orb.user-speaking {
|
||||
background: radial-gradient(circle, #3a1e3a 0%, #1a0d1a 100%);
|
||||
border-color: #a855f7;
|
||||
box-shadow: 0 0 20px #a855f740;
|
||||
animation: pulse-purple 0.6s ease-in-out infinite;
|
||||
}
|
||||
@keyframes pulse-blue { 0%,100%{box-shadow:0 0 20px #3b82f640} 50%{box-shadow:0 0 35px #3b82f680} }
|
||||
@keyframes pulse-green { 0%,100%{box-shadow:0 0 20px #22c55e40} 50%{box-shadow:0 0 35px #22c55e80} }
|
||||
@keyframes pulse-amber { 0%,100%{box-shadow:0 0 20px #f59e0b40} 50%{box-shadow:0 0 35px #f59e0b80} }
|
||||
@keyframes pulse-purple { 0%,100%{box-shadow:0 0 20px #a855f740} 50%{box-shadow:0 0 35px #a855f780} }
|
||||
#status {
|
||||
font-size: 0.9rem;
|
||||
color: #888;
|
||||
margin-bottom: 16px;
|
||||
min-height: 1.2em;
|
||||
}
|
||||
#transcript {
|
||||
font-size: 0.8rem;
|
||||
color: #555;
|
||||
margin-bottom: 20px;
|
||||
min-height: 2.4em;
|
||||
line-height: 1.4;
|
||||
font-style: italic;
|
||||
word-break: break-word;
|
||||
}
|
||||
#transcript .user-text { color: #7ba8d4; font-style: normal; }
|
||||
#transcript .bot-text { color: #6ab88a; font-style: normal; }
|
||||
#btn {
|
||||
background: #2a2a2a;
|
||||
border: 1px solid #3a3a3a;
|
||||
color: #e0e0e0;
|
||||
padding: 10px 28px;
|
||||
border-radius: 8px;
|
||||
font-size: 0.9rem;
|
||||
cursor: pointer;
|
||||
transition: background 0.2s;
|
||||
}
|
||||
#btn:hover { background: #333; }
|
||||
#btn:disabled { opacity: 0.4; cursor: default; }
|
||||
#btn.active { border-color: #ef4444; color: #ef4444; }
|
||||
</style>
|
||||
</head>
|
||||
<body>
|
||||
<div class="card">
|
||||
<h1>Adolf</h1>
|
||||
<p class="subtitle">Voice assistant</p>
|
||||
<div id="orb" onclick="toggle()">🎙️</div>
|
||||
<div id="status">Press to connect</div>
|
||||
<div id="transcript"></div>
|
||||
<button id="btn" onclick="toggle()">Connect</button>
|
||||
</div>
|
||||
|
||||
<script src="https://cdn.jsdelivr.net/npm/livekit-client/dist/livekit-client.umd.min.js"></script>
|
||||
<script>
|
||||
let room = null;
|
||||
let audioCtx = null;
|
||||
|
||||
// Unlock browser autoplay — must happen on first user gesture
|
||||
function unlockAudio() {
|
||||
if (!audioCtx) {
|
||||
audioCtx = new (window.AudioContext || window.webkitAudioContext)();
|
||||
if (audioCtx.state === 'suspended') audioCtx.resume();
|
||||
}
|
||||
}
|
||||
|
||||
function setUI(state, msg) {
|
||||
const orb = document.getElementById('orb');
|
||||
const status = document.getElementById('status');
|
||||
const btn = document.getElementById('btn');
|
||||
orb.className = state || '';
|
||||
status.textContent = msg;
|
||||
if (state === null) {
|
||||
btn.textContent = 'Connect';
|
||||
btn.classList.remove('active');
|
||||
orb.textContent = '🎙️';
|
||||
} else {
|
||||
btn.textContent = 'Disconnect';
|
||||
btn.classList.add('active');
|
||||
orb.textContent = state === 'thinking' ? '💭' :
|
||||
state === 'speaking' ? '🔊' :
|
||||
state === 'user-speaking' ? '🗣️' : '🎙️';
|
||||
}
|
||||
}
|
||||
|
||||
function addTranscript(role, text) {
|
||||
const div = document.getElementById('transcript');
|
||||
const cls = role === 'user' ? 'user-text' : 'bot-text';
|
||||
const prefix = role === 'user' ? 'You: ' : 'Adolf: ';
|
||||
div.innerHTML = `<span class="${cls}">${prefix}${text}</span>`;
|
||||
}
|
||||
|
||||
async function toggle() {
|
||||
unlockAudio();
|
||||
if (room) {
|
||||
room.disconnect();
|
||||
return;
|
||||
}
|
||||
|
||||
document.getElementById('btn').disabled = true;
|
||||
setUI('thinking', 'Connecting…');
|
||||
|
||||
try {
|
||||
const res = await fetch('/connect', { method: 'POST' });
|
||||
const { token, url } = await res.json();
|
||||
|
||||
room = new LivekitClient.Room({ adaptiveStream: true, dynacast: true });
|
||||
|
||||
room.on(LivekitClient.RoomEvent.Connected, () => {
|
||||
setUI('listening', 'Listening…');
|
||||
document.getElementById('btn').disabled = false;
|
||||
});
|
||||
|
||||
room.on(LivekitClient.RoomEvent.Disconnected, () => {
|
||||
setUI(null, 'Press to connect');
|
||||
document.getElementById('btn').disabled = false;
|
||||
document.getElementById('transcript').innerHTML = '';
|
||||
room = null;
|
||||
});
|
||||
|
||||
room.on(LivekitClient.RoomEvent.ActiveSpeakersChanged, (speakers) => {
|
||||
if (!room) return;
|
||||
const botSpeaking = speakers.some(s => s.identity === 'pipecat-bot');
|
||||
const userSpeaking = speakers.some(s => s.identity === 'user');
|
||||
if (botSpeaking) {
|
||||
setUI('speaking', 'Adolf is speaking…');
|
||||
} else if (userSpeaking) {
|
||||
setUI('user-speaking', 'Listening to you…');
|
||||
} else {
|
||||
setUI('listening', 'Listening…');
|
||||
}
|
||||
});
|
||||
|
||||
// Attach remote audio so browser plays it
|
||||
room.on(LivekitClient.RoomEvent.TrackSubscribed, (track, pub, participant) => {
|
||||
if (track.kind === 'audio') {
|
||||
// Remove old element if any
|
||||
const old = document.getElementById(`audio-${participant.identity}`);
|
||||
if (old) old.remove();
|
||||
const el = track.attach();
|
||||
el.id = `audio-${participant.identity}`;
|
||||
el.autoplay = true;
|
||||
// Resume audio context on attach to beat autoplay restrictions
|
||||
if (audioCtx && audioCtx.state === 'suspended') audioCtx.resume();
|
||||
document.body.appendChild(el);
|
||||
el.play().catch(() => {});
|
||||
}
|
||||
});
|
||||
|
||||
room.on(LivekitClient.RoomEvent.TrackUnsubscribed, (track) => {
|
||||
track.detach().forEach(el => el.remove());
|
||||
});
|
||||
|
||||
room.on(LivekitClient.RoomEvent.ParticipantConnected, (p) => {
|
||||
if (p.identity === 'pipecat-bot') {
|
||||
setUI('listening', 'Listening…');
|
||||
}
|
||||
});
|
||||
|
||||
// Data messages from bot (transcripts/events if pipecat sends them)
|
||||
room.on(LivekitClient.RoomEvent.DataReceived, (data, participant) => {
|
||||
try {
|
||||
const msg = JSON.parse(new TextDecoder().decode(data));
|
||||
if (msg.type === 'transcript' && msg.role === 'user') addTranscript('user', msg.text);
|
||||
if (msg.type === 'transcript' && msg.role === 'bot') addTranscript('bot', msg.text);
|
||||
} catch {}
|
||||
});
|
||||
|
||||
const wsUrl = url.replace(/^http/, 'ws');
|
||||
await room.connect(wsUrl, token);
|
||||
await room.localParticipant.setMicrophoneEnabled(true);
|
||||
|
||||
} catch (err) {
|
||||
console.error(err);
|
||||
setUI(null, 'Error: ' + err.message);
|
||||
document.getElementById('btn').disabled = false;
|
||||
room = null;
|
||||
}
|
||||
}
|
||||
</script>
|
||||
</body>
|
||||
</html>
|
||||
122
ai/pipecat/test_pipeline.py
Normal file
122
ai/pipecat/test_pipeline.py
Normal file
@@ -0,0 +1,122 @@
|
||||
"""
|
||||
End-to-end pipeline test:
|
||||
1. Call /connect to get a room + token
|
||||
2. Join the LiveKit room as a Python client
|
||||
3. Publish TTS audio (pre-generated) as microphone input
|
||||
4. Capture bot's audio response and save to file
|
||||
"""
|
||||
import asyncio
|
||||
import wave
|
||||
import struct
|
||||
import httpx
|
||||
import numpy as np
|
||||
from livekit import rtc
|
||||
|
||||
PIPECAT_URL = "http://localhost:8882"
|
||||
TTS_URL = "http://host.docker.internal:8881"
|
||||
OUTPUT_FILE = "/tmp/bot_response.wav"
|
||||
SAMPLE_RATE = 48000
|
||||
NUM_CHANNELS = 1
|
||||
|
||||
|
||||
async def generate_tts_pcm(text: str) -> bytes:
|
||||
"""Get WAV audio from Silero TTS, return raw PCM int16 bytes."""
|
||||
async with httpx.AsyncClient(timeout=30) as c:
|
||||
r = await c.post(f"{TTS_URL}/v1/audio/speech", json={
|
||||
"input": text, "voice": "onyx", "response_format": "wav"
|
||||
})
|
||||
r.raise_for_status()
|
||||
# Skip WAV header (44 bytes) to get raw PCM
|
||||
return r.content[44:]
|
||||
|
||||
|
||||
async def main():
|
||||
# Step 1 — create room
|
||||
print("[test] Creating room...")
|
||||
async with httpx.AsyncClient() as c:
|
||||
r = await c.post(f"{PIPECAT_URL}/connect")
|
||||
r.raise_for_status()
|
||||
creds = r.json()
|
||||
print(f"[test] Room: {creds['room']} URL: {creds['url']}")
|
||||
|
||||
# Step 2 — generate test audio
|
||||
test_phrase = "Привет! Как тебя зовут?"
|
||||
print(f"[test] Generating TTS for: {test_phrase!r}")
|
||||
pcm_bytes = await generate_tts_pcm(test_phrase)
|
||||
print(f"[test] TTS PCM: {len(pcm_bytes)} bytes (~{len(pcm_bytes)//(SAMPLE_RATE*2):.1f}s)")
|
||||
|
||||
# Step 3 — join room
|
||||
room = rtc.Room()
|
||||
received_frames: list[bytes] = []
|
||||
|
||||
@room.on("track_subscribed")
|
||||
def on_track(track, pub, participant):
|
||||
if track.kind == rtc.TrackKind.KIND_AUDIO and participant.identity == "pipecat-bot":
|
||||
print(f"[test] Subscribed to bot audio track")
|
||||
audio_stream = rtc.AudioStream(track, sample_rate=SAMPLE_RATE, num_channels=NUM_CHANNELS)
|
||||
asyncio.ensure_future(_collect_audio(audio_stream, received_frames))
|
||||
|
||||
ws_url = creds["url"].replace("https://", "wss://").replace("http://", "ws://")
|
||||
# Connect internally via host.docker.internal
|
||||
internal_url = "ws://host.docker.internal:7880"
|
||||
print(f"[test] Connecting to LiveKit at {internal_url}...")
|
||||
await room.connect(internal_url, creds["token"])
|
||||
print(f"[test] Connected. Waiting for bot to join...")
|
||||
|
||||
# Wait for bot participant
|
||||
for _ in range(20):
|
||||
if any(p.identity == "pipecat-bot" for p in room.remote_participants.values()):
|
||||
break
|
||||
await asyncio.sleep(0.5)
|
||||
print(f"[test] Participants: {[p.identity for p in room.remote_participants.values()]}")
|
||||
|
||||
# Step 4 — publish audio as microphone
|
||||
print("[test] Publishing audio track...")
|
||||
source = rtc.AudioSource(SAMPLE_RATE, NUM_CHANNELS)
|
||||
local_track = rtc.LocalAudioTrack.create_audio_track("microphone", source)
|
||||
opts = rtc.TrackPublishOptions(source=rtc.TrackSource.SOURCE_MICROPHONE)
|
||||
await room.local_participant.publish_track(local_track, opts)
|
||||
|
||||
# Send PCM in 20ms chunks
|
||||
chunk_samples = SAMPLE_RATE * 20 // 1000 # 960 samples per chunk
|
||||
chunk_bytes = chunk_samples * 2 # int16
|
||||
print(f"[test] Sending {len(pcm_bytes) // chunk_bytes} audio chunks...")
|
||||
for i in range(0, len(pcm_bytes), chunk_bytes):
|
||||
chunk = pcm_bytes[i:i + chunk_bytes]
|
||||
if len(chunk) < chunk_bytes:
|
||||
chunk = chunk + b'\x00' * (chunk_bytes - len(chunk))
|
||||
samples = np.frombuffer(chunk, dtype=np.int16)
|
||||
frame = rtc.AudioFrame(
|
||||
data=samples.tobytes(),
|
||||
sample_rate=SAMPLE_RATE,
|
||||
num_channels=NUM_CHANNELS,
|
||||
samples_per_channel=chunk_samples,
|
||||
)
|
||||
await source.capture_frame(frame)
|
||||
await asyncio.sleep(0.02)
|
||||
|
||||
print("[test] Audio sent. Waiting for bot response (up to 30s)...")
|
||||
await asyncio.sleep(30)
|
||||
|
||||
await room.disconnect()
|
||||
|
||||
# Step 5 — save response
|
||||
if received_frames:
|
||||
total = b"".join(received_frames)
|
||||
print(f"[test] Received {len(total)} bytes of bot audio ({len(total)//(SAMPLE_RATE*2):.1f}s)")
|
||||
with wave.open(OUTPUT_FILE, "wb") as wf:
|
||||
wf.setnchannels(NUM_CHANNELS)
|
||||
wf.setsampwidth(2)
|
||||
wf.setframerate(SAMPLE_RATE)
|
||||
wf.writeframes(total)
|
||||
print(f"[test] Saved to {OUTPUT_FILE}")
|
||||
else:
|
||||
print("[test] No audio received from bot!")
|
||||
|
||||
|
||||
async def _collect_audio(stream: rtc.AudioStream, buf: list):
|
||||
async for event in stream:
|
||||
buf.append(bytes(event.frame.data))
|
||||
|
||||
|
||||
asyncio.run(main())
|
||||
Reference in New Issue
Block a user