openai: add AI stack (litellm + langfuse + pipecat + silero-tts) and oO aliases

- LiteLLM proxy with langfuse callbacks, postgres backends, and OpenRouter fallbacks.
- Langfuse observability UI.
- Pipecat voice pipeline (LiveKit + STT + TTS + LLM) and Silero TTS build contexts.
- Ollama tuned for GPU (OLLAMA_NUM_GPU=999, mem_limit=4g, max 2 loaded models).
- open-webui wired to litellm + faster-whisper + silero for voice.
- litellm-config.yaml publishes oO's model aliases (tip-generator, embedder, judge)
  pointing at the host ollama on :11434 so ml/serving can call them via LiteLLM.

.env skipped (secrets).

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
Alvis
2026-04-20 14:28:24 +00:00
parent 52190b63b8
commit 85033136d8
8 changed files with 1020 additions and 7 deletions

18
openai/pipecat/Dockerfile Normal file
View File

@@ -0,0 +1,18 @@
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends gcc g++ && rm -rf /var/lib/apt/lists/*
# CPU torch first — prevents silero-vad from pulling in the CUDA variant
RUN pip install --no-cache-dir torch --index-url https://download.pytorch.org/whl/cpu
RUN pip install --no-cache-dir \
"pipecat-ai[openai,livekit,silero]" \
"livekit-api" \
fastapi \
"uvicorn[standard]"
WORKDIR /app
COPY . .
EXPOSE 8882
CMD ["uvicorn", "bot:app", "--host", "0.0.0.0", "--port", "8882"]

228
openai/pipecat/bot.py Normal file
View File

@@ -0,0 +1,228 @@
import asyncio
import os
import re
import uuid
import logging
from fastapi import FastAPI
from fastapi.responses import HTMLResponse
from fastapi.staticfiles import StaticFiles
from pydantic import BaseModel
from livekit import api as lkapi
from pipecat.audio.vad.silero import SileroVADAnalyzer
from pipecat.audio.vad.vad_analyzer import VADParams
from pipecat.frames.frames import TextFrame
from pipecat.pipeline.pipeline import Pipeline
from pipecat.pipeline.runner import PipelineRunner
from pipecat.pipeline.task import PipelineParams, PipelineTask
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
from pipecat.processors.frame_processor import FrameProcessor, FrameDirection
from pipecat.services.openai.llm import OpenAILLMService
from pipecat.services.openai.stt import OpenAISTTService
from pipecat.services.openai.tts import OpenAITTSService
from pipecat.transports.livekit.transport import LiveKitTransport, LiveKitParams
# ── TTS text normalizer ──────────────────────────────────────────────────────
# Replaces symbols and abbreviations with spoken Russian words so Silero TTS
# doesn't truncate on unknown characters.
_NORM_RULES: list[tuple[re.Pattern, str]] = [
# Temperature: +12°C / -5°С / 12 °C → плюс двенадцать градусов цельсия
(re.compile(r"([+-]?\d+)\s*°\s*[CСcс]", re.IGNORECASE), r"\1 градусов цельсия"),
# Bare degree sign: 90° → 90 градусов
(re.compile(r"(\d+)\s*°"), r"\1 градусов"),
# Percent
(re.compile(r"(\d+)\s*%"), r"\1 процентов"),
# Speed: m/s, м/с, km/h, км/ч
(re.compile(r"\bm/s\b", re.IGNORECASE), "метров в секунду"),
(re.compile(r"\bм/с\b"), "метров в секунду"),
(re.compile(r"\bkm/h\b", re.IGNORECASE), "километров в час"),
(re.compile(r"\bкм/ч\b"), "километров в час"),
# Currency
(re.compile(r"\$\s*(\d+)"), r"\1 долларов"),
(re.compile(r"(\d+)\s*\$"), r"\1 долларов"),
(re.compile(r"\s*(\d+)"), r"\1 евро"),
(re.compile(r"(\d+)\s*€"), r"\1 евро"),
(re.compile(r"(\d+)\s*₽"), r"\1 рублей"),
# Plus/minus signs before numbers
(re.compile(r"\+(\d)"), r"плюс \1"),
(re.compile(r"-(\d)"), r"минус \1"),
# Common abbreviations
(re.compile(r"\г\b"), "килограмм"),
(re.compile(r"\bг\b(?=\s|$)"), "грамм"),
(re.compile(r"\bмм\b"), "миллиметров"),
(re.compile(r"\bсм\b"), "сантиметров"),
(re.compile(r"\bкм\b"), "километров"),
# Strip remaining special chars that TTS can't handle
(re.compile(r"[°•·†‡§¶©®™«»<>{}[\]|\\~^`]"), ""),
]
def normalize_for_tts(text: str) -> str:
"""Replace symbols with spoken Russian equivalents."""
for pattern, replacement in _NORM_RULES:
text = pattern.sub(replacement, text)
return text
class TTSTextNormalizer(FrameProcessor):
"""Intercepts TextFrames between LLM and TTS, normalizing symbols to words."""
async def process_frame(self, frame, direction: FrameDirection = FrameDirection.DOWNSTREAM):
await super().process_frame(frame, direction)
if isinstance(frame, TextFrame):
original = frame.text
normalized = normalize_for_tts(original)
if normalized != original:
logger.debug(f"TTSTextNormalizer: {original!r}{normalized!r}")
await self.push_frame(TextFrame(text=normalized), direction)
else:
await self.push_frame(frame, direction)
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
# ── Config ────────────────────────────────────────────────────────────────────
LIVEKIT_URL = os.getenv("LIVEKIT_URL", "ws://host.docker.internal:7880") # bot connects here
LIVEKIT_PUBLIC_URL = os.getenv("LIVEKIT_PUBLIC_URL", "wss://lk.alogins.net") # browser connects here
LIVEKIT_API_KEY = os.getenv("LIVEKIT_API_KEY", "devkey")
LIVEKIT_SECRET = os.getenv("LIVEKIT_SECRET", "")
ADOLF_URL = os.getenv("ADOLF_URL", "http://host.docker.internal:8000/v1")
STT_URL = os.getenv("STT_URL", "http://host.docker.internal:8880/v1")
TTS_URL = os.getenv("TTS_URL", "http://host.docker.internal:8881/v1")
STT_MODEL = os.getenv("STT_MODEL", "deepdml/faster-whisper-large-v3-turbo-ct2")
TTS_VOICE = os.getenv("TTS_VOICE", "onyx")
SYSTEM_PROMPT = "You are Adolf, a helpful voice assistant. Keep replies concise — 1-3 sentences. No markdown."
app = FastAPI(title="Pipecat Voice Bot")
app.mount("/static", StaticFiles(directory="static"), name="static")
# ── LiveKit helpers ───────────────────────────────────────────────────────────
def _lk_token(room: str, identity: str, is_bot: bool = False) -> str:
grants = lkapi.VideoGrants(
room_join=True,
room=room,
can_publish=True,
can_subscribe=True,
can_publish_data=True,
)
token = (
lkapi.AccessToken(LIVEKIT_API_KEY, LIVEKIT_SECRET)
.with_identity(identity)
.with_name("Adolf Bot" if is_bot else identity)
.with_grants(grants)
)
return token.to_jwt()
async def _create_room(room_name: str) -> None:
lk = lkapi.LiveKitAPI(LIVEKIT_URL, LIVEKIT_API_KEY, LIVEKIT_SECRET)
try:
await lk.room.create_room(
lkapi.CreateRoomRequest(name=room_name, empty_timeout=300, max_participants=5)
)
finally:
await lk.aclose()
# ── Pipecat pipeline ──────────────────────────────────────────────────────────
async def _run_bot(room_name: str) -> None:
bot_token = _lk_token(room_name, "pipecat-bot", is_bot=True)
transport = LiveKitTransport(
url=LIVEKIT_URL,
token=bot_token,
room_name=room_name,
params=LiveKitParams(
audio_in_enabled=True,
audio_out_enabled=True,
vad_enabled=True,
vad_analyzer=SileroVADAnalyzer(params=VADParams(
stop_secs=0.8, # wait 0.8s of silence before end-of-speech
start_secs=0.2, # start speech detection after 0.2s
confidence=0.85, # high confidence to avoid triggering on ambient noise
)),
),
)
stt = OpenAISTTService(
api_key="dummy",
base_url=STT_URL,
model=STT_MODEL,
language="ru",
)
llm = OpenAILLMService(
api_key="dummy",
base_url=ADOLF_URL,
model="adolf-light",
)
tts = OpenAITTSService(
api_key="dummy",
base_url=TTS_URL,
model="silero",
voice=TTS_VOICE,
)
messages = [{"role": "system", "content": SYSTEM_PROMPT}]
context = OpenAILLMContext(messages)
context_aggregator = llm.create_context_aggregator(context)
normalizer = TTSTextNormalizer()
pipeline = Pipeline([
transport.input(),
stt,
context_aggregator.user(),
llm,
normalizer,
tts,
transport.output(),
context_aggregator.assistant(),
])
task = PipelineTask(pipeline, params=PipelineParams(allow_interruptions=False))
@transport.event_handler("on_participant_disconnected")
async def on_disconnect(transport, participant):
identity = participant if isinstance(participant, str) else getattr(participant, "identity", str(participant))
logger.info(f"Participant {identity} left — stopping bot")
await task.cancel()
runner = PipelineRunner()
logger.info(f"Bot starting in room={room_name}")
await runner.run(task)
logger.info(f"Bot done in room={room_name}")
# ── API ───────────────────────────────────────────────────────────────────────
class ConnectResponse(BaseModel):
room: str
token: str
url: str
@app.post("/connect", response_model=ConnectResponse)
async def connect():
room_name = f"voice-{uuid.uuid4().hex[:6]}"
await _create_room(room_name)
user_token = _lk_token(room_name, "user")
asyncio.create_task(_run_bot(room_name))
return ConnectResponse(room=room_name, token=user_token, url=LIVEKIT_PUBLIC_URL)
@app.get("/health")
async def health():
return {"status": "ok"}
@app.get("/", response_class=HTMLResponse)
async def index():
with open("static/index.html") as f:
return f.read()

View File

@@ -0,0 +1,241 @@
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>Adolf Voice</title>
<style>
* { box-sizing: border-box; margin: 0; padding: 0; }
body {
font-family: system-ui, sans-serif;
background: #0f0f0f;
color: #e0e0e0;
display: flex;
align-items: center;
justify-content: center;
min-height: 100vh;
}
.card {
background: #1a1a1a;
border: 1px solid #2a2a2a;
border-radius: 16px;
padding: 40px;
text-align: center;
width: 360px;
}
h1 { font-size: 1.4rem; font-weight: 600; margin-bottom: 8px; }
.subtitle { color: #666; font-size: 0.85rem; margin-bottom: 32px; }
#orb {
width: 100px;
height: 100px;
border-radius: 50%;
background: radial-gradient(circle, #3a3a3a 0%, #1a1a1a 100%);
border: 2px solid #333;
margin: 0 auto 24px;
cursor: pointer;
transition: all 0.3s ease;
display: flex;
align-items: center;
justify-content: center;
font-size: 2rem;
user-select: none;
}
#orb.listening {
background: radial-gradient(circle, #1e3a5f 0%, #0d1f33 100%);
border-color: #3b82f6;
box-shadow: 0 0 20px #3b82f640;
animation: pulse-blue 1.5s ease-in-out infinite;
}
#orb.speaking {
background: radial-gradient(circle, #1e4034 0%, #0d2018 100%);
border-color: #22c55e;
box-shadow: 0 0 20px #22c55e40;
animation: pulse-green 0.8s ease-in-out infinite;
}
#orb.thinking {
background: radial-gradient(circle, #3a2e1e 0%, #1a160d 100%);
border-color: #f59e0b;
box-shadow: 0 0 20px #f59e0b40;
animation: pulse-amber 1s ease-in-out infinite;
}
#orb.user-speaking {
background: radial-gradient(circle, #3a1e3a 0%, #1a0d1a 100%);
border-color: #a855f7;
box-shadow: 0 0 20px #a855f740;
animation: pulse-purple 0.6s ease-in-out infinite;
}
@keyframes pulse-blue { 0%,100%{box-shadow:0 0 20px #3b82f640} 50%{box-shadow:0 0 35px #3b82f680} }
@keyframes pulse-green { 0%,100%{box-shadow:0 0 20px #22c55e40} 50%{box-shadow:0 0 35px #22c55e80} }
@keyframes pulse-amber { 0%,100%{box-shadow:0 0 20px #f59e0b40} 50%{box-shadow:0 0 35px #f59e0b80} }
@keyframes pulse-purple { 0%,100%{box-shadow:0 0 20px #a855f740} 50%{box-shadow:0 0 35px #a855f780} }
#status {
font-size: 0.9rem;
color: #888;
margin-bottom: 16px;
min-height: 1.2em;
}
#transcript {
font-size: 0.8rem;
color: #555;
margin-bottom: 20px;
min-height: 2.4em;
line-height: 1.4;
font-style: italic;
word-break: break-word;
}
#transcript .user-text { color: #7ba8d4; font-style: normal; }
#transcript .bot-text { color: #6ab88a; font-style: normal; }
#btn {
background: #2a2a2a;
border: 1px solid #3a3a3a;
color: #e0e0e0;
padding: 10px 28px;
border-radius: 8px;
font-size: 0.9rem;
cursor: pointer;
transition: background 0.2s;
}
#btn:hover { background: #333; }
#btn:disabled { opacity: 0.4; cursor: default; }
#btn.active { border-color: #ef4444; color: #ef4444; }
</style>
</head>
<body>
<div class="card">
<h1>Adolf</h1>
<p class="subtitle">Voice assistant</p>
<div id="orb" onclick="toggle()">🎙️</div>
<div id="status">Press to connect</div>
<div id="transcript"></div>
<button id="btn" onclick="toggle()">Connect</button>
</div>
<script src="https://cdn.jsdelivr.net/npm/livekit-client/dist/livekit-client.umd.min.js"></script>
<script>
let room = null;
let audioCtx = null;
// Unlock browser autoplay — must happen on first user gesture
function unlockAudio() {
if (!audioCtx) {
audioCtx = new (window.AudioContext || window.webkitAudioContext)();
if (audioCtx.state === 'suspended') audioCtx.resume();
}
}
function setUI(state, msg) {
const orb = document.getElementById('orb');
const status = document.getElementById('status');
const btn = document.getElementById('btn');
orb.className = state || '';
status.textContent = msg;
if (state === null) {
btn.textContent = 'Connect';
btn.classList.remove('active');
orb.textContent = '🎙️';
} else {
btn.textContent = 'Disconnect';
btn.classList.add('active');
orb.textContent = state === 'thinking' ? '💭' :
state === 'speaking' ? '🔊' :
state === 'user-speaking' ? '🗣️' : '🎙️';
}
}
function addTranscript(role, text) {
const div = document.getElementById('transcript');
const cls = role === 'user' ? 'user-text' : 'bot-text';
const prefix = role === 'user' ? 'You: ' : 'Adolf: ';
div.innerHTML = `<span class="${cls}">${prefix}${text}</span>`;
}
async function toggle() {
unlockAudio();
if (room) {
room.disconnect();
return;
}
document.getElementById('btn').disabled = true;
setUI('thinking', 'Connecting…');
try {
const res = await fetch('/connect', { method: 'POST' });
const { token, url } = await res.json();
room = new LivekitClient.Room({ adaptiveStream: true, dynacast: true });
room.on(LivekitClient.RoomEvent.Connected, () => {
setUI('listening', 'Listening…');
document.getElementById('btn').disabled = false;
});
room.on(LivekitClient.RoomEvent.Disconnected, () => {
setUI(null, 'Press to connect');
document.getElementById('btn').disabled = false;
document.getElementById('transcript').innerHTML = '';
room = null;
});
room.on(LivekitClient.RoomEvent.ActiveSpeakersChanged, (speakers) => {
if (!room) return;
const botSpeaking = speakers.some(s => s.identity === 'pipecat-bot');
const userSpeaking = speakers.some(s => s.identity === 'user');
if (botSpeaking) {
setUI('speaking', 'Adolf is speaking…');
} else if (userSpeaking) {
setUI('user-speaking', 'Listening to you…');
} else {
setUI('listening', 'Listening…');
}
});
// Attach remote audio so browser plays it
room.on(LivekitClient.RoomEvent.TrackSubscribed, (track, pub, participant) => {
if (track.kind === 'audio') {
// Remove old element if any
const old = document.getElementById(`audio-${participant.identity}`);
if (old) old.remove();
const el = track.attach();
el.id = `audio-${participant.identity}`;
el.autoplay = true;
// Resume audio context on attach to beat autoplay restrictions
if (audioCtx && audioCtx.state === 'suspended') audioCtx.resume();
document.body.appendChild(el);
el.play().catch(() => {});
}
});
room.on(LivekitClient.RoomEvent.TrackUnsubscribed, (track) => {
track.detach().forEach(el => el.remove());
});
room.on(LivekitClient.RoomEvent.ParticipantConnected, (p) => {
if (p.identity === 'pipecat-bot') {
setUI('listening', 'Listening…');
}
});
// Data messages from bot (transcripts/events if pipecat sends them)
room.on(LivekitClient.RoomEvent.DataReceived, (data, participant) => {
try {
const msg = JSON.parse(new TextDecoder().decode(data));
if (msg.type === 'transcript' && msg.role === 'user') addTranscript('user', msg.text);
if (msg.type === 'transcript' && msg.role === 'bot') addTranscript('bot', msg.text);
} catch {}
});
const wsUrl = url.replace(/^http/, 'ws');
await room.connect(wsUrl, token);
await room.localParticipant.setMicrophoneEnabled(true);
} catch (err) {
console.error(err);
setUI(null, 'Error: ' + err.message);
document.getElementById('btn').disabled = false;
room = null;
}
}
</script>
</body>
</html>

View File

@@ -0,0 +1,122 @@
"""
End-to-end pipeline test:
1. Call /connect to get a room + token
2. Join the LiveKit room as a Python client
3. Publish TTS audio (pre-generated) as microphone input
4. Capture bot's audio response and save to file
"""
import asyncio
import wave
import struct
import httpx
import numpy as np
from livekit import rtc
PIPECAT_URL = "http://localhost:8882"
TTS_URL = "http://host.docker.internal:8881"
OUTPUT_FILE = "/tmp/bot_response.wav"
SAMPLE_RATE = 48000
NUM_CHANNELS = 1
async def generate_tts_pcm(text: str) -> bytes:
"""Get WAV audio from Silero TTS, return raw PCM int16 bytes."""
async with httpx.AsyncClient(timeout=30) as c:
r = await c.post(f"{TTS_URL}/v1/audio/speech", json={
"input": text, "voice": "onyx", "response_format": "wav"
})
r.raise_for_status()
# Skip WAV header (44 bytes) to get raw PCM
return r.content[44:]
async def main():
# Step 1 — create room
print("[test] Creating room...")
async with httpx.AsyncClient() as c:
r = await c.post(f"{PIPECAT_URL}/connect")
r.raise_for_status()
creds = r.json()
print(f"[test] Room: {creds['room']} URL: {creds['url']}")
# Step 2 — generate test audio
test_phrase = "Привет! Как тебя зовут?"
print(f"[test] Generating TTS for: {test_phrase!r}")
pcm_bytes = await generate_tts_pcm(test_phrase)
print(f"[test] TTS PCM: {len(pcm_bytes)} bytes (~{len(pcm_bytes)//(SAMPLE_RATE*2):.1f}s)")
# Step 3 — join room
room = rtc.Room()
received_frames: list[bytes] = []
@room.on("track_subscribed")
def on_track(track, pub, participant):
if track.kind == rtc.TrackKind.KIND_AUDIO and participant.identity == "pipecat-bot":
print(f"[test] Subscribed to bot audio track")
audio_stream = rtc.AudioStream(track, sample_rate=SAMPLE_RATE, num_channels=NUM_CHANNELS)
asyncio.ensure_future(_collect_audio(audio_stream, received_frames))
ws_url = creds["url"].replace("https://", "wss://").replace("http://", "ws://")
# Connect internally via host.docker.internal
internal_url = "ws://host.docker.internal:7880"
print(f"[test] Connecting to LiveKit at {internal_url}...")
await room.connect(internal_url, creds["token"])
print(f"[test] Connected. Waiting for bot to join...")
# Wait for bot participant
for _ in range(20):
if any(p.identity == "pipecat-bot" for p in room.remote_participants.values()):
break
await asyncio.sleep(0.5)
print(f"[test] Participants: {[p.identity for p in room.remote_participants.values()]}")
# Step 4 — publish audio as microphone
print("[test] Publishing audio track...")
source = rtc.AudioSource(SAMPLE_RATE, NUM_CHANNELS)
local_track = rtc.LocalAudioTrack.create_audio_track("microphone", source)
opts = rtc.TrackPublishOptions(source=rtc.TrackSource.SOURCE_MICROPHONE)
await room.local_participant.publish_track(local_track, opts)
# Send PCM in 20ms chunks
chunk_samples = SAMPLE_RATE * 20 // 1000 # 960 samples per chunk
chunk_bytes = chunk_samples * 2 # int16
print(f"[test] Sending {len(pcm_bytes) // chunk_bytes} audio chunks...")
for i in range(0, len(pcm_bytes), chunk_bytes):
chunk = pcm_bytes[i:i + chunk_bytes]
if len(chunk) < chunk_bytes:
chunk = chunk + b'\x00' * (chunk_bytes - len(chunk))
samples = np.frombuffer(chunk, dtype=np.int16)
frame = rtc.AudioFrame(
data=samples.tobytes(),
sample_rate=SAMPLE_RATE,
num_channels=NUM_CHANNELS,
samples_per_channel=chunk_samples,
)
await source.capture_frame(frame)
await asyncio.sleep(0.02)
print("[test] Audio sent. Waiting for bot response (up to 30s)...")
await asyncio.sleep(30)
await room.disconnect()
# Step 5 — save response
if received_frames:
total = b"".join(received_frames)
print(f"[test] Received {len(total)} bytes of bot audio ({len(total)//(SAMPLE_RATE*2):.1f}s)")
with wave.open(OUTPUT_FILE, "wb") as wf:
wf.setnchannels(NUM_CHANNELS)
wf.setsampwidth(2)
wf.setframerate(SAMPLE_RATE)
wf.writeframes(total)
print(f"[test] Saved to {OUTPUT_FILE}")
else:
print("[test] No audio received from bot!")
async def _collect_audio(stream: rtc.AudioStream, buf: list):
async for event in stream:
buf.append(bytes(event.frame.data))
asyncio.run(main())