Switch from Bifrost to LiteLLM; add Matrix channel; update rules

Infrastructure: - docker-compose.yml: replace bifrost container with LiteLLM proxy (host.docker.internal:4000); complex model → deepseek-r1:free via OpenRouter; add Matrix URL env var; mount logs volume - bifrost-config.json: add auth_config + postgres config_store (archived) Routing: - router.py: full semantic 3-tier classifier rewrite — nomic-embed-text centroids for light/medium/complex; regex pre-classifiers for all tiers; Russian utterance sets expanded - agent.py: wire LiteLLM URL; add dry_run support; add Matrix channel Channels: - channels.py: add Matrix adapter (_matrix_send via mx- session prefix) Rules / docs: - agent-pipeline.md: remove /think prefix requirement; document automatic complex tier classification - llm-inference.md: update BIFROST_URL → LITELLM_URL references; add remote model note for complex tier - ARCHITECTURE.md: deleted (superseded by README.md) Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-03-24 02:14:13 +00:00
parent 54cb940279
commit 1f5e272600
8 changed files with 730 additions and 406 deletions
--- a/docker-compose.yml
+++ b/docker-compose.yml
@@ -1,19 +1,4 @@
 services:
-  bifrost:
-    image: maximhq/bifrost
-    container_name: bifrost
-    ports:
-      - "8080:8080"
-    volumes:
-      - ./bifrost-config.json:/app/data/config.json:ro
-    environment:
-      - APP_DIR=/app/data
-      - APP_PORT=8080
-      - LOG_LEVEL=info
-    extra_hosts:
-      - "host.docker.internal:host-gateway"
-    restart: unless-stopped
-
  deepagents:
    build: .
    container_name: deepagents
@@ -21,25 +6,28 @@ services:
      - "8000:8000"
    environment:
      - PYTHONUNBUFFERED=1
-      # Bifrost gateway — all LLM inference goes through here
-      - BIFROST_URL=http://bifrost:8080/v1
+      # LiteLLM proxy — all LLM inference goes through here
+      - LITELLM_URL=http://host.docker.internal:4000/v1
+      - LITELLM_API_KEY=sk-fjQC1BxAiGFSMs
      # Direct Ollama GPU URL — used only by VRAMManager for flush/prewarm
      - OLLAMA_BASE_URL=http://host.docker.internal:11436
      - DEEPAGENTS_MODEL=qwen3:4b
-      - DEEPAGENTS_COMPLEX_MODEL=qwen3:8b
+      - DEEPAGENTS_COMPLEX_MODEL=deepseek/deepseek-r1:free
      - DEEPAGENTS_ROUTER_MODEL=qwen2.5:1.5b
      - SEARXNG_URL=http://host.docker.internal:11437
      - GRAMMY_URL=http://grammy:3001
+      - MATRIX_URL=http://host.docker.internal:3002
      - CRAWL4AI_URL=http://crawl4ai:11235
      - ROUTECHECK_URL=http://routecheck:8090
      - ROUTECHECK_TOKEN=${ROUTECHECK_TOKEN}
+    volumes:
+      - ./logs:/app/logs
    extra_hosts:
      - "host.docker.internal:host-gateway"
    depends_on:
      - openmemory
      - grammy
      - crawl4ai
-      - bifrost
      - routecheck
    restart: unless-stopped