Per DESIGN-a2a-agents.md v2.1 §2-3b: models are the scarce queued resource, version-controlled here rather than hardcoded in callers. - model-registry.yaml: kimi (main reasoning, quota-gated), local-small (ollama/gemma3:4b, always-on cheap tier), bge-m3 (embedder + routing classifier, never-evict), tei-reranker (never-evict, interactive- critical), paid-fallback (metered, opt-in only, unreachable by default via empty routing.metered_opt_in). GPU residency policy carries the never-evict set, co-residency groups, and measured baseline (bge-m3+gemma3:4b+tei-reranker ~6.2/8GB on the GTX 1070). - model_registry.py: resolve(tier) picks an available model without the caller naming one, gated so a metered model is only reachable with both allow_metered=True and an opted-in virtual key; to_probe_config() bridges registry quota data into kb_worker.py's existing Probe classes (no duplicated probe logic); preload_check() expresses the §3b pre-load VRAM check purely from registry data. Gap noted for follow-up: bge-m3 has no litellm-config.yaml model_list entry yet (embedder there still points at ollama/nomic-embed-text on a different port) — out of scope here, registry documents it as-is.
193 lines
9.4 KiB
YAML
193 lines
9.4 KiB
YAML
# Model registry — models are the scarce queued resource.
|
|
#
|
|
# Per DESIGN-a2a-agents.md v2.1 §2-3b (commit df2071d5), kanboard task #133
|
|
# (A2A-1). Version-controlled here; the "model plane" (§3) and the fabric's
|
|
# workers/routers read this data — they do not duplicate it. Lifecycle a(t)
|
|
# *probe mechanics* (QuotaProbe, GPUResidencyProbe, ...) live in
|
|
# kanboard/bin/kb_worker.py; this registry supplies the *parameters* those
|
|
# probes consume (commands, fields, thresholds, VRAM footprints).
|
|
#
|
|
# Scope constraint (alvis, §3a): NO METERED API BY DEFAULT. The workflow is
|
|
# Claude Code (a flat-subscription runtime -> agent registry #134, not here)
|
|
# + the Kimi wrapper + a local GPU embedder + a small weak local model. The
|
|
# governor arbitrates quota and GPU, not money. Any metered model below is
|
|
# `metered: true, opt_in_required: true` and carries no default route to it
|
|
# (see routing.metered_opt_in: [] at the bottom — empty means unreachable).
|
|
#
|
|
# Read with model_registry.py (same directory): resolve(), preload_check().
|
|
|
|
schema_version: 1
|
|
|
|
models:
|
|
# ── kimi — main reasoning ──────────────────────────────────────────────
|
|
# Flat Moonshot/Kimi subscription via `kimi login`, wrapped by two
|
|
# independent Kimi-CLI containers (own OAuth creds volume each). Not
|
|
# behind LiteLLM today — callers hit the wrapper HTTP endpoints directly.
|
|
- id: kimi
|
|
role: "main reasoning (adolf-llm / hindsight-llm Kimi-CLI wrappers)"
|
|
litellm_model_name: null
|
|
endpoints:
|
|
- name: adolf-llm
|
|
purpose: "Adolf's conversational backbone"
|
|
url: "http://adolf-llm:8010"
|
|
usage_url: "http://localhost:8010/usage"
|
|
- name: hindsight-llm
|
|
purpose: "Hindsight's structured-extraction LLM (HINDSIGHT_API_LLM_MODEL)"
|
|
url: "http://hindsight-llm:8012/v1"
|
|
model_name: "openai/hindsight-llm"
|
|
tier: large
|
|
context_tokens: 200000 # Moonshot Kimi K2 context window; re-verify if the CLI's pinned model changes
|
|
tool_use_quality: high
|
|
lifecycle: quota-gated
|
|
quota:
|
|
probe_command: ["kimi-usage", "--compact"]
|
|
windows:
|
|
- name: 5h
|
|
field: "window_5h.pct" # adolf-llm server.js normalizeKimiUsage() field name
|
|
approx_limit: "~60 msgs/5h"
|
|
- name: weekly
|
|
field: "weekly.pct"
|
|
approx_limit: "~300 msgs/wk"
|
|
threshold_pct: 95
|
|
gpu_residency: null
|
|
cost_class: subscription # flat-rate, not metered — quota is the constraint, not spend
|
|
metered: false
|
|
opt_in_required: false
|
|
|
|
# ── local-small — the cheap tier ───────────────────────────────────────
|
|
# ollama/gemma3:4b on the GPU ollama instance. Already the live model for
|
|
# Hindsight consolidation/reflect (HINDSIGHT_API_CONSOLIDATION_LLM_MODEL /
|
|
# HINDSIGHT_API_REFLECT_LLM_MODEL, kb#88) and exposed via LiteLLM.
|
|
- id: local-small
|
|
role: "cheap tier — ollama small/weak local model (background extraction, consolidation, reflect)"
|
|
litellm_model_name: "ollama/gemma3:4b" # openai/litellm-config.yaml model_list entry
|
|
endpoints:
|
|
- name: ollama-direct
|
|
url: "http://host.docker.internal:11436"
|
|
- name: via-litellm
|
|
url: "http://litellm:4000/v1"
|
|
tier: small
|
|
context_tokens: 8192 # gemma3:4b default ctx; re-verify with `ollama show gemma3:4b` if raised
|
|
tool_use_quality: low
|
|
lifecycle: always-on
|
|
quota: null
|
|
gpu_residency:
|
|
vram_mb: 4000 # approx measured footprint, within the shared 8GB card (see gpu_residency_policy below)
|
|
never_evict: false # evictable — a bigger model may push it out; that's a silent regression to catch, not prevent here
|
|
co_residency_group: interactive-local
|
|
cost_class: free
|
|
metered: false
|
|
opt_in_required: false
|
|
|
|
# ── bge-m3 — embedder + routing classifier ─────────────────────────────
|
|
# Never-evict: it's both Hindsight's recall embedder AND (design §3a) the
|
|
# embedding model LiteLLM Auto Router's semantic-router classifier will
|
|
# use for tier/complexity routing — losing it degrades both recall AND
|
|
# routing at once.
|
|
- id: bge-m3
|
|
role: "embedder — also the routing classifier (§3a, LiteLLM Auto Router / semantic-router)"
|
|
litellm_model_name: null # NOT YET wired into litellm-config.yaml — gap, see model_registry.py module docstring
|
|
endpoints:
|
|
- name: ollama-direct
|
|
url: "http://host.docker.internal:11436"
|
|
openai_compatible_path: "/v1/embeddings"
|
|
tier: small
|
|
context_tokens: 8192
|
|
tool_use_quality: "n/a" # embedder, not a chat/tool-use model
|
|
lifecycle: always-on
|
|
quota: null
|
|
gpu_residency:
|
|
vram_mb: 1200
|
|
never_evict: true
|
|
co_residency_group: interactive-local
|
|
cost_class: free
|
|
metered: false
|
|
opt_in_required: false
|
|
|
|
# ── tei-reranker — interactive-critical, never-evict ───────────────────
|
|
# Not an LLM (cross-encoder rerank sidecar for Hindsight recall, kb#87)
|
|
# but carries the same GPU-residency stakes as bge-m3, so it's tracked
|
|
# here rather than invented as a separate registry class.
|
|
- id: tei-reranker
|
|
role: "cross-encoder reranker sidecar for Hindsight recall (interactive-critical)"
|
|
litellm_model_name: null # TEI-compatible /rerank API; not routed through LiteLLM
|
|
endpoints:
|
|
- name: tei-reranker
|
|
url: "http://tei-reranker:80" # host-published :8014
|
|
tier: small
|
|
context_tokens: null
|
|
tool_use_quality: "n/a"
|
|
lifecycle: always-on
|
|
quota: null
|
|
gpu_residency:
|
|
vram_mb: 1000
|
|
never_evict: true
|
|
co_residency_group: interactive-local
|
|
cost_class: free
|
|
metered: false
|
|
opt_in_required: false
|
|
|
|
# ── paid-fallback — optional, opt-in only ──────────────────────────────
|
|
# §3a: "Any paid deployment in the LiteLLM config must be explicitly
|
|
# enabled per agent via its virtual key; nothing routes to a metered
|
|
# model implicitly." routing.metered_opt_in below is the enforcement
|
|
# point: empty list = no caller has opted in = unreachable by resolve().
|
|
- id: paid-fallback
|
|
role: "optional metered fallback (e.g. Haiku) — disabled by default"
|
|
litellm_model_name: "judge" # litellm-config.yaml's existing entry (anthropic/claude-haiku-4-5-20251001)
|
|
endpoints: []
|
|
tier: large
|
|
context_tokens: 200000
|
|
tool_use_quality: high
|
|
lifecycle: cost-gated
|
|
quota:
|
|
probe_command: null # wire to a LiteLLM virtual-key budget probe (kb_worker.py BudgetProbe) once a caller opts in
|
|
windows: []
|
|
threshold_pct: null
|
|
gpu_residency: null
|
|
cost_class: metered
|
|
metered: true
|
|
opt_in_required: true
|
|
|
|
# ── GPU residency policy (§3b) ──────────────────────────────────────────
|
|
# "a local model's a(t) is not 1": a(t) = f(VRAM headroom). Never-evict
|
|
# models are excluded from eviction math entirely — their VRAM is a fixed
|
|
# reservation. Everything else in a co-residency group must fit in what's
|
|
# left. preload_check semantics documented here; implemented generically
|
|
# in model_registry.py so it reads this data instead of hardcoding numbers.
|
|
gpu_residency_policy:
|
|
card: "GTX 1070, 8192 MB (single GPU today; §8 — more GPUs become a placement problem, same policy, more slots)"
|
|
total_vram_mb: 8192
|
|
# Measured 2026-07-21: bge-m3 + gemma3:4b + tei-reranker ~= 6.2/8 GB.
|
|
# Loading something bigger than local-small's footprint on top evicts
|
|
# tei-reranker (LRU-ish ollama/torch behavior) -> silent recall-latency
|
|
# regression. This is the regression the pre-load check exists to catch.
|
|
measured_baseline_mb: 6200
|
|
never_evict_ids: [bge-m3, tei-reranker]
|
|
co_residency_groups:
|
|
interactive-local: [bge-m3, tei-reranker, local-small]
|
|
preload_check:
|
|
description: >
|
|
Before a worker pulls a candidate model onto the GPU it must pass
|
|
this check (see model_registry.py:preload_check): reserve every
|
|
never_evict model's vram_mb unconditionally, subtract whatever else
|
|
is currently resident, and require the candidate's own vram_mb to
|
|
fit in what's left of total_vram_mb. A failing check means "park,
|
|
don't load" — never silently evict a never-evict model.
|
|
|
|
# ── routing ───────────────────────────────────────────────────────────────
|
|
# Tier pools a caller can ask for without naming a model (design §2: "target
|
|
# = constraint-set"). metered_opt_in lists the virtual keys that have
|
|
# explicitly opted into paid-fallback; empty = no metered model is reachable
|
|
# by anyone, satisfying the "no metered API by default" acceptance bar.
|
|
routing:
|
|
tiers:
|
|
small: [local-small]
|
|
# paid-fallback listed as a large-tier candidate AFTER kimi so resolve()
|
|
# can fail over to it when kimi's a(t)=0 (quota parked) — but only for a
|
|
# caller that both passes allow_metered=True AND appears in
|
|
# metered_opt_in below. With metered_opt_in empty (the shipped default)
|
|
# resolve() skips it unconditionally, so it stays unreachable.
|
|
large: [kimi, paid-fallback]
|
|
metered_opt_in: [] # e.g. ["agent:torgash"] once a human explicitly opts a specific virtual key in
|