From b5aaceb65a356309336dc627693728a6ed1508d4 Mon Sep 17 00:00:00 2001 From: alvis Date: Tue, 21 Jul 2026 12:07:11 +0000 Subject: [PATCH] Add model registry: schema + populate (kb#133, A2A-1) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Per DESIGN-a2a-agents.md v2.1 §2-3b: models are the scarce queued resource, version-controlled here rather than hardcoded in callers. - model-registry.yaml: kimi (main reasoning, quota-gated), local-small (ollama/gemma3:4b, always-on cheap tier), bge-m3 (embedder + routing classifier, never-evict), tei-reranker (never-evict, interactive- critical), paid-fallback (metered, opt-in only, unreachable by default via empty routing.metered_opt_in). GPU residency policy carries the never-evict set, co-residency groups, and measured baseline (bge-m3+gemma3:4b+tei-reranker ~6.2/8GB on the GTX 1070). - model_registry.py: resolve(tier) picks an available model without the caller naming one, gated so a metered model is only reachable with both allow_metered=True and an opted-in virtual key; to_probe_config() bridges registry quota data into kb_worker.py's existing Probe classes (no duplicated probe logic); preload_check() expresses the §3b pre-load VRAM check purely from registry data. Gap noted for follow-up: bge-m3 has no litellm-config.yaml model_list entry yet (embedder there still points at ollama/nomic-embed-text on a different port) — out of scope here, registry documents it as-is. --- openai/model-registry.yaml | 192 +++++++++++++++++++++++ openai/model_registry.py | 301 +++++++++++++++++++++++++++++++++++++ 2 files changed, 493 insertions(+) create mode 100644 openai/model-registry.yaml create mode 100755 openai/model_registry.py diff --git a/openai/model-registry.yaml b/openai/model-registry.yaml new file mode 100644 index 0000000..747e8e9 --- /dev/null +++ b/openai/model-registry.yaml @@ -0,0 +1,192 @@ +# Model registry — models are the scarce queued resource. +# +# Per DESIGN-a2a-agents.md v2.1 §2-3b (commit df2071d5), kanboard task #133 +# (A2A-1). Version-controlled here; the "model plane" (§3) and the fabric's +# workers/routers read this data — they do not duplicate it. Lifecycle a(t) +# *probe mechanics* (QuotaProbe, GPUResidencyProbe, ...) live in +# kanboard/bin/kb_worker.py; this registry supplies the *parameters* those +# probes consume (commands, fields, thresholds, VRAM footprints). +# +# Scope constraint (alvis, §3a): NO METERED API BY DEFAULT. The workflow is +# Claude Code (a flat-subscription runtime -> agent registry #134, not here) +# + the Kimi wrapper + a local GPU embedder + a small weak local model. The +# governor arbitrates quota and GPU, not money. Any metered model below is +# `metered: true, opt_in_required: true` and carries no default route to it +# (see routing.metered_opt_in: [] at the bottom — empty means unreachable). +# +# Read with model_registry.py (same directory): resolve(), preload_check(). + +schema_version: 1 + +models: + # ── kimi — main reasoning ────────────────────────────────────────────── + # Flat Moonshot/Kimi subscription via `kimi login`, wrapped by two + # independent Kimi-CLI containers (own OAuth creds volume each). Not + # behind LiteLLM today — callers hit the wrapper HTTP endpoints directly. + - id: kimi + role: "main reasoning (adolf-llm / hindsight-llm Kimi-CLI wrappers)" + litellm_model_name: null + endpoints: + - name: adolf-llm + purpose: "Adolf's conversational backbone" + url: "http://adolf-llm:8010" + usage_url: "http://localhost:8010/usage" + - name: hindsight-llm + purpose: "Hindsight's structured-extraction LLM (HINDSIGHT_API_LLM_MODEL)" + url: "http://hindsight-llm:8012/v1" + model_name: "openai/hindsight-llm" + tier: large + context_tokens: 200000 # Moonshot Kimi K2 context window; re-verify if the CLI's pinned model changes + tool_use_quality: high + lifecycle: quota-gated + quota: + probe_command: ["kimi-usage", "--compact"] + windows: + - name: 5h + field: "window_5h.pct" # adolf-llm server.js normalizeKimiUsage() field name + approx_limit: "~60 msgs/5h" + - name: weekly + field: "weekly.pct" + approx_limit: "~300 msgs/wk" + threshold_pct: 95 + gpu_residency: null + cost_class: subscription # flat-rate, not metered — quota is the constraint, not spend + metered: false + opt_in_required: false + + # ── local-small — the cheap tier ─────────────────────────────────────── + # ollama/gemma3:4b on the GPU ollama instance. Already the live model for + # Hindsight consolidation/reflect (HINDSIGHT_API_CONSOLIDATION_LLM_MODEL / + # HINDSIGHT_API_REFLECT_LLM_MODEL, kb#88) and exposed via LiteLLM. + - id: local-small + role: "cheap tier — ollama small/weak local model (background extraction, consolidation, reflect)" + litellm_model_name: "ollama/gemma3:4b" # openai/litellm-config.yaml model_list entry + endpoints: + - name: ollama-direct + url: "http://host.docker.internal:11436" + - name: via-litellm + url: "http://litellm:4000/v1" + tier: small + context_tokens: 8192 # gemma3:4b default ctx; re-verify with `ollama show gemma3:4b` if raised + tool_use_quality: low + lifecycle: always-on + quota: null + gpu_residency: + vram_mb: 4000 # approx measured footprint, within the shared 8GB card (see gpu_residency_policy below) + never_evict: false # evictable — a bigger model may push it out; that's a silent regression to catch, not prevent here + co_residency_group: interactive-local + cost_class: free + metered: false + opt_in_required: false + + # ── bge-m3 — embedder + routing classifier ───────────────────────────── + # Never-evict: it's both Hindsight's recall embedder AND (design §3a) the + # embedding model LiteLLM Auto Router's semantic-router classifier will + # use for tier/complexity routing — losing it degrades both recall AND + # routing at once. + - id: bge-m3 + role: "embedder — also the routing classifier (§3a, LiteLLM Auto Router / semantic-router)" + litellm_model_name: null # NOT YET wired into litellm-config.yaml — gap, see model_registry.py module docstring + endpoints: + - name: ollama-direct + url: "http://host.docker.internal:11436" + openai_compatible_path: "/v1/embeddings" + tier: small + context_tokens: 8192 + tool_use_quality: "n/a" # embedder, not a chat/tool-use model + lifecycle: always-on + quota: null + gpu_residency: + vram_mb: 1200 + never_evict: true + co_residency_group: interactive-local + cost_class: free + metered: false + opt_in_required: false + + # ── tei-reranker — interactive-critical, never-evict ─────────────────── + # Not an LLM (cross-encoder rerank sidecar for Hindsight recall, kb#87) + # but carries the same GPU-residency stakes as bge-m3, so it's tracked + # here rather than invented as a separate registry class. + - id: tei-reranker + role: "cross-encoder reranker sidecar for Hindsight recall (interactive-critical)" + litellm_model_name: null # TEI-compatible /rerank API; not routed through LiteLLM + endpoints: + - name: tei-reranker + url: "http://tei-reranker:80" # host-published :8014 + tier: small + context_tokens: null + tool_use_quality: "n/a" + lifecycle: always-on + quota: null + gpu_residency: + vram_mb: 1000 + never_evict: true + co_residency_group: interactive-local + cost_class: free + metered: false + opt_in_required: false + + # ── paid-fallback — optional, opt-in only ────────────────────────────── + # §3a: "Any paid deployment in the LiteLLM config must be explicitly + # enabled per agent via its virtual key; nothing routes to a metered + # model implicitly." routing.metered_opt_in below is the enforcement + # point: empty list = no caller has opted in = unreachable by resolve(). + - id: paid-fallback + role: "optional metered fallback (e.g. Haiku) — disabled by default" + litellm_model_name: "judge" # litellm-config.yaml's existing entry (anthropic/claude-haiku-4-5-20251001) + endpoints: [] + tier: large + context_tokens: 200000 + tool_use_quality: high + lifecycle: cost-gated + quota: + probe_command: null # wire to a LiteLLM virtual-key budget probe (kb_worker.py BudgetProbe) once a caller opts in + windows: [] + threshold_pct: null + gpu_residency: null + cost_class: metered + metered: true + opt_in_required: true + +# ── GPU residency policy (§3b) ────────────────────────────────────────── +# "a local model's a(t) is not 1": a(t) = f(VRAM headroom). Never-evict +# models are excluded from eviction math entirely — their VRAM is a fixed +# reservation. Everything else in a co-residency group must fit in what's +# left. preload_check semantics documented here; implemented generically +# in model_registry.py so it reads this data instead of hardcoding numbers. +gpu_residency_policy: + card: "GTX 1070, 8192 MB (single GPU today; §8 — more GPUs become a placement problem, same policy, more slots)" + total_vram_mb: 8192 + # Measured 2026-07-21: bge-m3 + gemma3:4b + tei-reranker ~= 6.2/8 GB. + # Loading something bigger than local-small's footprint on top evicts + # tei-reranker (LRU-ish ollama/torch behavior) -> silent recall-latency + # regression. This is the regression the pre-load check exists to catch. + measured_baseline_mb: 6200 + never_evict_ids: [bge-m3, tei-reranker] + co_residency_groups: + interactive-local: [bge-m3, tei-reranker, local-small] + preload_check: + description: > + Before a worker pulls a candidate model onto the GPU it must pass + this check (see model_registry.py:preload_check): reserve every + never_evict model's vram_mb unconditionally, subtract whatever else + is currently resident, and require the candidate's own vram_mb to + fit in what's left of total_vram_mb. A failing check means "park, + don't load" — never silently evict a never-evict model. + +# ── routing ─────────────────────────────────────────────────────────────── +# Tier pools a caller can ask for without naming a model (design §2: "target +# = constraint-set"). metered_opt_in lists the virtual keys that have +# explicitly opted into paid-fallback; empty = no metered model is reachable +# by anyone, satisfying the "no metered API by default" acceptance bar. +routing: + tiers: + small: [local-small] + # paid-fallback listed as a large-tier candidate AFTER kimi so resolve() + # can fail over to it when kimi's a(t)=0 (quota parked) — but only for a + # caller that both passes allow_metered=True AND appears in + # metered_opt_in below. With metered_opt_in empty (the shipped default) + # resolve() skips it unconditionally, so it stays unreachable. + large: [kimi, paid-fallback] + metered_opt_in: [] # e.g. ["agent:torgash"] once a human explicitly opts a specific virtual key in diff --git a/openai/model_registry.py b/openai/model_registry.py new file mode 100755 index 0000000..80aa9f5 --- /dev/null +++ b/openai/model_registry.py @@ -0,0 +1,301 @@ +#!/usr/bin/env python3 +"""model_registry — reads model-registry.yaml (kb#133, A2A-1). + +Per DESIGN-a2a-agents.md v2.1 §2-3b: the model registry is data, not logic. +a(t) probe *mechanics* (QuotaProbe, GPUResidencyProbe, ...) already live in +kanboard/bin/kb_worker.py — this module does not reimplement them. It gives +callers two things instead: + + * resolve(tier) — "an available model for tier X" without the + caller naming a model. Structural availability + (lifecycle, metered opt-in) is decided here from + registry data; live a(t) truth (is the quota + window open right now, is the GPU actually free) + is decided by an optional `probe_check` callback + the caller supplies (e.g. wired to kb_worker's + Probe classes via to_probe_config()). + * preload_check(...) — the §3b GPU pre-load check, expressed purely from + registry numbers (never-evict reservations + + candidate footprint) plus a headroom figure the + caller supplies. It does not shell nvidia-smi + itself — kb_worker.GPUResidencyProbe (or + `nvidia-smi` directly) is the live-read path; + this stays pure/testable. + +Usage (library): + from model_registry import load_registry, resolve, to_probe_config, preload_check + reg = load_registry() + model = resolve(reg, tier="large") # -> the "kimi" entry + cfg = to_probe_config(reg, "kimi") # -> kb_worker probe config dict + ok, reason = preload_check(reg, "local-small", headroom_mb=1900) + +Usage (CLI, for manual verification): + ./model_registry.py resolve --tier large + ./model_registry.py resolve --tier large --allow-metered --opted-in agent:torgash + ./model_registry.py probe-config --id kimi + ./model_registry.py preload-check --id local-small --headroom-mb 1900 + ./model_registry.py preload-check --id local-small --live # shells nvidia-smi + ./model_registry.py list +""" + +import argparse +import json +import os +import subprocess +import sys + +import yaml + +HERE = os.path.dirname(os.path.abspath(__file__)) +DEFAULT_REGISTRY_PATH = os.path.join(HERE, "model-registry.yaml") + + +class RegistryError(Exception): + pass + + +def load_registry(path=None): + """Load and lightly validate model-registry.yaml.""" + path = path or DEFAULT_REGISTRY_PATH + with open(path) as f: + reg = yaml.safe_load(f) + if not reg or "models" not in reg: + raise RegistryError(f"{path}: missing top-level 'models' list") + ids = [m["id"] for m in reg["models"]] + if len(ids) != len(set(ids)): + raise RegistryError(f"{path}: duplicate model ids in {ids}") + return reg + + +def get_model(registry, model_id): + for m in registry["models"]: + if m["id"] == model_id: + return m + raise RegistryError(f"unknown model id: {model_id!r}") + + +# --------------------------------------------------------------------------- +# resolve — "an available model for tier X" without the caller naming one. +# --------------------------------------------------------------------------- + +def resolve(registry, tier, allow_metered=False, opted_in_key=None, probe_check=None): + """Return the first model in `tier`'s pool that is structurally usable, + and (if probe_check is given) currently available. + + Structural filter (from registry data alone): + - candidate must be listed under routing.tiers[tier] + - a metered model is only a candidate at all when the CALLER passes + allow_metered=True AND opted_in_key appears in routing.metered_opt_in + (§3a: "no metered API by default" — an empty metered_opt_in list, the + shipped default, makes every metered model structurally unreachable + regardless of allow_metered). + + Live filter (optional): probe_check(model_dict) -> bool. Wire this to + kb_worker's Probe.available() (via to_probe_config below) when the + caller wants real a(t) truth instead of just structural eligibility. + """ + pools = registry.get("routing", {}).get("tiers", {}) + if tier not in pools: + raise RegistryError(f"unknown tier: {tier!r} (have: {sorted(pools)})") + opted_in = set(registry.get("routing", {}).get("metered_opt_in", []) or []) + + candidates = [] + for model_id in pools[tier]: + m = get_model(registry, model_id) + if m.get("metered"): + if not allow_metered: + continue + if not m.get("opt_in_required", True): + # Registry says this metered model doesn't need opt-in — treat + # as a data error rather than silently routing to it. + raise RegistryError( + f"model {model_id!r} is metered but opt_in_required=false; " + "fix the registry entry, this helper will not assume implicit access" + ) + if opted_in_key is None or opted_in_key not in opted_in: + continue + candidates.append(m) + + for m in candidates: + if probe_check is None or probe_check(m): + return m + + raise RegistryError( + f"no available model for tier={tier!r} " + f"(allow_metered={allow_metered}, opted_in_key={opted_in_key!r}); " + f"checked candidates: {[m['id'] for m in candidates] or pools[tier]}" + ) + + +# --------------------------------------------------------------------------- +# to_probe_config — bridges registry quota data into kb_worker's probe cfg +# shape (kanboard/bin/kb_worker.py PROBE_BUILDERS), so probes read registry +# numbers instead of the registry re-implementing probe logic. +# --------------------------------------------------------------------------- + +def to_probe_config(registry, model_id): + """Return a dict matching kb_worker.py's `build_probe(cfg)` input shape + for `model_id`'s lifecycle. Raises if the model has no probe-relevant + lifecycle (e.g. cost-gated with no probe_command wired yet).""" + m = get_model(registry, model_id) + lifecycle = m["lifecycle"] + + if lifecycle == "always-on": + return {"type": "always_on"} + + if lifecycle == "quota-gated": + q = m.get("quota") or {} + windows = q.get("windows") or [] + if not windows: + raise RegistryError(f"{model_id}: quota-gated but no quota.windows configured") + # kb_worker's QuotaProbe checks one field; the tightest (first-to-hit) + # window in practice is the short one — default to the first entry, + # callers needing multi-window gating build one probe per window. + window = windows[0] + return { + "type": "quota", + "command": q["probe_command"], + "field": window["field"], + "threshold_pct": q.get("threshold_pct", 95), + } + + if lifecycle == "cost-gated": + q = m.get("quota") or {} + if not q.get("probe_command"): + raise RegistryError( + f"{model_id}: cost-gated but no budget probe wired yet " + "(opt-in path incomplete — see registry comment)" + ) + return { + "type": "budget", + "command": q["probe_command"], + "field": q["field"], + "limit": q["limit"], + } + + if lifecycle == "on-demand": + ep = (m.get("endpoints") or [{}])[0] + if not ep.get("health_url"): + raise RegistryError(f"{model_id}: on-demand but no endpoint.health_url configured") + return {"type": "on_demand", "url": ep["health_url"]} + + raise RegistryError(f"{model_id}: unknown lifecycle {lifecycle!r}") + + +# --------------------------------------------------------------------------- +# preload_check — §3b GPU pre-load check, pure registry-data math. The live +# VRAM headroom READ is the caller's job (kb_worker.GPUResidencyProbe or +# nvidia-smi directly) — see live_headroom_mb() below for a thin convenience +# wrapper used only by this module's own CLI, not by the check itself. +# --------------------------------------------------------------------------- + +def preload_check(registry, candidate_id, headroom_mb, resident_ids=None): + """Would loading `candidate_id` fit, given `headroom_mb` free VRAM right + now (as reported by a live probe) and `resident_ids` already loaded? + + Never-evict models are never subtracted from headroom by a caller in the + first place (their footprint is a standing reservation baked into any + correct live headroom read) — this function just re-asserts that policy + from registry data: if `candidate_id` itself is never-evict, it always + passes (it's not something a worker "loads speculatively" and might be + told to skip); anything else must fit in the reported headroom. + """ + policy = registry.get("gpu_residency_policy") or {} + m = get_model(registry, candidate_id) + gr = m.get("gpu_residency") + if not gr: + return True, f"{candidate_id} has no gpu_residency entry (not a GPU-resident model)" + + if gr.get("never_evict"): + return True, f"{candidate_id} is in the never-evict set — always resident by policy" + + required_mb = gr["vram_mb"] + never_evict_ids = set(policy.get("never_evict_ids", [])) + resident_ids = set(resident_ids or []) + # Sanity: if the live headroom read already accounts for never-evict + # reservations (the expected contract — see kb_worker.GPUResidencyProbe's + # never_evict_reserved_mb param), this is just a straight comparison. + # If a caller passes raw total-minus-used instead, warn via the reason + # string rather than silently under/over-reserving. + reserved_hint = sum( + get_model(registry, mid)["gpu_residency"]["vram_mb"] + for mid in never_evict_ids + if mid not in resident_ids # already counted as "used" if resident_ids says so + ) + ok = headroom_mb >= required_mb + reason = ( + f"headroom={headroom_mb}MB required={required_mb}MB " + f"(never-evict reserve expected already netted out by the caller's probe: " + f"~{reserved_hint}MB across {sorted(never_evict_ids)})" + ) + return ok, reason + + +def live_headroom_mb(never_evict_reserved_mb=0): + """Convenience for manual CLI checks only — NOT used by preload_check() + itself. Shells nvidia-smi the same way kb_worker.GPUResidencyProbe does.""" + out = subprocess.run( + ["nvidia-smi", "--query-gpu=memory.used,memory.total", "--format=csv,noheader,nounits"], + capture_output=True, text=True, timeout=5, check=True, + ).stdout.strip().splitlines()[0] + used, total = (int(x) for x in out.split(",")) + return total - used - never_evict_reserved_mb + + +# --------------------------------------------------------------------------- +# CLI — manual verification only, not part of the library contract. +# --------------------------------------------------------------------------- + +def main(): + ap = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + ap.add_argument("--registry", default=None, help="path to model-registry.yaml (default: sibling file)") + sub = ap.add_subparsers(dest="cmd", required=True) + + p = sub.add_parser("resolve") + p.add_argument("--tier", required=True) + p.add_argument("--allow-metered", action="store_true") + p.add_argument("--opted-in", default=None, help="virtual key claimed to be opted in") + + p = sub.add_parser("probe-config") + p.add_argument("--id", required=True) + + p = sub.add_parser("preload-check") + p.add_argument("--id", required=True) + p.add_argument("--headroom-mb", type=float, default=None) + p.add_argument("--live", action="store_true", help="read live headroom via nvidia-smi instead of --headroom-mb") + p.add_argument("--resident", action="append", default=[], help="repeatable: id already resident") + + sub.add_parser("list") + + args = ap.parse_args() + reg = load_registry(args.registry) + + try: + if args.cmd == "resolve": + m = resolve(reg, args.tier, allow_metered=args.allow_metered, opted_in_key=args.opted_in) + print(json.dumps(m, indent=2)) + elif args.cmd == "probe-config": + print(json.dumps(to_probe_config(reg, args.id), indent=2)) + elif args.cmd == "preload-check": + headroom = args.headroom_mb + if args.live: + policy = reg.get("gpu_residency_policy") or {} + reserved = sum(get_model(reg, mid)["gpu_residency"]["vram_mb"] + for mid in policy.get("never_evict_ids", [])) + headroom = live_headroom_mb(never_evict_reserved_mb=reserved) + if headroom is None: + raise RegistryError("preload-check needs --headroom-mb or --live") + ok, reason = preload_check(reg, args.id, headroom, resident_ids=args.resident) + print(json.dumps({"ok": ok, "reason": reason})) + sys.exit(0 if ok else 1) + elif args.cmd == "list": + for m in reg["models"]: + print(f"{m['id']:16} tier={m['tier']:6} lifecycle={m['lifecycle']:13} " + f"metered={m['metered']} role={m['role']}") + except RegistryError as exc: + print(f"error: {exc}", file=sys.stderr) + sys.exit(2) + + +if __name__ == "__main__": + main()