From 713cb8f80f6c397699d9e06a01fc44d7194f6d95 Mon Sep 17 00:00:00 2001 From: jenkins Date: Mon, 28 Sep 2026 17:10:12 -0500 Subject: [PATCH] hermes: serve authenticated local inference on the LAN --- scripts/ops/hermes_lan_generate.sh | 40 +++++ services/hermes/NOTES.md | 27 ++++ services/hermes/kustomization.yaml | 1 + services/hermes/model-gate-configmap.yaml | 142 ++++++++++++++++++ services/hermes/model-gate-deployment.yaml | 32 +++- services/hermes/model-gate-lan-ingress.yaml | 47 ++++++ services/hermes/networkpolicy.yaml | 64 +++++++- ...ault_hermes_model_gate_lan_token_ensure.sh | 0 testing/tests/test_hermes_model_gate_lan.py | 124 +++++++++++++++ 9 files changed, 475 insertions(+), 2 deletions(-) create mode 100755 scripts/ops/hermes_lan_generate.sh create mode 100644 services/hermes/model-gate-lan-ingress.yaml mode change 100644 => 100755 services/vault/scripts/vault_hermes_model_gate_lan_token_ensure.sh create mode 100644 testing/tests/test_hermes_model_gate_lan.py diff --git a/scripts/ops/hermes_lan_generate.sh b/scripts/ops/hermes_lan_generate.sh new file mode 100755 index 00000000..409d1310 --- /dev/null +++ b/scripts/ops/hermes_lan_generate.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Call the private gateway with normal TLS verification and no public DNS hop. +set -euo pipefail + +if [[ ${1:-} == --help ]]; then + cat <<'USAGE' +Usage: HERMES_LAN_TOKEN_FILE=/path/to/token hermes_lan_generate.sh < prompt.txt +Without a token file, reads kv/atlas/hermes/model-gate-lan-api through Vault CLI. +Optional: HERMES_LAN_ADDRESS (192.168.22.50), HERMES_LAN_MAX_TOKENS (256). +USAGE + exit 0 +fi + +umask 077 +scratch=$(mktemp -d) +trap 'rm -rf -- "$scratch"' EXIT +if [[ -n ${HERMES_LAN_TOKEN_FILE:-} ]]; then + token=$(cat -- "$HERMES_LAN_TOKEN_FILE") +else + token=$(vault kv get -field=token kv/atlas/hermes/model-gate-lan-api) +fi +if [[ ! $token =~ ^[0-9a-f]{64}$ ]]; then + printf 'Invalid LAN gateway token\n' >&2 + exit 1 +fi +# Keep the bearer out of process arguments and shell tracing in the client. +printf 'header = "Authorization: Bearer %s"\n' "$token" > "$scratch/curl.conf" +unset token +python3 -c ' +import json, os, sys +json.dump({"model": "qwen2.5:14b-instruct-q4_0", "prompt": sys.stdin.read(), + "stream": False, "options": {"num_predict": int(os.getenv("HERMES_LAN_MAX_TOKENS", "256"))}}, sys.stdout) +' > "$scratch/request.json" +curl --fail-with-body --silent --show-error --connect-timeout 10 --max-time 310 \ + --noproxy worker.bstein.dev \ + --resolve "worker.bstein.dev:443:${HERMES_LAN_ADDRESS:-192.168.22.50}" \ + --config "$scratch/curl.conf" \ + --header 'Content-Type: application/json' \ + --data-binary "@$scratch/request.json" \ + https://worker.bstein.dev/local-model/api/generate diff --git a/services/hermes/NOTES.md b/services/hermes/NOTES.md index 1044a950..8a277078 100644 --- a/services/hermes/NOTES.md +++ b/services/hermes/NOTES.md @@ -1,5 +1,32 @@ # Hermes on Atlas: operator guide +## LAN model gateway + +The private base URL is `https://worker.bstein.dev/local-model`, reached through +`192.168.22.50:443` on the Atlas LAN. This separate Traefik LoadBalancer preserves +client IPs; the shared public LoadBalancer masks them. The service and middleware +allow only `192.168.22.0/24`, and the gateway also requires a bearer token from +Vault at `kv/atlas/hermes/model-gate-lan-api`, field `token`. + +Use `scripts/ops/hermes_lan_generate.sh < prompt.txt` from the LAN with a logged-in +Vault CLI, or set `HERMES_LAN_TOKEN_FILE` to a private file containing that token. +The helper connects directly to the LAN IP while verifying the existing +`worker.bstein.dev` TLS certificate. It needs no DNS override. Public DNS continues +to serve the worker dashboard; it does not select the LAN gateway automatically. + +`GET /healthz` and `POST /api/generate` are the only gateway operations. Generate +accepts `model: qwen2.5:14b-instruct-q4_0`, a string `prompt`, `stream: false`, and +optional `num_predict`, `temperature`, `top_p`, and `seed` inside `options`. +Requests are capped at 128 KiB and 2048 output tokens. Concurrent LAN generations +receive 429 and can retry. Requests use only the local Ollama service; no hosted +fallback, model management, tools, sessions, image, or voice APIs are exposed. + +A missing/wrong token returns 401, an unavailable credential or local model +returns 503, and a non-LAN source returns 403. Flux bootstraps the Vault roles +and creates the token only if absent. After deliberate token rotation, roll the +model-gate through its Flux-tracked deployment revision to reload the injected +credential. + This is the mental model and demonstration script for the operator instance at `triage.bstein.dev`. Read it once, then prove each section in the live UI. The consumer instance at `chat.bstein.dev` is intentionally separate and is not the diff --git a/services/hermes/kustomization.yaml b/services/hermes/kustomization.yaml index 4e1e6149..03cf7cb4 100644 --- a/services/hermes/kustomization.yaml +++ b/services/hermes/kustomization.yaml @@ -42,6 +42,7 @@ resources: - oauth2-proxy.yaml - agent-certificate.yaml - agent-ingress.yaml + - model-gate-lan-ingress.yaml - execution-worker-rbac.yaml - hux-evidence-rbac.yaml - chat-cluster-read-rbac.yaml diff --git a/services/hermes/model-gate-configmap.yaml b/services/hermes/model-gate-configmap.yaml index bf4efd0d..0aea42a9 100644 --- a/services/hermes/model-gate-configmap.yaml +++ b/services/hermes/model-gate-configmap.yaml @@ -10,6 +10,8 @@ data: """Normalize Jetson text requests and coordinate the titan-24 image handoff.""" from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + import hmac + import ipaddress import json import os from pathlib import Path @@ -22,12 +24,18 @@ data: LISTEN_HOST = os.environ.get("LISTEN_HOST", "0.0.0.0") LISTEN_PORT = int(os.environ.get("LISTEN_PORT", "8080")) + LAN_LISTEN_PORT = int(os.environ.get("LAN_LISTEN_PORT", "8082")) HANDOFF_PORT = int(os.environ.get("HANDOFF_PORT", "8081")) UPSTREAM_URL = os.environ.get("UPSTREAM_URL", "http://ollama.ai.svc.cluster.local:11434").rstrip("/") LOCAL_IMAGE_URL = os.environ.get("LOCAL_IMAGE_URL", "http://hermes-local-image.hermes.svc.cluster.local:9004").rstrip("/") HANDOFF_TIMEOUT_SEC = float(os.environ.get("HANDOFF_TIMEOUT_SEC", "1200")) IMAGE_NAMESPACE = os.environ.get("IMAGE_NAMESPACE", "hermes") IMAGE_DEPLOYMENT = os.environ.get("IMAGE_DEPLOYMENT", "hermes-local-image") + LAN_API_TOKEN_FILE = Path(os.environ.get("LAN_API_TOKEN_FILE", "/vault/secrets/lan-api-token")) + LAN_MODEL = "qwen2.5:14b-instruct-q4_0" + LAN_NETWORK = ipaddress.ip_network("192.168.22.0/24") + LAN_MAX_BODY = 131072 + _lan_inference = threading.BoundedSemaphore(1) KUBE_HOST = os.environ.get("KUBERNETES_SERVICE_HOST", "kubernetes.default.svc") KUBE_PORT = os.environ.get("KUBERNETES_SERVICE_PORT_HTTPS", "443") KUBE_TOKEN = Path("/var/run/secrets/kubernetes.io/serviceaccount/token") @@ -109,6 +117,138 @@ data: return False, last_error + def _normalize_lan_generate(body: bytes | None) -> bytes: + """Accept only the one supported stateless local generation operation.""" + + if not body: + raise ValueError("request body is required") + try: + payload = json.loads(body) + except (TypeError, ValueError, json.JSONDecodeError) as exc: + raise ValueError("invalid JSON") from exc + if not isinstance(payload, dict): + raise ValueError("JSON object is required") + if payload.get("model") != LAN_MODEL: + raise ValueError(f"model must be {LAN_MODEL}") + if payload.get("stream") is not False: + raise ValueError("stream must be false") + if not isinstance(payload.get("prompt"), str): + raise ValueError("prompt must be a string") + # Only stateless text and bounded sampling settings cross this boundary. + invalid = sorted(set(payload) - {"model", "prompt", "stream", "options"}) + if invalid: + raise ValueError(f"unsupported fields: {', '.join(invalid)}") + options = payload.get("options", {}) + if not isinstance(options, dict) or set(options) - {"num_predict", "temperature", "top_p", "seed"}: + raise ValueError("unsupported options") + count = options.get("num_predict", 256) + if type(count) is not int or not 1 <= count <= 2048: + raise ValueError("num_predict must be between 1 and 2048") + for key, upper in (("temperature", 2), ("top_p", 1)): + value = options.get(key, 1) + if type(value) not in (int, float) or not 0 <= value <= upper: + raise ValueError(f"invalid {key}") + if "seed" in options and (type(options["seed"]) is not int or not -1 <= options["seed"] <= 2147483647): + raise ValueError("invalid seed") + payload["options"] = {**options, "num_predict": count} + return json.dumps(payload, separators=(",", ":")).encode("utf-8") + + + def _is_lan_client(client_ip: str) -> bool: + """Validate the address appended by the trusted ingress proxy.""" + try: + return ipaddress.ip_address(client_ip) in LAN_NETWORK + except ValueError: + return False + + + class LanHandler(BaseHTTPRequestHandler): + """Narrow authenticated LAN API: health plus one pinned Ollama operation.""" + + protocol_version = "HTTP/1.1" + + def setup(self) -> None: + """Bound the time a client can hold a request body open.""" + super().setup() + self.connection.settimeout(10) + + def _json(self, status: int, payload: dict) -> None: + """End each request so rejected bodies cannot become another request.""" + body = json.dumps(payload, separators=(",", ":")).encode("utf-8") + self.send_response(status) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.send_header("Cache-Control", "no-store") + self.send_header("Connection", "close") + self.close_connection = True + self.end_headers() + self.wfile.write(body) + + def _authorized(self) -> bool: + """Require both the ingress-observed LAN address and the Vault token.""" + forwarded_for = self.headers.get("X-Forwarded-For", "").split(",")[-1].strip() + if not _is_lan_client(forwarded_for): + self._json(403, {"error": "LAN source required"}) + return False + try: + expected = LAN_API_TOKEN_FILE.read_text(encoding="utf-8").strip() + except OSError: + self._json(503, {"error": "LAN API credential unavailable"}) + return False + supplied = self.headers.get("Authorization", "") + if not expected or not supplied.startswith("Bearer ") or not hmac.compare_digest(supplied[7:].encode(), expected.encode()): + self._json(401, {"error": "Bearer authentication required"}) + return False + return True + + def do_GET(self) -> None: + """Expose authenticated readiness without model or session metadata.""" + if self.path != "/healthz": + self._json(404, {"error": "not found"}) + return + if self._authorized(): + self._json(200, {"status": "ok", "model": LAN_MODEL}) + + def do_POST(self) -> None: + """Validate and serialize bounded inference against local Ollama only.""" + if self.path != "/api/generate": + self._json(404, {"error": "not found"}) + return + if not self._authorized(): + return + try: + if self.headers.get("Transfer-Encoding"): + raise ValueError("Transfer-Encoding is unsupported") + length = int(self.headers.get("Content-Length", "0")) + if not 0 < length <= LAN_MAX_BODY: + self._json(413, {"error": "body must be between 1 and 131072 bytes"}) + return + body = _normalize_lan_generate(self.rfile.read(length)) + except ValueError as exc: + self._json(400, {"error": str(exc)}) + return + if not _lan_inference.acquire(blocking=False): + self._json(429, {"error": "local model is busy; retry later"}) + return + request = Request(f"{UPSTREAM_URL}/api/generate", data=body, headers={"Content-Type": "application/json"}, method="POST") + try: + try: + with urlopen(request, timeout=300) as response: + payload = json.load(response) + self._json(response.status, payload) + except HTTPError: + self._json(502, {"error": "local Ollama rejected the request"}) + except (TimeoutError, URLError, ValueError): + # This endpoint has one upstream only; never route externally. + self._json(503, {"error": "local Ollama unavailable"}) + finally: + _lan_inference.release() + + def log_message(self, format_string: str, *args) -> None: + # Deliberately log only transport metadata. Never log prompts or responses. + print(f"model-gate-lan {self.address_string()} {format_string % args}", flush=True) + + class Handler(BaseHTTPRequestHandler): """Proxy the non-preemptible titan-20 text fallback.""" @@ -234,4 +374,6 @@ data: if __name__ == "__main__": handoff = ThreadingHTTPServer((LISTEN_HOST, HANDOFF_PORT), HandoffHandler) threading.Thread(target=handoff.serve_forever, name="gpu-handoff", daemon=True).start() + lan_api = ThreadingHTTPServer((LISTEN_HOST, LAN_LISTEN_PORT), LanHandler) + threading.Thread(target=lan_api.serve_forever, name="lan-model-api", daemon=True).start() ThreadingHTTPServer((LISTEN_HOST, LISTEN_PORT), Handler).serve_forever() diff --git a/services/hermes/model-gate-deployment.yaml b/services/hermes/model-gate-deployment.yaml index cc747e07..ca4e640a 100644 --- a/services/hermes/model-gate-deployment.yaml +++ b/services/hermes/model-gate-deployment.yaml @@ -15,7 +15,19 @@ spec: template: metadata: annotations: - ai.bstein.dev/config-rev: "20260811-switchyard-model-fields" + ai.bstein.dev/config-rev: "20260928-lan-model-api-v1" + vault.hashicorp.com/agent-inject: "true" + vault.hashicorp.com/agent-pre-populate-only: "true" + vault.hashicorp.com/agent-init-first: "true" + vault.hashicorp.com/agent-run-as-user: "65532" + vault.hashicorp.com/agent-run-as-group: "65532" + vault.hashicorp.com/role: hermes-model-gate + vault.hashicorp.com/agent-inject-secret-lan-api-token: kv/data/atlas/hermes/model-gate-lan-api + vault.hashicorp.com/agent-inject-template-lan-api-token: | + {{- with secret "kv/data/atlas/hermes/model-gate-lan-api" -}} + {{ .Data.data.token }} + {{- end -}} + vault.hashicorp.com/agent-inject-perms-lan-api-token: "0400" labels: app: hermes-model-gate spec: @@ -64,6 +76,8 @@ spec: containerPort: 8080 - name: handoff containerPort: 8081 + - name: lan-api + containerPort: 8082 env: - name: UPSTREAM_URL value: http://ollama.ai.svc.cluster.local:11434 @@ -134,6 +148,22 @@ spec: --- apiVersion: v1 kind: Service +metadata: + name: hermes-model-gate-lan-api + namespace: hermes + labels: + app: hermes-model-gate +spec: + type: ClusterIP + selector: + app: hermes-model-gate + ports: + - name: http + port: 8082 + targetPort: lan-api +--- +apiVersion: v1 +kind: Service metadata: name: hermes-gpu-handoff namespace: hermes diff --git a/services/hermes/model-gate-lan-ingress.yaml b/services/hermes/model-gate-lan-ingress.yaml new file mode 100644 index 00000000..6c2e8af1 --- /dev/null +++ b/services/hermes/model-gate-lan-ingress.yaml @@ -0,0 +1,47 @@ +# services/hermes/model-gate-lan-ingress.yaml +apiVersion: traefik.io/v1alpha1 +kind: Middleware +metadata: + name: hermes-model-gate-lan-allowlist + namespace: hermes +spec: + ipAllowList: + sourceRange: + - 192.168.22.0/24 +--- +apiVersion: traefik.io/v1alpha1 +kind: Middleware +metadata: + name: hermes-model-gate-lan-prefix + namespace: hermes +spec: + stripPrefix: + prefixes: + - /local-model +--- +apiVersion: networking.k8s.io/v1 +kind: Ingress +metadata: + name: hermes-model-gate-lan + namespace: hermes + annotations: + traefik.ingress.kubernetes.io/router.entrypoints: websecure + traefik.ingress.kubernetes.io/router.middlewares: hermes-hermes-model-gate-lan-allowlist@kubernetescrd,hermes-hermes-model-gate-lan-prefix@kubernetescrd + traefik.ingress.kubernetes.io/router.tls: "true" +spec: + ingressClassName: traefik + tls: + - hosts: + - worker.bstein.dev + secretName: hermes-sites-tls + rules: + - host: worker.bstein.dev + http: + paths: + - path: /local-model + pathType: Prefix + backend: + service: + name: hermes-model-gate-lan-api + port: + number: 8082 diff --git a/services/hermes/networkpolicy.yaml b/services/hermes/networkpolicy.yaml index dbddb0fb..1817dafb 100644 --- a/services/hermes/networkpolicy.yaml +++ b/services/hermes/networkpolicy.yaml @@ -29,7 +29,7 @@ spec: podSelector: matchLabels: app: hermes-model-gate - policyTypes: [Ingress] + policyTypes: [Ingress, Egress] ingress: - from: - podSelector: @@ -48,6 +48,68 @@ spec: app: ariadne ports: - {protocol: TCP, port: 8081} + - from: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: traefik + podSelector: + matchLabels: + app.kubernetes.io/name: traefik + ports: + - {protocol: TCP, port: 8082} + egress: + # The LAN listener and the original internal model-gate listeners share a + # pod. Constrain the entire pod so no prompt/response can leave the cluster: + # the only inference destination is the local Ollama Service. + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: kube-system + podSelector: + matchLabels: + k8s-app: kube-dns + ports: + - {protocol: UDP, port: 53} + - {protocol: TCP, port: 53} + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: vault + podSelector: + matchLabels: + app: vault + ports: + - {protocol: TCP, port: 8200} + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: ai + podSelector: + matchLabels: + app: ollama + ports: + - {protocol: TCP, port: 11434} + - to: + - podSelector: + matchLabels: + app: hermes-local-image + ports: + - {protocol: TCP, port: 9004} + # Kubernetes API for the existing image-handoff reads: the ClusterIP + # (matched pre-DNAT on some CNIs) plus the real apiserver endpoints on + # 6443, mirroring hermes-chat-tenant-isolation. + - to: + - ipBlock: + cidr: 10.43.0.1/32 + - ipBlock: + cidr: 192.168.22.11/32 + - ipBlock: + cidr: 192.168.22.12/32 + - ipBlock: + cidr: 192.168.22.13/32 + ports: + - {protocol: TCP, port: 443} + - {protocol: TCP, port: 6443} --- apiVersion: networking.k8s.io/v1 kind: NetworkPolicy diff --git a/services/vault/scripts/vault_hermes_model_gate_lan_token_ensure.sh b/services/vault/scripts/vault_hermes_model_gate_lan_token_ensure.sh old mode 100644 new mode 100755 diff --git a/testing/tests/test_hermes_model_gate_lan.py b/testing/tests/test_hermes_model_gate_lan.py new file mode 100644 index 00000000..9d835000 --- /dev/null +++ b/testing/tests/test_hermes_model_gate_lan.py @@ -0,0 +1,124 @@ +"""Exercise the LAN gateway's authentication and local-only HTTP boundary.""" + +import http.client +import io +import json +from pathlib import Path +import threading +from urllib.error import URLError + +import pytest +import yaml + + +@pytest.fixture +def gateway(tmp_path): + """Run the shipped handler with an isolated token and intercepted upstream.""" + path = Path(__file__).parents[2] / "services/hermes/model-gate-configmap.yaml" + namespace = {"__name__": "test_gateway"} + exec(compile(yaml.safe_load(path.read_text())["data"]["model_gate.py"], str(path), "exec"), namespace) + token = tmp_path / "token" + token.write_text("a" * 64) + namespace["LAN_API_TOKEN_FILE"] = token + calls = [] + + def upstream(request, timeout): + calls.append(request) + response = io.BytesIO(b'{"response":"LAN_OK","done":true}') + response.status = 200 + return response + + namespace["urlopen"] = upstream + server = namespace["ThreadingHTTPServer"](("127.0.0.1", 0), namespace["LanHandler"]) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + + def request(method="GET", path="/healthz", body=None, **headers): + defaults = {"Authorization": "Bearer " + "a" * 64, "X-Forwarded-For": "192.168.22.8"} + defaults.update(headers) + connection = http.client.HTTPConnection(*server.server_address, timeout=3) + connection.request(method, path, body=body, headers=defaults) + response = connection.getresponse() + result = response.status, json.loads(response.read()) + connection.close() + return result + + yield namespace, calls, request + server.shutdown() + server.server_close() + thread.join(timeout=2) + + +@pytest.mark.parametrize("address", ["", "203.0.113.1", "192.168.22.evil", "192.168.22.8, 203.0.113.1"]) +def test_non_lan_and_spoofed_addresses_are_rejected(gateway, address): + _, calls, request = gateway + assert request(**{"X-Forwarded-For": address})[0] == 403 + assert not calls + + +@pytest.mark.parametrize("token", ["", "Bearer wrong", "Basic abc", "Bearer é"]) +def test_token_is_required(gateway, token): + _, calls, request = gateway + assert request(Authorization=token)[0] == 401 + assert not calls + + +def test_health_requires_a_readable_credential(gateway): + namespace, calls, request = gateway + assert request()[0] == 200 + namespace["LAN_API_TOKEN_FILE"].unlink() + assert request()[0] == 503 + assert not calls + + +def test_generation_uses_only_local_upstream_without_forwarding_token(gateway): + namespace, calls, request = gateway + body = json.dumps({"model": namespace["LAN_MODEL"], "prompt": "Say LAN_OK", "stream": False}) + assert request("POST", "/api/generate", body) == (200, {"response": "LAN_OK", "done": True}) + assert calls[0].full_url == "http://ollama.ai.svc.cluster.local:11434/api/generate" + assert "Authorization" not in calls[0].headers + assert json.loads(calls[0].data)["options"]["num_predict"] == 256 + + +@pytest.mark.parametrize("changes", [ + {"model": "external/model"}, {"stream": True}, {"tools": []}, {"images": []}, + {"context": []}, {"keep_alive": 0}, {"options": {"num_predict": -1}}, + {"options": {"num_predict": 2049}}, {"options": {"num_gpu": 100}}, + {"options": {"temperature": float("nan")}}, {"prompt": []}, +]) +def test_unsupported_payloads_never_reach_ollama(gateway, changes): + namespace, calls, request = gateway + payload = {"model": namespace["LAN_MODEL"], "prompt": "hello", "stream": False, **changes} + assert request("POST", "/api/generate", json.dumps(payload))[0] == 400 + assert not calls + + +@pytest.mark.parametrize("length,status", [("-1", 413), ("131073", 413), ("invalid", 400)]) +def test_invalid_request_lengths_are_rejected_before_read(gateway, length, status): + _, calls, request = gateway + assert request("POST", "/api/generate", "{}", **{"Content-Length": length})[0] == status + assert not calls + + +def test_outage_and_concurrent_request_fail_without_fallback(gateway): + namespace, calls, request = gateway + body = json.dumps({"model": namespace["LAN_MODEL"], "prompt": "hello", "stream": False}) + namespace["_lan_inference"].acquire() + assert request("POST", "/api/generate", body)[0] == 429 + namespace["_lan_inference"].release() + + def unavailable(*args, **kwargs): + raise URLError("local service unavailable") + + namespace["urlopen"] = unavailable + assert request("POST", "/api/generate", body)[0] == 503 + assert namespace["_lan_inference"].acquire(blocking=False) + namespace["_lan_inference"].release() + assert not calls + + +@pytest.mark.parametrize("path", ["/api/pull", "/api/delete", "/v1/chat/completions", "/api/chat"]) +def test_model_management_and_agent_routes_are_not_exposed(gateway, path): + _, calls, request = gateway + assert request("POST", path, "{}")[0] == 404 + assert not calls