From 1e084493a97331f1eca0c175ecee5c7df8422049 Mon Sep 17 00:00:00 2001 From: jenkins Date: Mon, 28 Sep 2026 19:18:27 -0500 Subject: [PATCH] ai: suppress backend content logs and verify isolated failures --- services/ai-llm/gpu-deployment.yaml | 4 +- services/hermes/LAN_API.md | 12 ++++-- services/logging/fluent-bit-helmrelease.yaml | 2 +- testing/tests/test_lan_generate.py | 43 ++++++++++++++++++++ 4 files changed, 55 insertions(+), 6 deletions(-) diff --git a/services/ai-llm/gpu-deployment.yaml b/services/ai-llm/gpu-deployment.yaml index 2de30d25..97b968c9 100644 --- a/services/ai-llm/gpu-deployment.yaml +++ b/services/ai-llm/gpu-deployment.yaml @@ -30,7 +30,9 @@ spec: args: - | while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done - exec ollama serve + # Backend parser diagnostics can contain submitted schema text. + # Keep request status/timing at the gateway; retain no backend output. + exec ollama serve >/dev/null 2>&1 env: - {name: OLLAMA_HOST, value: "0.0.0.0:11434"} - {name: OLLAMA_MODELS, value: /models} diff --git a/services/hermes/LAN_API.md b/services/hermes/LAN_API.md index d2629c8c..a9567ab6 100644 --- a/services/hermes/LAN_API.md +++ b/services/hermes/LAN_API.md @@ -202,9 +202,12 @@ is accepted from clients or returned to them. Traefik has access logs and tracing disabled. Gateway logs contain only a fixed route category and status; errors never echo request fields or backend bodies. -Ollama's debug logging is disabled; observed logs contain HTTP/timing metadata. +Ollama's stdout/stderr are suppressed, including runner output, so backend parser +diagnostics cannot persist submitted schema text. Health, token counts, latency, +and GPU metrics remain available. Both inference and gateway pods carry `fluentbit.io/exclude: "true"`, honored by -the deployed collector. Fluent Bit also explicitly excludes their log paths. +the deployed collector's existing `K8S-Logging.Exclude On` filter. No global +collector configuration change is required. This matters because central OpenSearch uses Longhorn storage whose configured backup target is external B2 (`s3://atlas-soteria@us-west-004/`). The endpoint does not send its logs or contents into that pipeline. Existing OpenTelemetry exports @@ -216,14 +219,15 @@ request tracing. The model volume is local-path, outside Longhorn backups. Changes are Git/Flux managed: the dedicated MetalLB LAN address, Traefik LAN LoadBalancer, allowlist/prefix ingress, scoped Vault token, restricted gateway listener/NetworkPolicy, native request validation, model pins, timeout transport, -and Fluent Bit exclusions. The GPU reservation and dedicated runtime are also +and pod-scoped log collection exclusions. The GPU reservation and dedicated runtime are also Flux managed; the existing Jetson deployment remains untouched. To disable the endpoint, remove `model-gate-lan-ingress.yaml` from `services/hermes/kustomization.yaml`, commit and push the reviewed change, then reconcile `hermes` with source. Flux pruning removes the route and its middleware and transport. Revert that commit to restore it. This does not restart Ollama. -Keep log exclusions while any real-data inference remains enabled. To roll back +Keep backend output suppression and pod log exclusions while any real-data +inference remains enabled. To roll back code, revert the relevant deployment commit(s) through Git, retain the log exclusions, and reconcile `hermes`; do not use manual kubectl edits. diff --git a/services/logging/fluent-bit-helmrelease.yaml b/services/logging/fluent-bit-helmrelease.yaml index 1da3af10..34b755ff 100644 --- a/services/logging/fluent-bit-helmrelease.yaml +++ b/services/logging/fluent-bit-helmrelease.yaml @@ -74,7 +74,7 @@ spec: Name tail Tag kube.* Path /var/log/containers/*.log - Exclude_Path /var/log/containers/*_POD_*.log,/var/log/containers/ollama-*_ai_ollama-*.log,/var/log/containers/hermes-model-gate-*_hermes_model-gate-*.log + Exclude_Path /var/log/containers/*_POD_*.log Parser cri Mem_Buf_Limit 50MB Skip_Long_Lines On diff --git a/testing/tests/test_lan_generate.py b/testing/tests/test_lan_generate.py index 3844b449..153a87af 100644 --- a/testing/tests/test_lan_generate.py +++ b/testing/tests/test_lan_generate.py @@ -2,6 +2,8 @@ import io import json +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +import threading from urllib.error import HTTPError from urllib.request import ProxyHandler @@ -83,3 +85,44 @@ def test_timeout_releases_capacity_and_redacts_error(monkeypatch): assert status == 504 and "PRIVATE_PROMPT" not in json.dumps(result) assert api._inference.acquire(blocking=False) api._inference.release() + + +@pytest.mark.parametrize("failure,expected", [(503, 503), (307, 502)]) +def test_isolated_http_backend_failure_never_routes_elsewhere(monkeypatch, failure, expected): + """A fake local backend fails after preflight; no shared model is contacted.""" + calls = [] + + class Backend(BaseHTTPRequestHandler): + """Emulate metadata and a failing native generation operation.""" + + def do_GET(self): + calls.append(self.path) + result = ({"version": api.RUNTIME} if self.path == "/api/version" else + {"models": [{"name": api.MODEL, "digest": api.DIGEST}]}) + self.send_response(200) + self.end_headers() + self.wfile.write(json.dumps(result).encode()) + + def do_POST(self): + calls.append(self.path) + self.rfile.read(int(self.headers["Content-Length"])) + self.send_response(failure) + self.send_header("Location", "/unapproved-fallback") + self.end_headers() + self.wfile.write(b'{"error":"PRIVATE_UPSTREAM_BODY"}') + + def log_message(self, *args): + pass + + server = ThreadingHTTPServer(("127.0.0.1", 0), Backend) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + monkeypatch.setattr(api, "UPSTREAM", f"http://127.0.0.1:{server.server_port}") + try: + status, result = api.generate(body()) + assert status == expected and "PRIVATE_UPSTREAM_BODY" not in json.dumps(result) + assert calls == ["/api/version", "/api/tags", "/api/generate"] + finally: + server.shutdown() + server.server_close() + thread.join(timeout=2)