ai: suppress backend content logs and verify isolated failures
This commit is contained in:
parent
45e9fa5635
commit
1e084493a9
@ -30,7 +30,9 @@ spec:
|
||||
args:
|
||||
- |
|
||||
while ! test -f /models/.pilot-v1-ready || ! test -f /reservation/reserved; do sleep 5; done
|
||||
exec ollama serve
|
||||
# Backend parser diagnostics can contain submitted schema text.
|
||||
# Keep request status/timing at the gateway; retain no backend output.
|
||||
exec ollama serve >/dev/null 2>&1
|
||||
env:
|
||||
- {name: OLLAMA_HOST, value: "0.0.0.0:11434"}
|
||||
- {name: OLLAMA_MODELS, value: /models}
|
||||
|
||||
@ -202,9 +202,12 @@ is accepted from clients or returned to them.
|
||||
|
||||
Traefik has access logs and tracing disabled. Gateway logs contain only a fixed
|
||||
route category and status; errors never echo request fields or backend bodies.
|
||||
Ollama's debug logging is disabled; observed logs contain HTTP/timing metadata.
|
||||
Ollama's stdout/stderr are suppressed, including runner output, so backend parser
|
||||
diagnostics cannot persist submitted schema text. Health, token counts, latency,
|
||||
and GPU metrics remain available.
|
||||
Both inference and gateway pods carry `fluentbit.io/exclude: "true"`, honored by
|
||||
the deployed collector. Fluent Bit also explicitly excludes their log paths.
|
||||
the deployed collector's existing `K8S-Logging.Exclude On` filter. No global
|
||||
collector configuration change is required.
|
||||
This matters because central OpenSearch uses Longhorn storage whose configured
|
||||
backup target is external B2 (`s3://atlas-soteria@us-west-004/`). The endpoint does
|
||||
not send its logs or contents into that pipeline. Existing OpenTelemetry exports
|
||||
@ -216,14 +219,15 @@ request tracing. The model volume is local-path, outside Longhorn backups.
|
||||
Changes are Git/Flux managed: the dedicated MetalLB LAN address, Traefik LAN
|
||||
LoadBalancer, allowlist/prefix ingress, scoped Vault token, restricted gateway
|
||||
listener/NetworkPolicy, native request validation, model pins, timeout transport,
|
||||
and Fluent Bit exclusions. The GPU reservation and dedicated runtime are also
|
||||
and pod-scoped log collection exclusions. The GPU reservation and dedicated runtime are also
|
||||
Flux managed; the existing Jetson deployment remains untouched.
|
||||
|
||||
To disable the endpoint, remove `model-gate-lan-ingress.yaml` from
|
||||
`services/hermes/kustomization.yaml`, commit and push the reviewed change, then
|
||||
reconcile `hermes` with source. Flux pruning removes the route and its middleware
|
||||
and transport. Revert that commit to restore it. This does not restart Ollama.
|
||||
Keep log exclusions while any real-data inference remains enabled. To roll back
|
||||
Keep backend output suppression and pod log exclusions while any real-data
|
||||
inference remains enabled. To roll back
|
||||
code, revert the relevant deployment commit(s) through Git, retain the log
|
||||
exclusions, and reconcile `hermes`; do not use manual kubectl edits.
|
||||
|
||||
|
||||
@ -74,7 +74,7 @@ spec:
|
||||
Name tail
|
||||
Tag kube.*
|
||||
Path /var/log/containers/*.log
|
||||
Exclude_Path /var/log/containers/*_POD_*.log,/var/log/containers/ollama-*_ai_ollama-*.log,/var/log/containers/hermes-model-gate-*_hermes_model-gate-*.log
|
||||
Exclude_Path /var/log/containers/*_POD_*.log
|
||||
Parser cri
|
||||
Mem_Buf_Limit 50MB
|
||||
Skip_Long_Lines On
|
||||
|
||||
@ -2,6 +2,8 @@
|
||||
|
||||
import io
|
||||
import json
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
import threading
|
||||
from urllib.error import HTTPError
|
||||
from urllib.request import ProxyHandler
|
||||
|
||||
@ -83,3 +85,44 @@ def test_timeout_releases_capacity_and_redacts_error(monkeypatch):
|
||||
assert status == 504 and "PRIVATE_PROMPT" not in json.dumps(result)
|
||||
assert api._inference.acquire(blocking=False)
|
||||
api._inference.release()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("failure,expected", [(503, 503), (307, 502)])
|
||||
def test_isolated_http_backend_failure_never_routes_elsewhere(monkeypatch, failure, expected):
|
||||
"""A fake local backend fails after preflight; no shared model is contacted."""
|
||||
calls = []
|
||||
|
||||
class Backend(BaseHTTPRequestHandler):
|
||||
"""Emulate metadata and a failing native generation operation."""
|
||||
|
||||
def do_GET(self):
|
||||
calls.append(self.path)
|
||||
result = ({"version": api.RUNTIME} if self.path == "/api/version" else
|
||||
{"models": [{"name": api.MODEL, "digest": api.DIGEST}]})
|
||||
self.send_response(200)
|
||||
self.end_headers()
|
||||
self.wfile.write(json.dumps(result).encode())
|
||||
|
||||
def do_POST(self):
|
||||
calls.append(self.path)
|
||||
self.rfile.read(int(self.headers["Content-Length"]))
|
||||
self.send_response(failure)
|
||||
self.send_header("Location", "/unapproved-fallback")
|
||||
self.end_headers()
|
||||
self.wfile.write(b'{"error":"PRIVATE_UPSTREAM_BODY"}')
|
||||
|
||||
def log_message(self, *args):
|
||||
pass
|
||||
|
||||
server = ThreadingHTTPServer(("127.0.0.1", 0), Backend)
|
||||
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
||||
thread.start()
|
||||
monkeypatch.setattr(api, "UPSTREAM", f"http://127.0.0.1:{server.server_port}")
|
||||
try:
|
||||
status, result = api.generate(body())
|
||||
assert status == expected and "PRIVATE_UPSTREAM_BODY" not in json.dumps(result)
|
||||
assert calls == ["/api/version", "/api/tags", "/api/generate"]
|
||||
finally:
|
||||
server.shutdown()
|
||||
server.server_close()
|
||||
thread.join(timeout=2)
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user