gpu(titan-24): isolate FLUX render processes
All checks were successful
Tests / Declarative: Post Actions passed: 248

This commit is contained in:
jenkins 2026-08-11 15:13:08 -03:00
parent 021a6b4593
commit 8f67b979d9
3 changed files with 93 additions and 4 deletions

View File

@ -9,8 +9,11 @@ import io
import json import json
import os import os
import ssl import ssl
import subprocess
import sys
import threading import threading
import time import time
from contextlib import redirect_stdout
from datetime import UTC, datetime from datetime import UTC, datetime
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path from pathlib import Path
@ -36,6 +39,9 @@ LEASE_IDLE_OWNER = os.environ.get("LEASE_IDLE_OWNER", "hermes")
LEASE_IMAGE_OWNER = os.environ.get("LEASE_IMAGE_OWNER", "hermes-image") LEASE_IMAGE_OWNER = os.environ.get("LEASE_IMAGE_OWNER", "hermes-image")
MAX_BODY_BYTES = int(os.environ.get("HERMES_LOCAL_IMAGE_MAX_BODY", str(96 << 20))) MAX_BODY_BYTES = int(os.environ.get("HERMES_LOCAL_IMAGE_MAX_BODY", str(96 << 20)))
QUEUE_TIMEOUT_SECONDS = float(os.environ.get("HERMES_LOCAL_IMAGE_QUEUE_TIMEOUT", "1200")) QUEUE_TIMEOUT_SECONDS = float(os.environ.get("HERMES_LOCAL_IMAGE_QUEUE_TIMEOUT", "1200"))
RENDER_TIMEOUT_SECONDS = float(
os.environ.get("HERMES_LOCAL_IMAGE_RENDER_TIMEOUT", "1200")
)
GPU_ACTIVITY_URL = os.environ.get( GPU_ACTIVITY_URL = os.environ.get(
"HERMES_LOCAL_IMAGE_GPU_ACTIVITY_URL", "HERMES_LOCAL_IMAGE_GPU_ACTIVITY_URL",
"http://nvidia-process-exporter.monitoring.svc.cluster.local:9401/metrics", "http://nvidia-process-exporter.monitoring.svc.cluster.local:9401/metrics",
@ -245,8 +251,8 @@ def _decode_image(value: str):
return Image.open(io.BytesIO(raw)).convert("RGB") return Image.open(io.BytesIO(raw)).convert("RGB")
def _render(payload: dict[str, Any]) -> dict[str, Any]: def _render_in_process(payload: dict[str, Any]) -> dict[str, Any]:
"""Load FLUX for one request, render, then fully release its GPU memory.""" """Load and run FLUX inside the disposable CUDA worker process."""
import torch import torch
from diffusers import Flux2KleinPipeline from diffusers import Flux2KleinPipeline
@ -305,6 +311,52 @@ def _render(payload: dict[str, Any]) -> dict[str, Any]:
torch.cuda.ipc_collect() torch.cuda.ipc_collect()
def _render(payload: dict[str, Any]) -> dict[str, Any]:
"""Render in a child process so process exit releases every CUDA allocation."""
_set_state(phase="render-worker", last_model=MODEL_ID)
completed = subprocess.run(
[sys.executable, str(Path(__file__).resolve()), "--render-worker"],
input=json.dumps(payload, separators=(",", ":")).encode("utf-8"),
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
timeout=RENDER_TIMEOUT_SECONDS,
check=False,
)
if completed.returncode != 0:
detail = completed.stderr.decode("utf-8", errors="replace").strip()
if len(detail) > 4000:
detail = detail[-4000:]
raise RuntimeError(detail or f"FLUX worker exited {completed.returncode}")
try:
result = json.loads(completed.stdout)
except (UnicodeDecodeError, json.JSONDecodeError) as exc:
raise RuntimeError("FLUX worker returned an invalid response") from exc
if not isinstance(result, dict) or not result.get("success"):
raise RuntimeError("FLUX worker returned an unsuccessful response")
return result
def _run_render_worker() -> int:
"""Read one bounded request and emit one JSON response for the parent."""
raw = sys.stdin.buffer.read(MAX_BODY_BYTES + 1)
if not raw or len(raw) > MAX_BODY_BYTES:
print("invalid render worker request size", file=sys.stderr)
return 2
try:
payload = json.loads(raw)
if not isinstance(payload, dict):
raise ValueError("JSON object required")
# Keep library progress output off the machine-readable stdout channel.
with redirect_stdout(sys.stderr):
result = _render_in_process(payload)
sys.stdout.write(json.dumps(result, separators=(",", ":")))
sys.stdout.flush()
return 0
except Exception as exc:
print(f"{type(exc).__name__}: {exc}", file=sys.stderr, flush=True)
return 1
def _generate(payload: dict[str, Any]) -> dict[str, Any]: def _generate(payload: dict[str, Any]) -> dict[str, Any]:
"""Serialize renders and perform an exclusive lease-backed GPU handoff.""" """Serialize renders and perform an exclusive lease-backed GPU handoff."""
acquired = _generation_lock.acquire(timeout=QUEUE_TIMEOUT_SECONDS) acquired = _generation_lock.acquire(timeout=QUEUE_TIMEOUT_SECONDS)
@ -422,5 +474,7 @@ class Handler(BaseHTTPRequestHandler):
if __name__ == "__main__": if __name__ == "__main__":
if sys.argv[1:] == ["--render-worker"]:
raise SystemExit(_run_render_worker())
_recover_stale_image_lease() _recover_stale_image_lease()
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever() ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()

View File

@ -21,7 +21,7 @@ spec:
annotations: annotations:
ai.bstein.dev/model: black-forest-labs/FLUX.2-klein-4B ai.bstein.dev/model: black-forest-labs/FLUX.2-klein-4B
ai.bstein.dev/gpu: titan-24 lease-shared image, desktop, and Wolf lane ai.bstein.dev/gpu: titan-24 lease-shared image, desktop, and Wolf lane
ai.bstein.dev/config-rev: "20260811-flux2-klein-interactive-guard" ai.bstein.dev/config-rev: "20260811-flux2-klein-process-isolation"
spec: spec:
serviceAccountName: hermes-gpu-runtime serviceAccountName: hermes-gpu-runtime
runtimeClassName: nvidia runtimeClassName: nvidia
@ -43,7 +43,7 @@ spec:
sizeLimit: 2Gi sizeLimit: 2Gi
containers: containers:
- name: local-image - name: local-image
image: registry.bstein.dev/bstein/hermes-local-image@sha256:769a16f2b56b1de401f7dc4adafb522d39c8509d8de741668f69ee7a206502f7 image: registry.bstein.dev/bstein/hermes-local-image@sha256:b158e33adab1305f93e3857786ca577b9cf3a4540d556c2a847ba9eb411de0dc
imagePullPolicy: IfNotPresent imagePullPolicy: IfNotPresent
ports: ports:
- name: local-image - name: local-image
@ -77,6 +77,10 @@ spec:
value: "3" value: "3"
- name: HERMES_LOCAL_IMAGE_GPU_ACTIVITY_SAMPLE_INTERVAL - name: HERMES_LOCAL_IMAGE_GPU_ACTIVITY_SAMPLE_INTERVAL
value: "1" value: "1"
- name: HERMES_LOCAL_IMAGE_RENDER_TIMEOUT
value: "1200"
- name: PYTORCH_CUDA_ALLOC_CONF
value: expandable_segments:True
- name: HF_HOME - name: HF_HOME
value: /models/huggingface value: /models/huggingface
- name: HOME - name: HOME

View File

@ -724,6 +724,37 @@ def test_local_image_gpu_guard_distinguishes_idle_and_active_desktop(monkeypatch
assert "41%" in active["gpu_guard_reason"] assert "41%" in active["gpu_guard_reason"]
def test_local_flux_renderer_uses_a_disposable_cuda_worker(monkeypatch):
"""A completed render must not leave its CUDA context in the API process."""
source = ROOT / "dockerfiles" / "hermes-local-image-server.py"
spec = importlib.util.spec_from_file_location("hermes_local_image_worker", source)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
calls = []
def run(command, **kwargs):
calls.append((command, kwargs))
return SimpleNamespace(
returncode=0,
stdout=b'{"success":true,"route":"local","image_b64":"cG5n"}',
stderr=b"",
)
monkeypatch.setattr(module.subprocess, "run", run)
result = module._render({"prompt": "black cat", "aspect_ratio": "square"})
assert result["route"] == "local"
command, options = calls[0]
assert command[-1] == "--render-worker"
assert json.loads(options["input"]) == {
"prompt": "black cat",
"aspect_ratio": "square",
}
assert options["timeout"] == module.RENDER_TIMEOUT_SECONDS
assert options["check"] is False
def test_chat_auth_file_mount_survives_atomic_provider_refresh(): def test_chat_auth_file_mount_survives_atomic_provider_refresh():
statefulset = _documents(HERMES / "chat-statefulset.yaml")[0] statefulset = _documents(HERMES / "chat-statefulset.yaml")[0]
containers = statefulset["spec"]["template"]["spec"]["containers"] containers = statefulset["spec"]["template"]["spec"]["containers"]