atlas-iac/testing/tests/test_hermes_chat_local_runtime.py

173 lines
7.0 KiB
Python
Raw Normal View History

"""Local inference and sandbox execution contracts for Hermes chat."""
from __future__ import annotations
import importlib.util
import json
from pathlib import Path
from types import SimpleNamespace
from testing.tests.test_hermes_chat_support import (
HERMES,
ROOT,
_documents,
)
def test_titan20_serializes_classifier_and_local_chat_model_residency():
"""Classifier and local chat share one serialized resident Qwen weight."""
deployment = _documents(
Path(__file__).parents[2] / "services/ai-llm/deployment.yaml"
)[0]
pod = deployment["spec"]["template"]["spec"]
required = pod["affinity"]["nodeAffinity"][
"requiredDuringSchedulingIgnoredDuringExecution"
]["nodeSelectorTerms"][0]["matchExpressions"][0]
assert required["values"] == ["titan-20"]
container = pod["containers"][0]
env = {item["name"]: item["value"] for item in container["env"]}
assert env["OLLAMA_MAX_LOADED_MODELS"] == "1"
assert env["OLLAMA_NUM_PARALLEL"] == "1"
assert env["OLLAMA_KEEP_ALIVE"] == "-1"
assert env["OLLAMA_CONTEXT_LENGTH"] == "8192"
warm_command = " ".join(container["command"])
assert "--keepalive=-1" not in warm_command
models = next(item for item in pod["volumes"] if item["name"] == "models")
assert models["persistentVolumeClaim"]["claimName"] == ("ollama-models-titan20")
def test_local_image_gpu_guard_distinguishes_background_and_saturated_gpu(monkeypatch):
"""Lease-idle desktop spikes may coexist, but saturation still blocks FLUX."""
source = ROOT / "dockerfiles" / "hermes-local-image-server.py"
spec = importlib.util.spec_from_file_location("hermes_local_image_server", source)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
monkeypatch.setattr(module, "GPU_ACTIVITY_NODE", "titan-24")
monkeypatch.setattr(module, "GPU_ACTIVE_SM_PERCENT", 80.0)
monkeypatch.setattr(module, "GPU_MAX_EXTERNAL_MEMORY_BYTES", 3 << 30)
idle = module._parse_gpu_activity(
"\n".join(
[
'nvidia_process_gpu_sm_util_percent{node="titan-24",namespace="host",process="Xorg"} 0',
'nvidia_process_gpu_memory_used_bytes{node="titan-24",namespace="host",process="Xorg"} 1900000000',
'nvidia_process_gpu_sm_util_percent{node="titan-24",namespace="game-stream",process="wolf"} 3',
'nvidia_process_gpu_memory_used_bytes{node="titan-24",namespace="hermes",process="python"} 9000000000',
]
)
)
assert idle["interactive_active"] is False
assert idle["external_gpu_memory_bytes"] == 1900000000
assert idle["external_gpu_sm_percent"] == 3
background_spike = module._parse_gpu_activity(
'nvidia_process_gpu_sm_util_percent{node="titan-24",namespace="host",process="sway"} 41\n'
)
assert background_spike["interactive_active"] is False
active = module._parse_gpu_activity(
'nvidia_process_gpu_sm_util_percent{node="titan-24",namespace="host",process="steam"} 91\n'
)
assert active["interactive_active"] is True
assert "91%" in active["gpu_guard_reason"]
def test_local_flux_renderer_uses_a_disposable_cuda_worker(monkeypatch):
"""A completed render must not leave its CUDA context in the API process."""
source = ROOT / "dockerfiles" / "hermes-local-image-server.py"
spec = importlib.util.spec_from_file_location("hermes_local_image_worker", source)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
calls = []
def run(command, **kwargs):
calls.append((command, kwargs))
return SimpleNamespace(
returncode=0,
stdout=b'{"success":true,"route":"local","image_b64":"cG5n"}',
stderr=b"",
)
monkeypatch.setattr(module.subprocess, "run", run)
result = module._render({"prompt": "black cat", "aspect_ratio": "square"})
assert result["route"] == "local"
command, options = calls[0]
assert command[-1] == "--render-worker"
assert json.loads(options["input"]) == {
"prompt": "black cat",
"aspect_ratio": "square",
}
assert options["timeout"] == module.RENDER_TIMEOUT_SECONDS
assert options["check"] is False
def test_local_flux_uses_low_vram_offload_without_reducing_resolution():
"""The shared 3080 lane must trade time, not image size, for headroom."""
source = (ROOT / "dockerfiles" / "hermes-local-image-server.py").read_text()
assert "OFFLOAD_MODE = os.environ.get(" in source
assert '"HERMES_LOCAL_IMAGE_OFFLOAD_MODE", "sequential"' in source
assert "pipe.enable_sequential_cpu_offload()" in source
assert '"square": (1024, 1024)' in source
def test_chat_auth_and_relay_are_pod_lifetime_only():
statefulset = _documents(HERMES / "chat-statefulset.yaml")[0]
pod = statefulset["spec"]["template"]["spec"]
containers = statefulset["spec"]["template"]["spec"]["containers"]
for name in ("hermes", "webui"):
container = next(item for item in containers if item["name"] == name)
env = {item["name"]: item["value"] for item in container["env"]}
assert env["HERMES_AUTH_FILE"] == "/runtime-access/hermes-auth.json"
mount = next(
item
for item in container["volumeMounts"]
if item["name"] == "runtime-access"
)
assert mount["mountPath"] == "/runtime-access"
assert "subPath" not in mount
hermes_env = {
item["name"]: item["value"]
for item in next(item for item in containers if item["name"] == "hermes")["env"]
}
runtime = next(item for item in pod["volumes"] if item["name"] == "runtime-access")
assert runtime["emptyDir"] == {"medium": "Memory", "sizeLimit": "2Mi"}
assert not any(item["name"] == "provider-auth" for item in pod["volumes"])
init_command = next(
item for item in pod["initContainers"] if item["name"] == "init-config"
)["command"][2]
for key in (
"ANTHROPIC_API_KEY",
"API_SERVER_KEY",
"CLAUDE_CODE_OAUTH_TOKEN",
"GITEA_TOKEN",
"HERMES_IMAGE_BROKER_KEY",
"OPENAI_API_KEY",
):
assert key in init_command
assert "printf 'API_SERVER_KEY=%s" not in init_command
assert hermes_env["AGENT_BROWSER_EXECUTABLE_PATH"].endswith(
"/chrome-linux/headless_shell"
)
assert "--no-sandbox" in hermes_env["AGENT_BROWSER_ARGS"]
def test_sandbox_executes_python_with_bounded_output(tmp_path: Path, monkeypatch):
source = ROOT / "dockerfiles" / "hermes-chat-sandbox-server.py"
spec = importlib.util.spec_from_file_location("hermes_chat_sandbox_server", source)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
monkeypatch.setattr(module, "WORKSPACE", tmp_path)
result = module._execute("import math\nprint(math.comb(10, 3))")
assert result["success"] is True
assert result["stdout"] == "120\n"
assert result["stderr"] == ""