hermes(chat): parse large cluster reads before capping the output

The size cap was applied to the raw wire body, so a nodes list (huge
because of status.images) truncated before the image-stripping ran and
came back as a truncation notice. Accept up to 6 MiB on the wire to
parse and clean, then enforce the 384 KiB model-facing cap on the
stripped result - nodes now returns real data. Delivered via the
cluster-read ConfigMap; picked up on the next pod roll.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
This commit is contained in:
jenkins 2026-08-24 16:25:21 -03:00
parent b40300efaf
commit f7b3506a88
2 changed files with 18 additions and 8 deletions

View File

@ -23,6 +23,7 @@ TOKEN_FILE = Path("/var/run/secrets/kubernetes.io/serviceaccount/token")
CA_FILE = Path("/var/run/secrets/kubernetes.io/serviceaccount/ca.crt")
API_BASE = "https://kubernetes.default.svc"
MAX_RESPONSE_BYTES = 384 * 1024
MAX_WIRE_BYTES = 6 * 1024 * 1024 # accept a large response to clean, then cap the model-facing output
TIMEOUT_SECONDS = 8
SEGMENT = re.compile(r"^[a-z0-9][a-z0-9.\-]{0,252}$")
DENIED_RESOURCES = frozenset({"secrets"})
@ -148,19 +149,20 @@ def _handle_cluster_read(args: dict[str, Any] | None = None, **kwargs: Any) -> s
method="GET",
)
with urllib.request.urlopen(request, timeout=TIMEOUT_SECONDS, context=context) as response:
raw = response.read(MAX_RESPONSE_BYTES + 1)
raw = response.read(MAX_WIRE_BYTES + 1)
except urllib.error.HTTPError as error:
detail = error.read(2048).decode("utf-8", "replace")
return json.dumps({"error": f"cluster API returned {error.code}", "detail": detail[:400]})
except Exception as error: # noqa: BLE001 - tool result, never an exception into the loop
return json.dumps({"error": f"cluster API unavailable: {type(error).__name__}"})
over_wire = len(raw) > MAX_RESPONSE_BYTES
if over_wire:
# A truncated wire body is not parseable JSON; hand back the readable
# prefix rather than failing, so a large list still yields something.
if len(raw) > MAX_WIRE_BYTES:
# Genuinely enormous even before cleanup: a truncated wire body is not
# parseable, so hand back the readable prefix and ask to narrow.
return json.dumps({"truncated": True, "reason": "response exceeded size cap; narrow with namespace/name/label_selector",
"partial": raw[: MAX_RESPONSE_BYTES - 96].decode("utf-8", "replace")})
try:
# Parse the full response first, then strip heavy fields (e.g. node
# image lists) so the model-facing output cap applies to clean data.
payload = _strip_noise(json.loads(raw))
except ValueError:
return json.dumps({"error": "cluster API returned non-JSON data"})

View File

@ -81,13 +81,21 @@ def test_handler_is_get_only_and_bounded(monkeypatch, tmp_path):
assert "managedFields" not in json.dumps(out)
assert "last-applied" not in json.dumps(out)
def huge_open(request, timeout=None, context=None):
return FakeResponse(b'{"kind":"List","items":["' + b"x" * (mod.MAX_RESPONSE_BYTES + 100) + b'"]}')
def wire_huge(request, timeout=None, context=None):
return FakeResponse(b"x" * (mod.MAX_WIRE_BYTES + 100))
monkeypatch.setattr(mod.urllib.request, "urlopen", huge_open)
monkeypatch.setattr(mod.urllib.request, "urlopen", wire_huge)
out = json.loads(mod._handle_cluster_read({"resource": "pods"}))
assert out.get("truncated") is True and "partial" in out
def output_huge(request, timeout=None, context=None):
payload = {"kind": "List", "items": ["y" * (mod.MAX_RESPONSE_BYTES)]}
return FakeResponse(json.dumps(payload).encode())
monkeypatch.setattr(mod.urllib.request, "urlopen", output_huge)
out = json.loads(mod._handle_cluster_read({"resource": "pods"}))
assert out.get("truncated") is True
def node_open(request, timeout=None, context=None):
payload = {"kind": "NodeList", "items": [{"metadata": {"name": "n"},
"status": {"images": [{"names": ["x"], "sizeBytes": 1} for _ in range(500)]}}]}