From f7b3506a88248c24211289da32d7f69253a99584 Mon Sep 17 00:00:00 2001 From: jenkins Date: Mon, 24 Aug 2026 16:25:21 -0300 Subject: [PATCH] hermes(chat): parse large cluster reads before capping the output The size cap was applied to the raw wire body, so a nodes list (huge because of status.images) truncated before the image-stripping ran and came back as a truncation notice. Accept up to 6 MiB on the wire to parse and clean, then enforce the 384 KiB model-facing cap on the stripped result - nodes now returns real data. Delivered via the cluster-read ConfigMap; picked up on the next pod roll. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf --- services/hermes/plugins/cluster-read/__init__.py | 12 +++++++----- testing/tests/test_hermes_cluster_read_plugin.py | 14 +++++++++++--- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/services/hermes/plugins/cluster-read/__init__.py b/services/hermes/plugins/cluster-read/__init__.py index 921de89b..2747d33d 100644 --- a/services/hermes/plugins/cluster-read/__init__.py +++ b/services/hermes/plugins/cluster-read/__init__.py @@ -23,6 +23,7 @@ TOKEN_FILE = Path("/var/run/secrets/kubernetes.io/serviceaccount/token") CA_FILE = Path("/var/run/secrets/kubernetes.io/serviceaccount/ca.crt") API_BASE = "https://kubernetes.default.svc" MAX_RESPONSE_BYTES = 384 * 1024 +MAX_WIRE_BYTES = 6 * 1024 * 1024 # accept a large response to clean, then cap the model-facing output TIMEOUT_SECONDS = 8 SEGMENT = re.compile(r"^[a-z0-9][a-z0-9.\-]{0,252}$") DENIED_RESOURCES = frozenset({"secrets"}) @@ -148,19 +149,20 @@ def _handle_cluster_read(args: dict[str, Any] | None = None, **kwargs: Any) -> s method="GET", ) with urllib.request.urlopen(request, timeout=TIMEOUT_SECONDS, context=context) as response: - raw = response.read(MAX_RESPONSE_BYTES + 1) + raw = response.read(MAX_WIRE_BYTES + 1) except urllib.error.HTTPError as error: detail = error.read(2048).decode("utf-8", "replace") return json.dumps({"error": f"cluster API returned {error.code}", "detail": detail[:400]}) except Exception as error: # noqa: BLE001 - tool result, never an exception into the loop return json.dumps({"error": f"cluster API unavailable: {type(error).__name__}"}) - over_wire = len(raw) > MAX_RESPONSE_BYTES - if over_wire: - # A truncated wire body is not parseable JSON; hand back the readable - # prefix rather than failing, so a large list still yields something. + if len(raw) > MAX_WIRE_BYTES: + # Genuinely enormous even before cleanup: a truncated wire body is not + # parseable, so hand back the readable prefix and ask to narrow. return json.dumps({"truncated": True, "reason": "response exceeded size cap; narrow with namespace/name/label_selector", "partial": raw[: MAX_RESPONSE_BYTES - 96].decode("utf-8", "replace")}) try: + # Parse the full response first, then strip heavy fields (e.g. node + # image lists) so the model-facing output cap applies to clean data. payload = _strip_noise(json.loads(raw)) except ValueError: return json.dumps({"error": "cluster API returned non-JSON data"}) diff --git a/testing/tests/test_hermes_cluster_read_plugin.py b/testing/tests/test_hermes_cluster_read_plugin.py index dc62ba59..cc55e127 100644 --- a/testing/tests/test_hermes_cluster_read_plugin.py +++ b/testing/tests/test_hermes_cluster_read_plugin.py @@ -81,13 +81,21 @@ def test_handler_is_get_only_and_bounded(monkeypatch, tmp_path): assert "managedFields" not in json.dumps(out) assert "last-applied" not in json.dumps(out) - def huge_open(request, timeout=None, context=None): - return FakeResponse(b'{"kind":"List","items":["' + b"x" * (mod.MAX_RESPONSE_BYTES + 100) + b'"]}') + def wire_huge(request, timeout=None, context=None): + return FakeResponse(b"x" * (mod.MAX_WIRE_BYTES + 100)) - monkeypatch.setattr(mod.urllib.request, "urlopen", huge_open) + monkeypatch.setattr(mod.urllib.request, "urlopen", wire_huge) out = json.loads(mod._handle_cluster_read({"resource": "pods"})) assert out.get("truncated") is True and "partial" in out + def output_huge(request, timeout=None, context=None): + payload = {"kind": "List", "items": ["y" * (mod.MAX_RESPONSE_BYTES)]} + return FakeResponse(json.dumps(payload).encode()) + + monkeypatch.setattr(mod.urllib.request, "urlopen", output_huge) + out = json.loads(mod._handle_cluster_read({"resource": "pods"})) + assert out.get("truncated") is True + def node_open(request, timeout=None, context=None): payload = {"kind": "NodeList", "items": [{"metadata": {"name": "n"}, "status": {"images": [{"names": ["x"], "sizeBytes": 1} for _ in range(500)]}}]}