hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
"""Contracts for the handoff harness credential screen and output bounding."""
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
import json
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
from testing.tests.test_hermes_handoff_support import load_handoff_module
|
|
|
|
|
|
|
|
|
|
redaction = load_handoff_module("hermes_handoff_redaction")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
"text",
|
|
|
|
|
[
|
|
|
|
|
"token: ghp_ABCDEFGHIJKLMNOPQRSTUV1234",
|
|
|
|
|
"Authorization: Bearer abcdefghijklmnop",
|
|
|
|
|
"cookie: session=abcdefghijklmnop",
|
|
|
|
|
"password=hunter2000secret",
|
|
|
|
|
"-----BEGIN OPENSSH PRIVATE KEY-----\nabc\n-----END OPENSSH PRIVATE KEY-----",
|
|
|
|
|
"ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAI",
|
|
|
|
|
"eyJhbGciOiJIUzI1NiJ9.eyJzdWIiOiIxMjM0.SflKxwRJ",
|
|
|
|
|
"https://user:supersecretvalue@scm.example.dev/x",
|
|
|
|
|
"vault token hvs.CAESIJabcdefghijklmnop",
|
|
|
|
|
"aws AKIAIOSFODNN7EXAMPLE",
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
def test_credential_shapes_are_screened(text: str) -> None:
|
|
|
|
|
assert redaction.contains_credential_shape(text)
|
|
|
|
|
assert redaction.REDACTED in redaction.redact(text)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
"text",
|
|
|
|
|
[
|
|
|
|
|
"sha256:37ebf720c783ae908a602916ffccf88d43d205a157957f5dc4b487867aee45e7",
|
|
|
|
|
"ab346f55509d584e457fe26cf90be3078f7a375c",
|
|
|
|
|
'"automountServiceAccountToken": false',
|
|
|
|
|
'"key": "kubernetes.io/arch"',
|
|
|
|
|
"namespace: hermes",
|
|
|
|
|
"replicas: 3",
|
|
|
|
|
"checked_at: 2026-08-17T09:17:07.224400+00:00",
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
def test_ordinary_evidence_survives_screening(text: str) -> None:
|
|
|
|
|
assert redaction.redact(text) == text
|
|
|
|
|
assert not redaction.contains_credential_shape(text)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_screened_json_is_still_parsable_json() -> None:
|
|
|
|
|
"""A screen that corrupts structure would break every downstream evaluator."""
|
|
|
|
|
payload = {
|
|
|
|
|
"spec": {"automountServiceAccountToken": True, "key": "kubernetes.io/arch"},
|
|
|
|
|
"data": {"password": "a-long-enough-secret-value"},
|
|
|
|
|
}
|
|
|
|
|
screened = redaction.redact(json.dumps(payload, indent=2))
|
|
|
|
|
reparsed = json.loads(screened)
|
|
|
|
|
|
|
|
|
|
assert reparsed["spec"]["automountServiceAccountToken"] is True
|
|
|
|
|
assert reparsed["spec"]["key"] == "kubernetes.io/arch"
|
|
|
|
|
assert reparsed["data"]["password"] == redaction.REDACTED
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_high_entropy_runs_are_screened_but_hex_digests_are_not() -> None:
|
2026-08-17 13:11:34 +00:00
|
|
|
assert (
|
|
|
|
|
redaction.redact("Zm9vYmFyYmF6cXV4-Quux_1234567890ABCDEFghij")
|
|
|
|
|
!= redaction.REDACTED
|
|
|
|
|
)
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
long_mixed = "aB3" + "x9Y2z_" * 10
|
|
|
|
|
assert redaction.redact(long_mixed) == redaction.REDACTED
|
|
|
|
|
assert redaction.redact("f" * 64) == "f" * 64
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
("key", "sensitive"),
|
|
|
|
|
[
|
|
|
|
|
("ANTHROPIC_API_KEY", True),
|
|
|
|
|
("gitea_token", True),
|
|
|
|
|
("clientSecret", True),
|
|
|
|
|
("Set-Cookie", True),
|
|
|
|
|
("automountServiceAccountToken", True),
|
|
|
|
|
("key", False),
|
|
|
|
|
("namespace", False),
|
|
|
|
|
("", False),
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
def test_key_sensitivity_classification(key: str, sensitive: bool) -> None:
|
|
|
|
|
assert redaction.is_sensitive_key(key) is sensitive
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
("value", "possible"),
|
|
|
|
|
[
|
|
|
|
|
("true", False),
|
|
|
|
|
("False", False),
|
|
|
|
|
("3", False),
|
|
|
|
|
("12.5", False),
|
|
|
|
|
("short", False),
|
|
|
|
|
("", False),
|
|
|
|
|
("a-long-enough-secret", True),
|
|
|
|
|
('"another-long-secret"', True),
|
|
|
|
|
("1234567890123", False),
|
|
|
|
|
("-12.5e9", False),
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
def test_value_shape_gates_screening(value: str, possible: bool) -> None:
|
|
|
|
|
assert redaction.could_hold_secret(value) is possible
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_bound_truncates_to_a_byte_budget_and_reports_it() -> None:
|
|
|
|
|
text, truncated = redaction.bound("x" * 100, max_bytes=10)
|
|
|
|
|
assert truncated
|
2026-08-17 13:11:34 +00:00
|
|
|
assert len(text.encode()) <= 10
|
|
|
|
|
assert text == redaction.TRUNCATION_NOTE[:10]
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
|
|
|
|
|
text, truncated = redaction.bound("short", max_bytes=64)
|
|
|
|
|
assert (text, truncated) == ("short", False)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_bound_with_no_budget_keeps_nothing() -> None:
|
|
|
|
|
assert redaction.bound("content", max_bytes=0) == ("", True)
|
|
|
|
|
assert redaction.bound("", max_bytes=0) == ("", False)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_bound_never_splits_a_multibyte_character() -> None:
|
|
|
|
|
text, truncated = redaction.bound("é" * 10, max_bytes=5)
|
|
|
|
|
assert truncated
|
2026-08-17 13:11:34 +00:00
|
|
|
assert len(text.encode()) <= 5
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
|
|
|
|
|
|
2026-08-17 13:11:34 +00:00
|
|
|
def test_safe_text_screens_before_it_bounds() -> None:
|
|
|
|
|
"""A boundary must never preserve a useful credential prefix."""
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
payload = "password=" + "a" * 200
|
|
|
|
|
text, truncated = redaction.safe_text(payload, max_bytes=20)
|
2026-08-17 13:11:34 +00:00
|
|
|
assert not truncated
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
assert redaction.REDACTED in text
|
|
|
|
|
|
|
|
|
|
|
2026-08-17 13:11:34 +00:00
|
|
|
def test_no_byte_budget_exposes_a_credential_prefix() -> None:
|
|
|
|
|
payload = "password=A9secret-prefix-that-must-never-survive"
|
|
|
|
|
for budget in range(1, 64):
|
|
|
|
|
screened, _ = redaction.safe_text(payload, max_bytes=budget)
|
|
|
|
|
assert "A9" not in screened
|
|
|
|
|
assert len(screened.encode()) <= budget
|
|
|
|
|
|
|
|
|
|
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
def test_scrub_walks_nested_structures_and_keeps_useful_types() -> None:
|
|
|
|
|
scrubbed = redaction.scrub(
|
|
|
|
|
{
|
|
|
|
|
"token": "a-long-enough-secret",
|
|
|
|
|
"automountServiceAccountToken": False,
|
|
|
|
|
"items": [{"password": "another-long-secret"}, 3, None],
|
|
|
|
|
"note": ("tuple", "values"),
|
|
|
|
|
}
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert scrubbed["token"] == redaction.REDACTED
|
|
|
|
|
assert scrubbed["automountServiceAccountToken"] is False
|
|
|
|
|
assert scrubbed["items"][0]["password"] == redaction.REDACTED
|
|
|
|
|
assert scrubbed["items"][1] == 3
|
|
|
|
|
assert scrubbed["items"][2] is None
|
|
|
|
|
assert scrubbed["note"] == ["tuple", "values"]
|
|
|
|
|
|
|
|
|
|
|
2026-08-17 13:11:34 +00:00
|
|
|
def test_scrub_screens_short_numeric_nested_values_and_mapping_keys() -> None:
|
|
|
|
|
scrubbed = redaction.scrub(
|
|
|
|
|
{
|
|
|
|
|
"token": 1234,
|
|
|
|
|
"credentials": {"x": "y"},
|
|
|
|
|
"ghp_ABCDEF1234567890": "mapping-key",
|
|
|
|
|
"private_key": "x",
|
|
|
|
|
}
|
|
|
|
|
)
|
|
|
|
|
assert scrubbed["token"] == redaction.REDACTED
|
|
|
|
|
assert scrubbed["credentials"] == redaction.REDACTED
|
|
|
|
|
assert scrubbed["private_key"] == redaction.REDACTED
|
|
|
|
|
assert not any("ghp_" in key for key in scrubbed)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_unterminated_private_key_blocks_are_screened() -> None:
|
|
|
|
|
assert "PRIVATE KEY" not in redaction.redact("-----BEGIN PRIVATE KEY-----\nabc")
|
|
|
|
|
|
|
|
|
|
|
hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.
The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.
Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.
Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.
Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.
The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.
docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.
Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00
|
|
|
def test_screening_is_idempotent() -> None:
|
|
|
|
|
once = redaction.redact("api_key=AKIAIOSFODNN7EXAMPLE")
|
|
|
|
|
assert redaction.redact(once) == once
|
|
|
|
|
assert redaction.redact("") == ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_normalise_key_strips_separators() -> None:
|
|
|
|
|
assert redaction.normalise_key("Client-Secret_v2") == "clientsecretv2"
|