atlas-iac/testing/tests/test_hermes_handoff_policy.py
Hermes Agent 8f00545828 hermes: add a fail-closed full-handoff acceptance harness
Decides whether the Hermes platform handoff is fit to release, and refuses
to round an absence of evidence up to a pass.

The harness is read-only by default and classifies 71 checks PASS / FAIL /
NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a
harness-level problem: an unreachable vantage, a catalog entry whose evidence
no longer exists, an expired deadline, or an evaluator that raised.

Evidence comes from two vantages that cannot cover for each other: an external
read-only operator kubeconfig, and the Hermes agent probing itself from inside
its own pod. Before any check runs, the harness asks each vantage who it is and
stops if they are the same principal, because dual-vantage evidence from one
identity is a restatement rather than a corroboration. `--as` is rejected for
every operator-side command and reachable only as the inner command of a
`kubectl exec`, so impersonation can never stand in for a real self-probe. A
deny check needs a live refused request, not only an authorization review.

Two safety properties are structural rather than conventional, enforced where
an argv becomes a subprocess: the default mode mutates nothing (mutating verbs
require a server dry run; there is deliberately no live TokenRequest probe,
because a successful one would mint a real credential), and no probe can pull a
credential value into a report (no vault/sops/curl, secrets readable only with
-o name, environment probes list names, shell only through frozen reviewed
templates). Captures are bounded before they are screened, and the rendered
report is re-screened before it is written.

Mutation lives behind a separate arming flag with an exact confirmation phrase,
a caller-supplied unique ref, a preflight that refuses a protected push target
before any network call, and a cleanup whose verification is itself mandatory.
A default run reports those four checks NOT_RUN.

The catalog is declarative so a reviewer reads what is asserted rather than how
it is plumbed, and so structural properties can be proven over every entry
before a run. Catalog drift surfaces as NOT_RUN, which stops the release.

docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18
on top of the merged #13 baseline, the image build and Flux rollout, the
rollback point for each step, the go/no-go checklist, and the limits that are
asserted rather than exercised.

Validation: 295 handoff tests pass with 100% line coverage on all 15 new
modules; the full unit suite is 647 passed with two failures that reproduce
unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff
credential screen are clean; a live read-only run against Atlas returns NO_GO
for the pre-merge cluster with no unscreened fields in the report.
2026-08-17 10:14:17 +00:00

160 lines
7.2 KiB
Python

"""Contracts for the fail-closed command policy of the handoff harness."""
from __future__ import annotations
import pytest
from testing.tests.test_hermes_handoff_support import load_handoff_module
policy = load_handoff_module("hermes_handoff_policy")
def check(*argv: str, mode: str | None = None, inner: bool = False) -> None:
policy.check_argv(argv, mode or policy.READ_ONLY, allow_impersonation=inner)
@pytest.mark.parametrize(
"argv",
[
("kubectl", "--namespace", "hermes", "get", "pods"),
("kubectl", "auth", "can-i", "get", "secrets", "--all-namespaces"),
("kubectl", "get", "secrets", "--all-namespaces", "-o", "name"),
("kubectl", "get", "secrets", "--output=name"),
("kubectl", "logs", "deploy/x", "--tail", "1"),
("kubectl", "--context", "atlas", "get", "nodes", "-o", "json"),
("flux", "get", "kustomizations", "--all-namespaces"),
("helm", "list", "--all-namespaces"),
("git", "merge-base", "--is-ancestor", "a", "b"),
("git", "config", "--get", "remote.origin.url"),
("hermes", "kanban", "list", "--json"),
("hermes", "sessions", "list", "--source", "telegram"),
("hermes", "status"),
(policy.GITEA_CLIENT, "GET", "/api/v1/user"),
],
)
def test_read_only_commands_are_permitted(argv: tuple[str, ...]) -> None:
check(*argv)
@pytest.mark.parametrize(
("argv", "fragment"),
[
((), "empty command"),
(("vault", "read", "kv/x"), "forbidden binary"),
(("curl", "https://example.dev"), "forbidden binary"),
(("rm", "-rf", "/"), "allowlist"),
(("kubectl", "get", "pods", "--token", "abc"), "forbidden argument"),
(("kubectl", "get", "pods", "--as=system:admin"), "forbidden argument"),
(("kubectl", "get", "secrets", "-A"), "-o name"),
(("kubectl", "describe", "secrets", "x"), "-o name"),
(("kubectl", "delete", "-n", "get", "secrets"), "requires a dry run"),
(("kubectl", "port-forward", "svc/x", "80"), "outside the read-only boundary"),
(("kubectl", "exec", "pod"), "-- command boundary"),
(("kubectl", "exec", "pod", "--"), "inner command"),
(("kubectl",), "outside the read-only boundary"),
(("flux", "reconcile", "kustomization", "x"), "outside the read-only boundary"),
(("flux",), "requires a subcommand"),
(("helm", "upgrade", "x"), "outside the read-only boundary"),
(("git", "commit", "-m", "x"), "not permitted"),
(("git", "push", "origin", "HEAD:refs/heads/x"), "not permitted"),
(("git", "config", "user.name", "x"), "may only read"),
(("hermes", "kanban", "complete", "t_1"), "outside the read-only boundary"),
(("hermes",), "requires a subcommand"),
((policy.GITEA_CLIENT, "POST", "/api/v1/x"), "not permitted"),
((policy.GITEA_CLIENT, "GET"), "method and a path"),
(("sh", "-c", "rm -rf /"), "frozen template"),
(("sh", "echo"), "frozen template"),
(("git", "log", "/etc/shadow"), "credential path"),
],
)
def test_unsafe_commands_are_refused(argv: tuple[str, ...], fragment: str) -> None:
with pytest.raises(policy.PolicyError) as caught:
check(*argv)
assert fragment in str(caught.value)
def test_unknown_mode_is_refused() -> None:
with pytest.raises(policy.PolicyError, match="unknown mode"):
policy.check_argv(("kubectl", "get", "pods"), "whatever")
def test_dry_run_reaches_mutating_subcommands_without_writing() -> None:
check("kubectl", "--namespace", "hermes", "patch", "deploy/x", "--dry-run=server")
check("kubectl", "create", "configmap", "probe", "--dry-run=server")
def test_a_resource_named_after_a_subcommand_cannot_smuggle_a_delete() -> None:
"""`-n get` must not make a delete look like a read."""
with pytest.raises(policy.PolicyError, match="requires a dry run"):
check("kubectl", "delete", "-n", "get", "configmap", "x")
def test_impersonation_is_operator_forbidden_and_in_pod_permitted() -> None:
with pytest.raises(policy.PolicyError, match="forbidden argument"):
check("kubectl", "get", "namespaces", "--as", "system:admin")
check("kubectl", "get", "namespaces", "--as", "system:admin", inner=True)
def test_exec_validates_its_inner_command_and_tolerates_impersonation_inside() -> None:
inner = policy.shell("path_readable", path="/tmp")
check("kubectl", "exec", "--namespace", "hermes", "pod", "--", *inner)
check("kubectl", "exec", "pod", "--", "kubectl", "get", "ns", "--as", "system:admin")
with pytest.raises(policy.PolicyError, match="allowlist"):
check("kubectl", "exec", "pod", "--", "bash", "-c", "id")
def test_exec_of_a_shell_template_naming_a_secret_path_is_permitted() -> None:
"""The path scan stops at `--`; the tail is a reviewed template, not free text."""
inner = policy.shell("account_lock_state", account="hermes-agent", path="/host-etc/shadow")
check("kubectl", "exec", "--namespace", "hermes", "pod", "--", *inner)
def test_armed_mode_opens_exactly_the_mutations_it_should() -> None:
check("git", "push", "origin", "HEAD:refs/heads/x", mode=policy.ARMED)
check(policy.GITEA_CLIENT, "POST", "/api/v1/x", mode=policy.ARMED)
check(policy.GITEA_CLIENT, "DELETE", "/api/v1/x", mode=policy.ARMED)
with pytest.raises(policy.PolicyError, match="may not force"):
check("git", "push", "--force", "origin", "x", mode=policy.ARMED)
with pytest.raises(policy.PolicyError, match="may not force"):
check("git", "push", "--all", "origin", mode=policy.ARMED)
with pytest.raises(policy.PolicyError, match="not permitted"):
check(policy.GITEA_CLIENT, "PUT", "/api/v1/x", mode=policy.ARMED)
def test_every_frozen_template_renders_and_is_accepted() -> None:
parameters = {
"account": "hermes-agent",
"askpass": "/opt/coordinator/gitea_askpass.sh",
"fields": "state,model",
"limit": "200",
"path": "/host-etc/passwd",
"url": "https://scm.example.dev/atlas/repo.git",
}
for name in policy.SHELL_TEMPLATES:
required = policy.template_parameters(name)
argv = policy.shell(name, **{key: parameters[key] for key in required})
policy.check_argv(argv)
assert argv[0] == "sh"
def test_template_rendering_rejects_unknown_names_and_unsafe_parameters() -> None:
with pytest.raises(policy.PolicyError, match="unknown shell template"):
policy.render_shell("nope")
with pytest.raises(policy.PolicyError, match="expects parameters"):
policy.render_shell("path_mode")
with pytest.raises(policy.PolicyError, match="unsafe shell parameter"):
policy.render_shell("path_mode", path="/tmp; rm -rf /")
with pytest.raises(policy.PolicyError, match="unsafe shell parameter"):
policy.render_shell("path_mode", path="$(id)")
def test_positionals_skip_flags_and_stop_at_the_boundary() -> None:
argv = ("kubectl", "--namespace", "hermes", "get", "pods", "--", "ignored")
assert list(policy.positionals(argv)) == ["get", "pods"]
assert list(policy.positionals(("kubectl", "--all-namespaces", "get"))) == ["get"]
def test_a_subcommand_must_precede_its_resource_arguments() -> None:
with pytest.raises(policy.PolicyError, match="must precede"):
check("kubectl", "pods", "get")