Decides whether the Hermes platform handoff is fit to release, and refuses to round an absence of evidence up to a pass. The harness is read-only by default and classifies 71 checks PASS / FAIL / NOT_RUN / NOT_APPLICABLE. Any mandatory FAIL or NOT_RUN is NO_GO, and so is a harness-level problem: an unreachable vantage, a catalog entry whose evidence no longer exists, an expired deadline, or an evaluator that raised. Evidence comes from two vantages that cannot cover for each other: an external read-only operator kubeconfig, and the Hermes agent probing itself from inside its own pod. Before any check runs, the harness asks each vantage who it is and stops if they are the same principal, because dual-vantage evidence from one identity is a restatement rather than a corroboration. `--as` is rejected for every operator-side command and reachable only as the inner command of a `kubectl exec`, so impersonation can never stand in for a real self-probe. A deny check needs a live refused request, not only an authorization review. Two safety properties are structural rather than conventional, enforced where an argv becomes a subprocess: the default mode mutates nothing (mutating verbs require a server dry run; there is deliberately no live TokenRequest probe, because a successful one would mint a real credential), and no probe can pull a credential value into a report (no vault/sops/curl, secrets readable only with -o name, environment probes list names, shell only through frozen reviewed templates). Captures are bounded before they are screened, and the rendered report is re-screened before it is written. Mutation lives behind a separate arming flag with an exact confirmation phrase, a caller-supplied unique ref, a preflight that refuses a protected push target before any network call, and a cleanup whose verification is itself mandatory. A default run reports those four checks NOT_RUN. The catalog is declarative so a reviewer reads what is asserted rather than how it is plumbed, and so structural properties can be proven over every entry before a run. Catalog drift surfaces as NOT_RUN, which stops the release. docs/hermes_full_handoff_acceptance.md carries the merge order for PRs #14-#18 on top of the merged #13 baseline, the image build and Flux rollout, the rollback point for each step, the go/no-go checklist, and the limits that are asserted rather than exercised. Validation: 295 handoff tests pass with 100% line coverage on all 15 new modules; the full unit suite is 647 passed with two failures that reproduce unchanged on origin/main; Ruff, py_compile, kustomize render, and a diff credential screen are clean; a live read-only run against Atlas returns NO_GO for the pre-merge cluster with no unscreened fields in the report.
195 lines
7.9 KiB
Python
195 lines
7.9 KiB
Python
"""Structural contracts for the handoff acceptance catalog."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import datetime as dt
|
|
|
|
import pytest
|
|
|
|
from testing.tests.test_hermes_handoff_support import load_handoff_module
|
|
|
|
catalog = load_handoff_module("hermes_handoff_catalog")
|
|
harness_run = load_handoff_module("hermes_handoff_run")
|
|
evaluators = load_handoff_module("hermes_handoff_evaluators")
|
|
model = load_handoff_module("hermes_handoff_model")
|
|
policy = load_handoff_module("hermes_handoff_policy")
|
|
exec_module = load_handoff_module("hermes_handoff_exec")
|
|
|
|
NOW = dt.datetime(2026, 8, 17, 12, 0, 0, tzinfo=dt.timezone.utc)
|
|
TARGETS = catalog.Targets(now=NOW)
|
|
SPECS = harness_run.build_catalog(TARGETS)
|
|
BY_ID = {check.id: check for check in SPECS}
|
|
|
|
VANTAGES = {
|
|
catalog.OPERATOR: exec_module.operator_vantage(context="atlas-operator"),
|
|
catalog.SELF: exec_module.pod_vantage("hermes", "hermes-agent-1", "hermes"),
|
|
catalog.SWITCHYARD: exec_module.pod_vantage("hermes", "hermes-switchyard-1", "switchyard"),
|
|
catalog.NODE: exec_module.pod_vantage("hermes", "hermes-node-ssh-access-1", "key-reconciler"),
|
|
catalog.CHAT: exec_module.pod_vantage("hermes", "hermes-chat-tenant-0", "hermes"),
|
|
}
|
|
|
|
|
|
def test_the_catalog_is_structurally_valid() -> None:
|
|
assert harness_run.validate_catalog(SPECS) == []
|
|
assert len(SPECS) >= 60
|
|
|
|
|
|
def test_every_step_is_accepted_by_the_read_only_policy_from_its_vantage() -> None:
|
|
"""A catalog entry that cannot run under policy is a bug found before a run."""
|
|
for check in SPECS:
|
|
for step in check.steps:
|
|
addressed = VANTAGES[step.vantage].wrap(step.argv)
|
|
policy.check_argv(addressed, policy.READ_ONLY)
|
|
|
|
|
|
def test_no_step_mutates_outside_a_server_dry_run() -> None:
|
|
"""`auth can-i create ...` is a read; the subcommand is what decides."""
|
|
for check in SPECS:
|
|
for step in check.steps:
|
|
subcommand = next(policy.positionals(step.argv), "")
|
|
if subcommand in policy.KUBECTL_DRY_RUN_SUBCOMMANDS:
|
|
assert any(flag in step.argv for flag in policy.DRY_RUN_FLAGS), check.id
|
|
|
|
|
|
def test_no_operator_step_impersonates_and_at_least_one_in_pod_step_does() -> None:
|
|
operator_steps = [
|
|
step for check in SPECS for step in check.steps if step.vantage == catalog.OPERATOR
|
|
]
|
|
assert operator_steps
|
|
assert not any(set(step.argv) & set(policy.IMPERSONATION_ARGS) for step in operator_steps)
|
|
|
|
impersonation = BY_ID["access.impersonation-is-denied"]
|
|
attempts = [step for step in impersonation.steps if step.kind == model.ATTEMPT]
|
|
assert any("--as" in step.argv for step in attempts)
|
|
|
|
|
|
def test_every_deny_check_pairs_a_review_with_a_live_attempt() -> None:
|
|
deny_checks = [check for check in SPECS if check.rule == "denied"]
|
|
assert len(deny_checks) >= 6
|
|
for check in deny_checks:
|
|
assert any(step.kind == model.ATTEMPT for step in check.steps), check.id
|
|
|
|
|
|
def test_both_vantages_are_represented_across_the_catalog() -> None:
|
|
used = {step.vantage for check in SPECS for step in check.steps}
|
|
assert catalog.OPERATOR in used and catalog.SELF in used
|
|
assert used <= set(catalog.VANTAGE_NAMES)
|
|
|
|
|
|
def test_every_rule_named_by_a_check_is_registered() -> None:
|
|
for check in SPECS:
|
|
if check.scope == model.EPHEMERAL:
|
|
assert check.rule == "not_armed"
|
|
continue
|
|
assert check.rule in evaluators.EVALUATORS, check.id
|
|
|
|
|
|
def test_every_evidence_step_referenced_by_expect_exists() -> None:
|
|
for check in SPECS:
|
|
keys = {step.key for step in check.steps}
|
|
named = [check.expect.get("step")] + list(check.expect.get("steps", ()))
|
|
for key in filter(None, named):
|
|
assert key in keys, f"{check.id} names missing step {key}"
|
|
|
|
|
|
def test_the_default_catalog_reports_every_mutating_check_as_not_armed() -> None:
|
|
ephemeral = [check for check in SPECS if check.scope == model.EPHEMERAL]
|
|
assert len(ephemeral) == 4
|
|
for check in ephemeral:
|
|
assert check.mandatory is False
|
|
assert evaluators.evaluate(check, {}).status == model.NOT_RUN
|
|
|
|
|
|
def test_dependency_pull_requests_each_get_their_own_merge_gate() -> None:
|
|
for number in TARGETS.dependency_pull_requests:
|
|
check = BY_ID[f"baseline.dependency-pr-{number}-merged"]
|
|
assert check.mandatory
|
|
assert check.expect["fields"] == {"merged": True, "base.ref": "main"}
|
|
|
|
|
|
def test_the_baseline_commit_gate_names_the_configured_commit() -> None:
|
|
check = BY_ID["baseline.origin-main-descends-merged-work"]
|
|
assert TARGETS.baseline_commit in check.steps[0].argv
|
|
assert "--is-ancestor" in check.steps[0].argv
|
|
|
|
|
|
def test_optional_checks_appear_only_when_a_site_declares_what_to_expect() -> None:
|
|
"""A names_present over an empty expectation would pass on any workload."""
|
|
assert "pool.assignment-safety-knobs-are-configured" not in BY_ID
|
|
assert "surfaces.chat-runs-the-telegram-topic-revision" not in BY_ID
|
|
assert "surfaces.chat-telegram-sessions-are-continuous" not in BY_ID
|
|
|
|
configured = harness_run.build_catalog(
|
|
catalog.Targets(
|
|
now=NOW,
|
|
pool_worker_env=("HERMES_POOL_LEASE_SECONDS",),
|
|
chat_config_revision="20260816-telegram-topics",
|
|
expect_telegram_sessions=True,
|
|
)
|
|
)
|
|
identifiers = {check.id for check in configured}
|
|
assert "pool.assignment-safety-knobs-are-configured" in identifiers
|
|
assert "surfaces.chat-runs-the-telegram-topic-revision" in identifiers
|
|
assert "surfaces.chat-telegram-sessions-are-continuous" in identifiers
|
|
assert harness_run.validate_catalog(configured) == []
|
|
|
|
|
|
def test_bulk_evidence_is_parsed_but_not_pasted_into_the_report() -> None:
|
|
routing = BY_ID["routing.provider-effort-and-fallback-evidence"]
|
|
assert routing.steps[0].record is False
|
|
assert routing.steps[0].max_bytes == TARGETS.routing_tail_bytes
|
|
|
|
kanban = BY_ID["surfaces.kanban-activity-is-visible"]
|
|
assert kanban.steps[0].record is False
|
|
|
|
|
|
def test_freshness_gates_carry_the_run_clock() -> None:
|
|
for identifier in ("identity.codex-evidence-is-fresh", "identity.claude-evidence-is-fresh"):
|
|
assert BY_ID[identifier].expect["now"] == NOW
|
|
routing = BY_ID["routing.provider-effort-and-fallback-evidence"]
|
|
assert routing.expect["now"] == NOW
|
|
assert routing.expect["max_age_seconds"] == TARGETS.max_evidence_age_seconds
|
|
|
|
|
|
def test_check_identifiers_are_unique_and_grouped() -> None:
|
|
identifiers = [check.id for check in SPECS]
|
|
assert len(identifiers) == len(set(identifiers))
|
|
for check in SPECS:
|
|
assert check.id.split(".")[0]
|
|
assert check.group
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"identifier",
|
|
[
|
|
"identity.no-provider-api-key-in-pod",
|
|
"identity.no-provider-api-key-in-manifest",
|
|
"forge.worker-holds-no-git-credential-variable",
|
|
"access.secrets-are-denied",
|
|
"access.service-account-tokens-are-denied",
|
|
"access.workload-mutation-is-denied",
|
|
"access.pod-exec-is-denied",
|
|
"access.required-reads-are-allowed",
|
|
"access.pod-logs-are-allowed",
|
|
"access.flux-and-helm-status-are-allowed",
|
|
"access.namespace-view-agrees-across-vantages",
|
|
"nodes.account-password-is-locked",
|
|
"nodes.hardening-covers-every-node",
|
|
"build.builder-service-account-is-tokenless",
|
|
"build.harbor-immutability-rule-is-applied",
|
|
"build.deployed-image-is-digest-pinned",
|
|
"reliability.finalization-and-replay-patches-are-deployed",
|
|
"pool.three-workers-are-ready",
|
|
"pool.workers-occupy-distinct-nodes",
|
|
"pool.workers-own-distinct-volumes",
|
|
"pool.coordinator-retains-sole-state-ownership",
|
|
"surfaces.kanban-activity-is-visible",
|
|
"surfaces.chat-telegram-topic-state-is-durable",
|
|
"gitops.kustomizations-reconcile-cleanly",
|
|
"routing.provider-effort-and-fallback-evidence",
|
|
"scopes.agent-profile-is-distinct",
|
|
],
|
|
)
|
|
def test_the_handoff_requirements_each_have_a_mandatory_check(identifier: str) -> None:
|
|
assert BY_ID[identifier].mandatory is True
|