atlas-iac/testing/tests/test_hermes_cli_lane_goal.py

132 lines
3.9 KiB
Python
Raw Permalink Normal View History

"""Completion-gate tests for Hermes external Kanban workers."""
from __future__ import annotations
import importlib.util
import json
import sys
from pathlib import Path
import pytest
SCRIPT = (
Path(__file__).parents[2]
/ "services/hermes/scripts/cli_lane_goal.py"
)
SPEC = importlib.util.spec_from_file_location("cli_lane_goal_test", SCRIPT)
assert SPEC and SPEC.loader
goal = importlib.util.module_from_spec(SPEC)
sys.modules[SPEC.name] = goal
SPEC.loader.exec_module(goal)
def _result(**overrides):
value = {
"status": "completed",
"summary": "All acceptance criteria passed and the branch was pushed.",
"changed_files": ["src/example.py"],
"tests_run": ["pytest -q: 12 passed"],
"artifacts": [],
"blockers": [],
}
value.update(overrides)
return value
@pytest.mark.parametrize(
"result",
[
_result(status="incomplete", summary="The full suite is still running."),
_result(summary="The broad rerun remains active before the final push."),
_result(tests_run=["Full suite — in progress (11m)"]),
_result(blockers=["remote head was not verified"]),
],
)
def test_unfinished_completion_evidence_is_rejected(result):
hermes: close the review-role gaps found reviewing PR 22 Repairs the blockers from the independent review of the previous head. The role-aware verdict contract was correct but reachable only through one call site and only on cards written with real newlines, so most real review cards never used it. * The role-blind call is no longer a weaker classifier that can reject before the role-aware judge runs. Without a card there is no defensible role-dependent judgement, so `unfinished_result_reason()` applies only the card-independent checks. That closes the short-circuit at every call site, including the one PR15 moves to `cli_lane_execution`, and it makes a journalled terminal record accepted under one version of these semantics re-validate under any other instead of being quarantined into a re-dispatch of an already-accepted task. * The lane resolves the role from the card and passes it, and records it in the Kanban metadata. The verdict contract binds only where the lane can buy another turn: in single-shot mode a rejection discards the worker's real result, so review cards keep the relaxation without gaining any rejection single-shot mode did not already have. * Card scope expands the literal \n escapes the board stores in one-line bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives are honoured on the 6 of 78 live cards that carry no real newline, and the read-only, verdict and mutation heuristics stop being cut apart by them. * Card scope now ends at the first non-card H2 and at the runner's controller evidence, which is emitted under its own heading. Goal-controller rejection history can no longer sit inside the card, and an upstream heading rename fails closed instead of admitting history into role resolution. * The inference recognises the SHIP/BLOCK-shaped deliverables real cards actually use: 21 of 78 live cards resolve to review, up from 10, with no implementation card misclassified. Cards asking for a findings list rather than a verdict deliberately stay on the model judge, since the verdict is the review contract's only gate. * A declared review role no longer outranks mutation evidence: a report that changed files falls back to the implementation regime. * Judge reasons go through the agent runtime's canonical redactor, extended for the two shapes it deliberately passes through and this lane handles - `scheme://user:secret@host` and a credential named in prose - while a 40-hex commit SHA survives as evidence. Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of every live card on three boards with a hand-labelled expected role, the upgrade and single-shot properties against the previous gate, the upstream context-heading contract, and the end-to-end `execute_claim` shape that used to burn every goal turn. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-17 17:22:43 +00:00
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE)
def test_completed_pending_state_test_name_is_not_a_false_positive():
result = _result(tests_run=["pytest tests/test_pending_build_state.py: 8 passed"])
hermes: close the review-role gaps found reviewing PR 22 Repairs the blockers from the independent review of the previous head. The role-aware verdict contract was correct but reachable only through one call site and only on cards written with real newlines, so most real review cards never used it. * The role-blind call is no longer a weaker classifier that can reject before the role-aware judge runs. Without a card there is no defensible role-dependent judgement, so `unfinished_result_reason()` applies only the card-independent checks. That closes the short-circuit at every call site, including the one PR15 moves to `cli_lane_execution`, and it makes a journalled terminal record accepted under one version of these semantics re-validate under any other instead of being quarantined into a re-dispatch of an already-accepted task. * The lane resolves the role from the card and passes it, and records it in the Kanban metadata. The verdict contract binds only where the lane can buy another turn: in single-shot mode a rejection discards the worker's real result, so review cards keep the relaxation without gaining any rejection single-shot mode did not already have. * Card scope expands the literal \n escapes the board stores in one-line bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives are honoured on the 6 of 78 live cards that carry no real newline, and the read-only, verdict and mutation heuristics stop being cut apart by them. * Card scope now ends at the first non-card H2 and at the runner's controller evidence, which is emitted under its own heading. Goal-controller rejection history can no longer sit inside the card, and an upstream heading rename fails closed instead of admitting history into role resolution. * The inference recognises the SHIP/BLOCK-shaped deliverables real cards actually use: 21 of 78 live cards resolve to review, up from 10, with no implementation card misclassified. Cards asking for a findings list rather than a verdict deliberately stay on the model judge, since the verdict is the review contract's only gate. * A declared review role no longer outranks mutation evidence: a report that changed files falls back to the implementation regime. * Judge reasons go through the agent runtime's canonical redactor, extended for the two shapes it deliberately passes through and this lane handles - `scheme://user:secret@host` and a credential named in prose - while a 40-hex commit SHA survives as evidence. Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of every live card on three boards with a hand-labelled expected role, the upgrade and single-shot properties against the previous gate, the upstream context-heading contract, and the end-to-end `execute_claim` shape that used to burn every goal turn. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-17 17:22:43 +00:00
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE) is None
def test_completed_review_findings_are_not_task_blockers():
result = _result(
summary="Review complete; the pull request is not merge-ready.",
findings=["A casting-table variant fails open."],
)
hermes: close the review-role gaps found reviewing PR 22 Repairs the blockers from the independent review of the previous head. The role-aware verdict contract was correct but reachable only through one call site and only on cards written with real newlines, so most real review cards never used it. * The role-blind call is no longer a weaker classifier that can reject before the role-aware judge runs. Without a card there is no defensible role-dependent judgement, so `unfinished_result_reason()` applies only the card-independent checks. That closes the short-circuit at every call site, including the one PR15 moves to `cli_lane_execution`, and it makes a journalled terminal record accepted under one version of these semantics re-validate under any other instead of being quarantined into a re-dispatch of an already-accepted task. * The lane resolves the role from the card and passes it, and records it in the Kanban metadata. The verdict contract binds only where the lane can buy another turn: in single-shot mode a rejection discards the worker's real result, so review cards keep the relaxation without gaining any rejection single-shot mode did not already have. * Card scope expands the literal \n escapes the board stores in one-line bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives are honoured on the 6 of 78 live cards that carry no real newline, and the read-only, verdict and mutation heuristics stop being cut apart by them. * Card scope now ends at the first non-card H2 and at the runner's controller evidence, which is emitted under its own heading. Goal-controller rejection history can no longer sit inside the card, and an upstream heading rename fails closed instead of admitting history into role resolution. * The inference recognises the SHIP/BLOCK-shaped deliverables real cards actually use: 21 of 78 live cards resolve to review, up from 10, with no implementation card misclassified. Cards asking for a findings list rather than a verdict deliberately stay on the model judge, since the verdict is the review contract's only gate. * A declared review role no longer outranks mutation evidence: a report that changed files falls back to the implementation regime. * Judge reasons go through the agent runtime's canonical redactor, extended for the two shapes it deliberately passes through and this lane handles - `scheme://user:secret@host` and a credential named in prose - while a 40-hex commit SHA survives as evidence. Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of every live card on three boards with a hand-labelled expected role, the upgrade and single-shot properties against the previous gate, the upstream context-heading contract, and the end-to-end `execute_claim` shape that used to burn every goal turn. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-17 17:22:43 +00:00
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE) is None
class JudgeResponse:
def __init__(self, verdict: str, reason: str):
self.body = json.dumps(
{
"choices": [
{
"message": {
"content": json.dumps(
{"verdict": verdict, "reason": reason}
)
}
}
]
}
).encode()
def __enter__(self):
return self
def __exit__(self, *_args):
return False
def read(self):
return self.body
def test_goal_judge_uses_local_structured_verdict():
observed = {}
def request(req, timeout):
observed["payload"] = json.loads(req.data)
observed["timeout"] = timeout
return JudgeResponse("continue", "remote head evidence is missing")
accepted, reason = goal.judge_goal_completion(
"Run tests, push, and verify the remote head.",
_result(summary="Tests passed."),
open_request=request,
)
assert accepted is False
assert reason == "remote head evidence is missing"
assert observed["payload"]["model"] == "qwen2.5:14b-instruct-q4_0"
assert observed["payload"]["response_format"]["json_schema"]["strict"] is True
2026-08-16 13:29:10 -03:00
assert observed["timeout"] == 120
def test_goal_judge_fails_closed_on_invalid_response():
accepted, reason = goal.judge_goal_completion(
"Finish the task.",
_result(),
open_request=lambda *_args, **_kwargs: JudgeResponse("unknown", "bad"),
)
assert accepted is False
assert "local completion judge unavailable" in reason
def test_completed_report_with_corrupt_test_evidence_is_still_inspected():
"""A non-list tests_run field neither crashes nor blocks the summary gate."""
assert goal.unfinished_result_reason(_result(tests_run=None)) is None
assert goal.unfinished_result_reason(
_result(
tests_run={"suite": "still running"},
summary="The remaining work is still running.",
)
)