2026-08-16 10:51:34 -03:00
|
|
|
"""Completion-gate tests for Hermes external Kanban workers."""
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
import importlib.util
|
|
|
|
|
import json
|
|
|
|
|
import sys
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
SCRIPT = (
|
|
|
|
|
Path(__file__).parents[2]
|
|
|
|
|
/ "services/hermes/scripts/cli_lane_goal.py"
|
|
|
|
|
)
|
|
|
|
|
SPEC = importlib.util.spec_from_file_location("cli_lane_goal_test", SCRIPT)
|
|
|
|
|
assert SPEC and SPEC.loader
|
|
|
|
|
goal = importlib.util.module_from_spec(SPEC)
|
|
|
|
|
sys.modules[SPEC.name] = goal
|
|
|
|
|
SPEC.loader.exec_module(goal)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _result(**overrides):
|
|
|
|
|
value = {
|
|
|
|
|
"status": "completed",
|
|
|
|
|
"summary": "All acceptance criteria passed and the branch was pushed.",
|
|
|
|
|
"changed_files": ["src/example.py"],
|
|
|
|
|
"tests_run": ["pytest -q: 12 passed"],
|
|
|
|
|
"artifacts": [],
|
|
|
|
|
"blockers": [],
|
|
|
|
|
}
|
|
|
|
|
value.update(overrides)
|
|
|
|
|
return value
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
"result",
|
|
|
|
|
[
|
|
|
|
|
_result(status="incomplete", summary="The full suite is still running."),
|
|
|
|
|
_result(summary="The broad rerun remains active before the final push."),
|
|
|
|
|
_result(tests_run=["Full suite — in progress (11m)"]),
|
|
|
|
|
_result(blockers=["remote head was not verified"]),
|
|
|
|
|
],
|
|
|
|
|
)
|
|
|
|
|
def test_unfinished_completion_evidence_is_rejected(result):
|
hermes: close the review-role gaps found reviewing PR 22
Repairs the blockers from the independent review of the previous head. The
role-aware verdict contract was correct but reachable only through one call
site and only on cards written with real newlines, so most real review cards
never used it.
* The role-blind call is no longer a weaker classifier that can reject before
the role-aware judge runs. Without a card there is no defensible
role-dependent judgement, so `unfinished_result_reason()` applies only the
card-independent checks. That closes the short-circuit at every call site,
including the one PR15 moves to `cli_lane_execution`, and it makes a
journalled terminal record accepted under one version of these semantics
re-validate under any other instead of being quarantined into a re-dispatch
of an already-accepted task.
* The lane resolves the role from the card and passes it, and records it in
the Kanban metadata. The verdict contract binds only where the lane can buy
another turn: in single-shot mode a rejection discards the worker's real
result, so review cards keep the relaxation without gaining any rejection
single-shot mode did not already have.
* Card scope expands the literal \n escapes the board stores in one-line
bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives
are honoured on the 6 of 78 live cards that carry no real newline, and the
read-only, verdict and mutation heuristics stop being cut apart by them.
* Card scope now ends at the first non-card H2 and at the runner's controller
evidence, which is emitted under its own heading. Goal-controller rejection
history can no longer sit inside the card, and an upstream heading rename
fails closed instead of admitting history into role resolution.
* The inference recognises the SHIP/BLOCK-shaped deliverables real cards
actually use: 21 of 78 live cards resolve to review, up from 10, with no
implementation card misclassified. Cards asking for a findings list rather
than a verdict deliberately stay on the model judge, since the verdict is
the review contract's only gate.
* A declared review role no longer outranks mutation evidence: a report that
changed files falls back to the implementation regime.
* Judge reasons go through the agent runtime's canonical redactor, extended
for the two shapes it deliberately passes through and this lane handles -
`scheme://user:secret@host` and a credential named in prose - while a 40-hex
commit SHA survives as evidence.
Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of
every live card on three boards with a hand-labelled expected role, the
upgrade and single-shot properties against the previous gate, the upstream
context-heading contract, and the end-to-end `execute_claim` shape that used
to burn every goal turn.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-17 17:22:43 +00:00
|
|
|
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE)
|
2026-08-16 10:51:34 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_completed_pending_state_test_name_is_not_a_false_positive():
|
|
|
|
|
result = _result(tests_run=["pytest tests/test_pending_build_state.py: 8 passed"])
|
|
|
|
|
|
hermes: close the review-role gaps found reviewing PR 22
Repairs the blockers from the independent review of the previous head. The
role-aware verdict contract was correct but reachable only through one call
site and only on cards written with real newlines, so most real review cards
never used it.
* The role-blind call is no longer a weaker classifier that can reject before
the role-aware judge runs. Without a card there is no defensible
role-dependent judgement, so `unfinished_result_reason()` applies only the
card-independent checks. That closes the short-circuit at every call site,
including the one PR15 moves to `cli_lane_execution`, and it makes a
journalled terminal record accepted under one version of these semantics
re-validate under any other instead of being quarantined into a re-dispatch
of an already-accepted task.
* The lane resolves the role from the card and passes it, and records it in
the Kanban metadata. The verdict contract binds only where the lane can buy
another turn: in single-shot mode a rejection discards the worker's real
result, so review cards keep the relaxation without gaining any rejection
single-shot mode did not already have.
* Card scope expands the literal \n escapes the board stores in one-line
bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives
are honoured on the 6 of 78 live cards that carry no real newline, and the
read-only, verdict and mutation heuristics stop being cut apart by them.
* Card scope now ends at the first non-card H2 and at the runner's controller
evidence, which is emitted under its own heading. Goal-controller rejection
history can no longer sit inside the card, and an upstream heading rename
fails closed instead of admitting history into role resolution.
* The inference recognises the SHIP/BLOCK-shaped deliverables real cards
actually use: 21 of 78 live cards resolve to review, up from 10, with no
implementation card misclassified. Cards asking for a findings list rather
than a verdict deliberately stay on the model judge, since the verdict is
the review contract's only gate.
* A declared review role no longer outranks mutation evidence: a report that
changed files falls back to the implementation regime.
* Judge reasons go through the agent runtime's canonical redactor, extended
for the two shapes it deliberately passes through and this lane handles -
`scheme://user:secret@host` and a credential named in prose - while a 40-hex
commit SHA survives as evidence.
Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of
every live card on three boards with a hand-labelled expected role, the
upgrade and single-shot properties against the previous gate, the upstream
context-heading contract, and the end-to-end `execute_claim` shape that used
to burn every goal turn.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-17 17:22:43 +00:00
|
|
|
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE) is None
|
2026-08-16 10:51:34 -03:00
|
|
|
|
|
|
|
|
|
2026-08-16 11:54:26 -03:00
|
|
|
def test_completed_review_findings_are_not_task_blockers():
|
|
|
|
|
result = _result(
|
|
|
|
|
summary="Review complete; the pull request is not merge-ready.",
|
|
|
|
|
findings=["A casting-table variant fails open."],
|
|
|
|
|
)
|
|
|
|
|
|
hermes: close the review-role gaps found reviewing PR 22
Repairs the blockers from the independent review of the previous head. The
role-aware verdict contract was correct but reachable only through one call
site and only on cards written with real newlines, so most real review cards
never used it.
* The role-blind call is no longer a weaker classifier that can reject before
the role-aware judge runs. Without a card there is no defensible
role-dependent judgement, so `unfinished_result_reason()` applies only the
card-independent checks. That closes the short-circuit at every call site,
including the one PR15 moves to `cli_lane_execution`, and it makes a
journalled terminal record accepted under one version of these semantics
re-validate under any other instead of being quarantined into a re-dispatch
of an already-accepted task.
* The lane resolves the role from the card and passes it, and records it in
the Kanban metadata. The verdict contract binds only where the lane can buy
another turn: in single-shot mode a rejection discards the worker's real
result, so review cards keep the relaxation without gaining any rejection
single-shot mode did not already have.
* Card scope expands the literal \n escapes the board stores in one-line
bodies, so explicit `Hermes-Task-Role` / `Hermes-Expected-Output` directives
are honoured on the 6 of 78 live cards that carry no real newline, and the
read-only, verdict and mutation heuristics stop being cut apart by them.
* Card scope now ends at the first non-card H2 and at the runner's controller
evidence, which is emitted under its own heading. Goal-controller rejection
history can no longer sit inside the card, and an upstream heading rename
fails closed instead of admitting history into role resolution.
* The inference recognises the SHIP/BLOCK-shaped deliverables real cards
actually use: 21 of 78 live cards resolve to review, up from 10, with no
implementation card misclassified. Cards asking for a findings list rather
than a verdict deliberately stay on the model judge, since the verdict is
the review contract's only gate.
* A declared review role no longer outranks mutation evidence: a report that
changed files falls back to the implementation regime.
* Judge reasons go through the agent runtime's canonical redactor, extended
for the two shapes it deliberately passes through and this lane handles -
`scheme://user:secret@host` and a credential named in prose - while a 40-hex
commit SHA survives as evidence.
Regressions cover the recovered t_dbdcd739 incident, a verbatim snapshot of
every live card on three boards with a hand-labelled expected role, the
upgrade and single-shot properties against the previous gate, the upstream
context-heading contract, and the end-to-end `execute_claim` shape that used
to burn every goal turn.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-17 17:22:43 +00:00
|
|
|
assert goal.unfinished_result_reason(result, role=goal.IMPLEMENTATION_ROLE) is None
|
2026-08-16 11:54:26 -03:00
|
|
|
|
|
|
|
|
|
2026-08-16 10:51:34 -03:00
|
|
|
class JudgeResponse:
|
|
|
|
|
def __init__(self, verdict: str, reason: str):
|
|
|
|
|
self.body = json.dumps(
|
|
|
|
|
{
|
|
|
|
|
"choices": [
|
|
|
|
|
{
|
|
|
|
|
"message": {
|
|
|
|
|
"content": json.dumps(
|
|
|
|
|
{"verdict": verdict, "reason": reason}
|
|
|
|
|
)
|
|
|
|
|
}
|
|
|
|
|
}
|
|
|
|
|
]
|
|
|
|
|
}
|
|
|
|
|
).encode()
|
|
|
|
|
|
|
|
|
|
def __enter__(self):
|
|
|
|
|
return self
|
|
|
|
|
|
|
|
|
|
def __exit__(self, *_args):
|
|
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
def read(self):
|
|
|
|
|
return self.body
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_goal_judge_uses_local_structured_verdict():
|
|
|
|
|
observed = {}
|
|
|
|
|
|
|
|
|
|
def request(req, timeout):
|
|
|
|
|
observed["payload"] = json.loads(req.data)
|
|
|
|
|
observed["timeout"] = timeout
|
|
|
|
|
return JudgeResponse("continue", "remote head evidence is missing")
|
|
|
|
|
|
|
|
|
|
accepted, reason = goal.judge_goal_completion(
|
|
|
|
|
"Run tests, push, and verify the remote head.",
|
|
|
|
|
_result(summary="Tests passed."),
|
|
|
|
|
open_request=request,
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert accepted is False
|
|
|
|
|
assert reason == "remote head evidence is missing"
|
|
|
|
|
assert observed["payload"]["model"] == "qwen2.5:14b-instruct-q4_0"
|
|
|
|
|
assert observed["payload"]["response_format"]["json_schema"]["strict"] is True
|
2026-08-16 13:29:10 -03:00
|
|
|
assert observed["timeout"] == 120
|
2026-08-16 10:51:34 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_goal_judge_fails_closed_on_invalid_response():
|
|
|
|
|
accepted, reason = goal.judge_goal_completion(
|
|
|
|
|
"Finish the task.",
|
|
|
|
|
_result(),
|
|
|
|
|
open_request=lambda *_args, **_kwargs: JudgeResponse("unknown", "bad"),
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
assert accepted is False
|
|
|
|
|
assert "local completion judge unavailable" in reason
|
2026-08-17 15:03:59 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_completed_report_with_corrupt_test_evidence_is_still_inspected():
|
|
|
|
|
"""A non-list tests_run field neither crashes nor blocks the summary gate."""
|
|
|
|
|
assert goal.unfinished_result_reason(_result(tests_run=None)) is None
|
|
|
|
|
assert goal.unfinished_result_reason(
|
|
|
|
|
_result(
|
|
|
|
|
tests_run={"suite": "still running"},
|
|
|
|
|
summary="The remaining work is still running.",
|
|
|
|
|
)
|
|
|
|
|
)
|