All checks were successful
Tests / Declarative: Post Actions passed: 1404
Ariadne opens a pull request when it can: the repository is mapped, the file is in the write allowlist, and the change is one anchored snippet the validator can check. When any of that fails the incident escalates with a diagnosis and nothing else - even though the model that wrote the diagnosis frequently knows exactly what the fix is. That knowledge was discarded at the moment it was most useful, because the cases where no patch is possible are exactly the cases a maintainer has to do by hand. A diagnosis may now carry up to three code suggestions: the file, what is wrong there, and the code to change it to. Rendered into the issue under a heading that says the change was not applied, because a code block in an issue reads like something that already happened unless it is told otherwise. Deliberately not a patch, and the difference is the safety story. A patch must survive the validator because Ariadne acts on it. A suggestion is read by a person who is already going to edit that file, so being wrong costs them a moment's thought rather than a bad commit - which is why suggestions may describe changes too large or too diffuse for the patcher to have attempted, and why nothing here is anchored, applied or pushed. Bounded at three. A diagnosis that suggests a dozen changes has stopped diagnosing and started rewriting, and an issue that long buries the reason a person was called in the first place. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
175 lines
5.8 KiB
Python
175 lines
5.8 KiB
Python
"""Tests for code suggestions carried by an escalated diagnosis."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
|
|
import pytest
|
|
|
|
from ariadne.services import hermes_autotriage_decision as decision_module
|
|
from ariadne.services import hermes_code_suggestion as module
|
|
from ariadne.services import hermes_incident_body as body
|
|
|
|
|
|
SUGGESTION = {
|
|
"path": "ariadne/utils/errors.py",
|
|
"explanation": "safe_error_detail drops the response body, so the assertion never sees it.",
|
|
"code": 'def safe_error_detail(exc):\n return f"http {exc.response.status_code}: {exc.response.text}"',
|
|
}
|
|
|
|
|
|
def _payload(**overrides):
|
|
payload = {
|
|
"incident_id": "ariadne/408",
|
|
"classification": "pytest_test_failure",
|
|
"confidence": 0.9,
|
|
"facts": [],
|
|
"inferences": [],
|
|
"first_failed_gate": "tests",
|
|
"requested_action": None,
|
|
"human_required": True,
|
|
"reason": "a repository test failure",
|
|
}
|
|
payload.update(overrides)
|
|
return payload
|
|
|
|
|
|
def _parse(payload):
|
|
return decision_module.parse_triage_response(json.dumps(payload), "ariadne/408")
|
|
|
|
|
|
def test_a_response_without_the_field_still_validates() -> None:
|
|
"""Responses written before the field existed must keep parsing."""
|
|
|
|
outcome = _parse(_payload())
|
|
|
|
assert outcome.valid
|
|
assert outcome.decision.code_suggestions == []
|
|
|
|
|
|
def test_suggestions_are_parsed() -> None:
|
|
outcome = _parse(_payload(code_suggestions=[dict(SUGGESTION)]))
|
|
|
|
suggestion = outcome.decision.code_suggestions[0]
|
|
assert suggestion.path == "ariadne/utils/errors.py"
|
|
assert "drops the response body" in suggestion.explanation
|
|
assert "def safe_error_detail" in suggestion.code
|
|
|
|
|
|
def test_a_null_field_is_accepted() -> None:
|
|
assert _parse(_payload(code_suggestions=None)).decision.code_suggestions == []
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
("suggestions", "expected"),
|
|
[
|
|
("nope", "must be a list or null"),
|
|
([dict(SUGGESTION)] * 4, "at most 3 suggestions"),
|
|
(["not-an-object"], "[0] must be an object"),
|
|
([{"path": "a"}], "[0] must have exactly path, explanation, code"),
|
|
([{**SUGGESTION, "code": " "}], "[0] code must be a non-empty string"),
|
|
([{**SUGGESTION, "path": 7}], "[0] path must be a non-empty string"),
|
|
],
|
|
)
|
|
def test_a_malformed_field_is_rejected(suggestions, expected) -> None:
|
|
outcome = _parse(_payload(code_suggestions=suggestions))
|
|
|
|
assert not outcome.valid
|
|
assert expected in outcome.reject_reason
|
|
|
|
|
|
def test_a_long_suggestion_is_clipped() -> None:
|
|
outcome = _parse(
|
|
_payload(code_suggestions=[{**SUGGESTION, "code": "x" * 5000, "explanation": "y " * 900}])
|
|
)
|
|
|
|
suggestion = outcome.decision.code_suggestions[0]
|
|
assert len(suggestion.code) <= 1700
|
|
assert len(suggestion.explanation) <= 700
|
|
|
|
|
|
def test_a_suggestion_never_becomes_an_action() -> None:
|
|
"""It is advice in an issue; the gates must not see it as anything else."""
|
|
|
|
outcome = _parse(_payload(code_suggestions=[dict(SUGGESTION)], human_required=False))
|
|
allowed, reason = decision_module.authorize_action(
|
|
outcome,
|
|
{
|
|
"allowed_actions": ["retry_transient_infra"],
|
|
"action_classifications": {"pytest_test_failure": "retry_transient_infra"},
|
|
"autoremediation_enabled": True,
|
|
"min_confidence": 0.5,
|
|
},
|
|
prior_action_count=0,
|
|
build_is_terminal_failure=True,
|
|
job_allowlisted=True,
|
|
evidence_has_signature=True,
|
|
)
|
|
|
|
assert not allowed
|
|
assert reason == "requested_action_missing"
|
|
|
|
|
|
def test_the_issue_section_says_the_change_was_not_applied() -> None:
|
|
"""A code block in an issue reads as a change that was made."""
|
|
|
|
section = module.issue_section(module.from_payload({"code_suggestions": [dict(SUGGESTION)]}))
|
|
|
|
assert "## Suggested fix (not applied)" in section
|
|
assert "not written, validated, or pushed anywhere" in section
|
|
assert "**`ariadne/utils/errors.py`**" in section
|
|
assert section.count("```") == 2
|
|
assert module.issue_section(None) == ""
|
|
assert module.issue_section([]) == ""
|
|
|
|
|
|
def test_a_fence_inside_a_suggestion_cannot_break_the_block() -> None:
|
|
section = module.issue_section(
|
|
module.from_payload({"code_suggestions": [{**SUGGESTION, "code": "a\n```\nb"}]})
|
|
)
|
|
|
|
assert section.count("```") == 2
|
|
|
|
|
|
def test_the_detail_recorded_for_the_audit_trail() -> None:
|
|
detail = module.as_detail(module.from_payload({"code_suggestions": [dict(SUGGESTION)]}))
|
|
|
|
assert detail[0]["path"] == "ariadne/utils/errors.py"
|
|
assert module.as_detail(None) == []
|
|
|
|
|
|
def test_from_payload_tolerates_junk() -> None:
|
|
assert module.from_payload({}) == []
|
|
assert module.from_payload({"code_suggestions": "nope"}) == []
|
|
assert module.from_payload({"code_suggestions": ["x"]}) == []
|
|
|
|
|
|
def test_the_issue_body_carries_the_suggestion_and_keeps_its_marker() -> None:
|
|
rendered = body.issue_body(
|
|
{
|
|
"incident_id": "ariadne/408",
|
|
"job": "ariadne",
|
|
"build_number": 408,
|
|
"classification": "pytest_test_failure",
|
|
"reason": "a repository test failure",
|
|
"run_id": "run-1",
|
|
"code_suggestions": module.from_payload({"code_suggestions": [dict(SUGGESTION)]}),
|
|
}
|
|
)
|
|
|
|
assert "## Suggested fix (not applied)" in rendered
|
|
assert "def safe_error_detail" in rendered
|
|
assert rendered.index("Suggested fix") < rendered.index("## Links")
|
|
assert rendered.rstrip().endswith("-->")
|
|
|
|
|
|
def test_the_prompt_asks_for_suggestions_without_promising_to_apply_them() -> None:
|
|
from ariadne.services import hermes_triage_prompt
|
|
|
|
prompt = hermes_triage_prompt.build_prompt("ariadne/408", "ariadne", {})
|
|
|
|
assert '"code_suggestions": <array or null>' in prompt
|
|
assert "at most three" in prompt
|
|
assert "nothing you put here is applied" in prompt
|
|
assert "prefer correcting the code under test" in prompt
|