jenkins 681b040885 hermes(hux): close Wave A review findings in events, memory, privacy and organization
F3 memory edits go through the same privacy shaping as proposals; F5 seq is
derived from the ledger tail so a crash between append and checkpoint never
duplicates; F7 transitions re-read under the lock and always write with the
loaded revision; F9 secret scrub on titles, passages, claims and notebooks
and forget blanks the conversation document; F13 no ghost conversations
from notices, idempotency under the lock, artifact titles searchable,
normalised paths.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
2026-08-24 00:48:20 -03:00

385 lines
19 KiB
Python

"""HUX-08 research model: sources, passages, citations and notebooks.
A source says where evidence came from, a passage is the exact excerpt, a
citation ties a claim in a message to passages with a support verdict, and a
notebook collects them per research question. The service never fetches a
``uri`` (SO-19, SO-29): URIs are stored metadata under an allowlisted scheme.
Every referenced id resolves under the caller's own subtree (SO-33); a foreign
or unknown id is ``404`` so ids cannot be probed. Duplicates are merged on a
server-computed ``dedupe_key`` and the original record is returned.
"""
from __future__ import annotations
import hashlib
from typing import Any
from urllib.parse import urlsplit, urlunsplit
from hux.artifacts import remember, replay
from hux import redaction
from hux.contracts import load_all, validate_record
from hux.errors import Conflict, Invalid, NotFound, TooLarge
from hux.http import Request, Response, Router, page
from hux.identity import Identity
from hux.store import TenantStore, check_id, new_id, now_iso
FAMILY = "research"
SOURCES, PASSAGES, CITATIONS, NOTEBOOKS = "sources", "passages", "citations", "notebooks"
SCHEMES = frozenset({"http", "https", "artifact", "memory", "file"})
SOURCE_KINDS = ("web", "document", "artifact", "memory", "tool_output", "dataset")
CLASSIFICATIONS = ("primary", "secondary", "unknown")
SUPPORT = ("supports", "partially_supports", "contradicts", "unverified")
NOTEBOOK_TRANSITIONS = {"open": frozenset({"answered", "abandoned"}), "answered": frozenset(), "abandoned": frozenset()}
MAX_SOURCES, MAX_PASSAGES = 10000, 20000
_SCHEMAS: dict[str, dict[str, Any]] = {}
def _schemas() -> dict[str, dict[str, Any]]:
if not _SCHEMAS:
_SCHEMAS.update(load_all())
return _SCHEMAS
def _check(record: dict[str, Any]) -> None:
problems = validate_record(record, _schemas())
if problems:
raise Invalid(f"{record.get('schema')} record is not valid", problems)
def _body(request: Request) -> dict[str, Any]:
if not isinstance(request.body, dict):
raise Invalid("body must be a JSON object")
return request.body
def _string(body: dict[str, Any], key: str, limit: int, required: bool = True) -> str | None:
value = body.get(key)
if value is None:
if required:
raise Invalid(f"{key} is required")
return None
if not isinstance(value, str) or not value.strip() or len(value) > limit:
raise Invalid(f"{key} must be a non-empty string of at most {limit} characters")
return value
def _clean(value: Any) -> Any:
"""Secret-scrub free text before it is stored (F9); ids and enums never carry secrets so they skip this."""
return redaction.scrub_value(value, [])
def _sha(*parts: str) -> str:
return "sha256:" + hashlib.sha256("\n".join(parts).encode("utf-8")).hexdigest()
def _lookup(store: TenantStore, ledger: str, key: str) -> str | None:
for row in store.read(FAMILY, ledger):
if row.get("key") == key:
return row["id"]
return None
def _resolve(store: TenantStore, family: str, record_id: Any) -> dict[str, Any]:
"""Load a record under the caller's subtree; malformed or missing is 404 (SO-33).
Sources, passages and citations are immutable, so the store's ``revision``
bookkeeping is stripped before they are served; notebooks keep theirs.
"""
try:
record = store.get(family, check_id(record_id))
except (Invalid, NotFound) as error:
raise NotFound(f"{family[:-1]} not found") from error
return record if family == NOTEBOOKS else {k: v for k, v in record.items() if k != "revision"}
def _id_list(body: dict[str, Any], key: str, maximum: int = 32) -> list[str]:
values = body.get(key, [])
if not isinstance(values, list) or not all(isinstance(v, str) for v in values):
raise Invalid(f"{key} must be a list of ids")
unique = list(dict.fromkeys(values))
if len(unique) > maximum:
raise TooLarge(f"{key} exceeds {maximum} ids")
return unique
def _emit(store: TenantStore, identity: Identity, conversation_id: str | None, kind: str, summary: str, detail: dict[str, Any], evidence: list[dict[str, Any]]) -> None:
if conversation_id is None:
return
try:
from hux.events import emit
except ModuleNotFoundError:
return
emit(store, identity, conversation_id, kind, summary, detail=detail, evidence=evidence)
# -- sources ---------------------------------------------------------------------
def normalise_uri(uri: str) -> str:
"""Canonical form used for dedupe: lowercase scheme and host, no fragment, no trailing slash."""
parts = urlsplit(uri.strip())
if parts.scheme.lower() not in SCHEMES:
raise Invalid("uri scheme must be one of http, https, artifact, memory, file")
path = parts.path.rstrip("/") or ("/" if parts.netloc else "")
return urlunsplit((parts.scheme.lower(), parts.netloc.lower(), path, parts.query, ""))
def create_source(request: Request) -> Response:
"""``POST /hux/v1/sources``: record where evidence came from; duplicates return the original."""
body = _body(request)
key = request.idempotency_key()
kind = _string(body, "kind", 40)
if kind not in SOURCE_KINDS:
raise Invalid("unknown source kind")
classification = body.get("classification", "unknown")
if classification not in CLASSIFICATIONS:
raise Invalid("unknown classification")
title = _clean(_string(body, "title", 300))
uri = _string(body, "uri", 2000, required=False)
dedupe = _sha("uri", normalise_uri(uri)) if uri is not None else _sha("title", kind, title.strip().lower())
stamp = now_iso()
with request.store.lock(FAMILY):
existing = replay(request.store, FAMILY, key) or _lookup(request.store, "source_keys", dedupe)
if existing is not None:
request.audit("research.source_create", existing, reason="deduplicated")
return Response(200, _resolve(request.store, SOURCES, existing), {"HUX-Replayed": "true"})
if request.store.count(SOURCES) >= MAX_SOURCES:
raise TooLarge("tenant has reached 10000 sources")
record: dict[str, Any] = {
"schema": "hux.source.v1", "id": new_id("src"), "kind": kind, "title": title,
"classification": classification, "retrieved_at": stamp, "dedupe_key": dedupe,
"provenance": {"surface": request.identity.surface, "actor": _actor(request.identity), "recorded_at": stamp},
}
if uri is not None:
record["uri"] = uri
for field, limit in (("publisher", 200), ("published_at", 40), ("content_hash", 71)):
if _string(body, field, limit, required=False) is not None:
record[field] = body[field]
conversation_id = body.get("conversation_id")
if conversation_id is not None:
record["provenance"]["conversation_id"] = check_id(conversation_id)
_check(record)
stored = record | {}
request.store.put(SOURCES, record)
request.store.append(FAMILY, "source_keys", {"key": dedupe, "id": stored["id"]})
remember(request.store, FAMILY, key, stored["id"])
request.audit("research.source_create", stored["id"])
return Response(201, stored)
def _actor(identity: Identity) -> dict[str, str]:
if identity.trust == "worker":
return {"type": "system", "id": "worker"}
return {"type": "user", "id": identity.subject}
def get_source(request: Request) -> Response:
"""``GET /hux/v1/sources/{id}``: one source record."""
record = _resolve(request.store, SOURCES, request.params["id"])
request.audit("research.source_get", record["id"])
return Response(200, record)
# -- passages --------------------------------------------------------------------
def create_passage(request: Request) -> Response:
"""``POST /hux/v1/passages``: an exact excerpt of a source; hash and dedupe key are server-computed."""
body = _body(request)
key = request.idempotency_key()
source = _resolve(request.store, SOURCES, body.get("source_id"))
text = _clean(_string(body, "text", 4000))
text_hash = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
dedupe = _sha("passage", source["id"], text_hash)
with request.store.lock(FAMILY):
existing = replay(request.store, FAMILY, key) or _lookup(request.store, "passage_keys", dedupe)
if existing is not None:
request.audit("research.passage_create", existing, reason="deduplicated")
return Response(200, _resolve(request.store, PASSAGES, existing), {"HUX-Replayed": "true"})
if request.store.count(PASSAGES) >= MAX_PASSAGES:
raise TooLarge("tenant has reached 20000 passages")
record: dict[str, Any] = {"schema": "hux.passage.v1", "id": new_id("psg"), "source_id": source["id"], "text": text, "hash": text_hash, "dedupe_key": dedupe}
if body.get("locator") is not None:
record["locator"] = body["locator"]
_check(record)
stored = record | {}
request.store.put(PASSAGES, record)
request.store.append(FAMILY, "passage_keys", {"key": dedupe, "id": stored["id"]})
remember(request.store, FAMILY, key, stored["id"])
request.audit("research.passage_create", stored["id"])
return Response(201, stored)
# -- citations -------------------------------------------------------------------
def _message_ledger(message_id: str) -> str:
return "msg_" + hashlib.sha256(message_id.encode("utf-8")).hexdigest()[:40]
def attach_citation(request: Request) -> Response:
"""``POST /hux/v1/messages/{id}/citations``: tie a claim to passages with a support verdict."""
body = _body(request)
key = request.idempotency_key()
message_id = request.params["id"]
if len(message_id) > 120:
raise Invalid("message id is too long")
claim = _clean(_string(body, "claim", 1000))
support = body.get("support", "unverified")
if support not in SUPPORT:
raise Invalid("unknown support verdict")
passage_ids = _id_list(body, "passage_ids")
if not passage_ids:
raise Invalid("at least one passage_id is required")
passages = [_resolve(request.store, PASSAGES, pid) for pid in passage_ids]
dedupe = _sha("citation", message_id, claim, *sorted(passage_ids))
with request.store.lock(FAMILY):
existing = replay(request.store, FAMILY, key) or _lookup(request.store, "citation_keys", dedupe)
if existing is not None:
request.audit("research.citation_attach", existing, reason="deduplicated")
return Response(200, _resolve(request.store, CITATIONS, existing), {"HUX-Replayed": "true"})
record: dict[str, Any] = {"schema": "hux.citation.v1", "id": new_id("cit"), "message_id": message_id, "claim": claim, "passage_ids": passage_ids, "support": support, "dedupe_key": dedupe}
if _string(body, "note", 500, required=False) is not None:
record["note"] = _clean(body["note"])
_check(record)
stored = record | {}
request.store.put(CITATIONS, record)
request.store.append(FAMILY, "citation_keys", {"key": dedupe, "id": stored["id"]})
request.store.append(FAMILY, _message_ledger(message_id), {"id": stored["id"]})
remember(request.store, FAMILY, key, stored["id"])
request.audit("research.citation_attach", stored["id"])
evidence = [{"kind": "passage", "id": p["id"], "hash": p["hash"]} for p in passages]
conversation_id = body.get("conversation_id")
_emit(request.store, request.identity, None if conversation_id is None else check_id(conversation_id), "citation.attached", f"Citation attached ({support})", {"message_id": message_id, "citation_id": stored["id"], "support": support}, evidence)
return Response(201, stored)
def list_citations(request: Request) -> Response:
"""``GET /hux/v1/messages/{id}/citations``: citations with passages and sources embedded."""
items = []
for row in request.store.read(FAMILY, _message_ledger(request.params["id"])):
citation = _resolve(request.store, CITATIONS, row["id"])
passages = [_resolve(request.store, PASSAGES, pid) for pid in citation["passage_ids"]]
sources = {p["source_id"]: _resolve(request.store, SOURCES, p["source_id"]) for p in passages}
items.append({"citation": citation, "passages": passages, "sources": list(sources.values())})
request.audit("research.citation_list", request.params["id"])
return page(items)
def validate_citations(store: TenantStore, citation_ids: list[str]) -> list[str]:
"""Integrity problems for a set of citations: dangling passages, passage/source mismatch, contradictions without a note."""
problems: list[str] = []
for citation_id in citation_ids:
try:
citation = _resolve(store, CITATIONS, citation_id)
except NotFound:
problems.append(f"{citation_id}: citation does not resolve")
continue
if citation["support"] == "contradicts" and not citation.get("note"):
problems.append(f"{citation_id}: contradicts without a note")
for passage_id in citation["passage_ids"]:
try:
passage = _resolve(store, PASSAGES, passage_id)
except NotFound:
problems.append(f"{citation_id}: passage {passage_id} does not resolve")
continue
if not store.exists(SOURCES, passage["source_id"]):
problems.append(f"{citation_id}: passage {passage_id} names missing source {passage['source_id']}")
return problems
# -- notebooks -------------------------------------------------------------------
def _text_list(body: dict[str, Any], key: str, current: list[str]) -> list[str]:
values = body.get(key)
if values is None:
return current
if not isinstance(values, list) or not all(isinstance(v, str) and v.strip() for v in values):
raise Invalid(f"{key} must be a list of non-empty strings")
return _clean(values)
def create_notebook(request: Request) -> Response:
"""``POST /hux/v1/notebooks``: open a research question for a conversation."""
body = _body(request)
key = request.idempotency_key()
with request.store.lock(FAMILY):
existing = replay(request.store, FAMILY, key)
if existing is not None:
request.audit("research.notebook_create", existing, reason="idempotent_replay")
return Response(200, request.store.get(NOTEBOOKS, existing), {"HUX-Replayed": "true"})
record: dict[str, Any] = {
"schema": "hux.research_notebook.v1", "id": new_id("nb"), "conversation_id": check_id(body.get("conversation_id")),
"question": _clean(_string(body, "question", 1000)), "status": "open", "source_ids": [], "passage_ids": [], "citation_ids": [],
"assumptions": _text_list(body, "assumptions", []), "unresolved_questions": _text_list(body, "unresolved_questions", []),
"updated_at": now_iso(), "notes": [], "revision": 1,
}
_check(record)
stored = request.store.put(NOTEBOOKS, record)
remember(request.store, FAMILY, key, stored["id"])
request.audit("research.notebook_create", stored["id"])
return Response(201, stored, {"ETag": "1"})
def get_notebook(request: Request) -> Response:
"""``GET /hux/v1/notebooks/{id}``: one notebook."""
record = _resolve(request.store, NOTEBOOKS, request.params["id"])
request.audit("research.notebook_get", record["id"])
return Response(200, record, {"ETag": str(record["revision"])})
def _merged_ids(store: TenantStore, family: str, current: list[str], body: dict[str, Any], key: str) -> list[str]:
added = _id_list(body, key, maximum=500)
for record_id in added:
_resolve(store, family, record_id)
return list(dict.fromkeys([*current, *added]))
def patch_notebook(request: Request) -> Response:
"""``PATCH /hux/v1/notebooks/{id}``: add references and notes, replace assumptions, move status (If-Match required)."""
body = _body(request)
expected = request.if_match()
if expected is None:
raise Invalid("If-Match is required")
with request.store.lock(FAMILY):
notebook = _resolve(request.store, NOTEBOOKS, request.params["id"])
if expected != notebook["revision"]:
raise Conflict(f"revision {expected} does not match current revision {notebook['revision']}", [str(notebook["revision"])])
stamp = now_iso()
updated = {
**notebook,
"source_ids": _merged_ids(request.store, SOURCES, notebook["source_ids"], body, "add_source_ids"),
"passage_ids": _merged_ids(request.store, PASSAGES, notebook["passage_ids"], body, "add_passage_ids"),
"citation_ids": _merged_ids(request.store, CITATIONS, notebook["citation_ids"], body, "add_citation_ids"),
"assumptions": _text_list(body, "assumptions", notebook["assumptions"]),
"unresolved_questions": _text_list(body, "unresolved_questions", notebook["unresolved_questions"]),
"updated_at": stamp,
}
notes = body.get("add_notes", [])
if not isinstance(notes, list) or not all(isinstance(n, dict) for n in notes):
raise Invalid("add_notes must be a list of objects")
for note in notes:
entry = {"at": stamp, "text": _clean(_string(note, "text", 2000))}
if note.get("source_id") is not None:
entry["source_id"] = _resolve(request.store, SOURCES, note["source_id"])["id"]
updated["notes"] = [*updated["notes"], entry]
status = body.get("status")
if status is not None and status != notebook["status"]:
if status not in NOTEBOOK_TRANSITIONS or status not in NOTEBOOK_TRANSITIONS[notebook["status"]]:
raise Conflict(f"notebook cannot move from {notebook['status']} to {status}")
updated["status"] = status
_check(updated)
stored = request.store.put(NOTEBOOKS, updated, expected_revision=expected)
request.audit("research.notebook_patch", stored["id"])
return Response(200, stored, {"ETag": str(stored["revision"])})
def register(router: Router) -> None:
"""Attach HUX-08 routes."""
card = "HUX-08"
router.add("POST", "/hux/v1/sources", card, "research.source_create", create_source)
router.add("GET", "/hux/v1/sources/{id}", card, "research.source_get", get_source)
router.add("POST", "/hux/v1/passages", card, "research.passage_create", create_passage)
router.add("POST", "/hux/v1/messages/{id}/citations", card, "research.citation_attach", attach_citation)
router.add("GET", "/hux/v1/messages/{id}/citations", card, "research.citation_list", list_citations)
router.add("POST", "/hux/v1/notebooks", card, "research.notebook_create", create_notebook)
router.add("GET", "/hux/v1/notebooks/{id}", card, "research.notebook_get", get_notebook)
router.add("PATCH", "/hux/v1/notebooks/{id}", card, "research.notebook_patch", patch_notebook)