F3 memory edits go through the same privacy shaping as proposals; F5 seq is derived from the ledger tail so a crash between append and checkpoint never duplicates; F7 transitions re-read under the lock and always write with the loaded revision; F9 secret scrub on titles, passages, claims and notebooks and forget blanks the conversation document; F13 no ghost conversations from notices, idempotency under the lock, artifact titles searchable, normalised paths. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01RNPhwu2bsaRNg3DETSAZoM
385 lines
19 KiB
Python
385 lines
19 KiB
Python
"""HUX-08 research model: sources, passages, citations and notebooks.
|
|
|
|
A source says where evidence came from, a passage is the exact excerpt, a
|
|
citation ties a claim in a message to passages with a support verdict, and a
|
|
notebook collects them per research question. The service never fetches a
|
|
``uri`` (SO-19, SO-29): URIs are stored metadata under an allowlisted scheme.
|
|
Every referenced id resolves under the caller's own subtree (SO-33); a foreign
|
|
or unknown id is ``404`` so ids cannot be probed. Duplicates are merged on a
|
|
server-computed ``dedupe_key`` and the original record is returned.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import hashlib
|
|
from typing import Any
|
|
from urllib.parse import urlsplit, urlunsplit
|
|
|
|
from hux.artifacts import remember, replay
|
|
from hux import redaction
|
|
from hux.contracts import load_all, validate_record
|
|
from hux.errors import Conflict, Invalid, NotFound, TooLarge
|
|
from hux.http import Request, Response, Router, page
|
|
from hux.identity import Identity
|
|
from hux.store import TenantStore, check_id, new_id, now_iso
|
|
|
|
FAMILY = "research"
|
|
SOURCES, PASSAGES, CITATIONS, NOTEBOOKS = "sources", "passages", "citations", "notebooks"
|
|
SCHEMES = frozenset({"http", "https", "artifact", "memory", "file"})
|
|
SOURCE_KINDS = ("web", "document", "artifact", "memory", "tool_output", "dataset")
|
|
CLASSIFICATIONS = ("primary", "secondary", "unknown")
|
|
SUPPORT = ("supports", "partially_supports", "contradicts", "unverified")
|
|
NOTEBOOK_TRANSITIONS = {"open": frozenset({"answered", "abandoned"}), "answered": frozenset(), "abandoned": frozenset()}
|
|
MAX_SOURCES, MAX_PASSAGES = 10000, 20000
|
|
_SCHEMAS: dict[str, dict[str, Any]] = {}
|
|
|
|
|
|
def _schemas() -> dict[str, dict[str, Any]]:
|
|
if not _SCHEMAS:
|
|
_SCHEMAS.update(load_all())
|
|
return _SCHEMAS
|
|
|
|
|
|
def _check(record: dict[str, Any]) -> None:
|
|
problems = validate_record(record, _schemas())
|
|
if problems:
|
|
raise Invalid(f"{record.get('schema')} record is not valid", problems)
|
|
|
|
|
|
def _body(request: Request) -> dict[str, Any]:
|
|
if not isinstance(request.body, dict):
|
|
raise Invalid("body must be a JSON object")
|
|
return request.body
|
|
|
|
|
|
def _string(body: dict[str, Any], key: str, limit: int, required: bool = True) -> str | None:
|
|
value = body.get(key)
|
|
if value is None:
|
|
if required:
|
|
raise Invalid(f"{key} is required")
|
|
return None
|
|
if not isinstance(value, str) or not value.strip() or len(value) > limit:
|
|
raise Invalid(f"{key} must be a non-empty string of at most {limit} characters")
|
|
return value
|
|
|
|
|
|
def _clean(value: Any) -> Any:
|
|
"""Secret-scrub free text before it is stored (F9); ids and enums never carry secrets so they skip this."""
|
|
return redaction.scrub_value(value, [])
|
|
|
|
|
|
def _sha(*parts: str) -> str:
|
|
return "sha256:" + hashlib.sha256("\n".join(parts).encode("utf-8")).hexdigest()
|
|
|
|
|
|
def _lookup(store: TenantStore, ledger: str, key: str) -> str | None:
|
|
for row in store.read(FAMILY, ledger):
|
|
if row.get("key") == key:
|
|
return row["id"]
|
|
return None
|
|
|
|
|
|
def _resolve(store: TenantStore, family: str, record_id: Any) -> dict[str, Any]:
|
|
"""Load a record under the caller's subtree; malformed or missing is 404 (SO-33).
|
|
|
|
Sources, passages and citations are immutable, so the store's ``revision``
|
|
bookkeeping is stripped before they are served; notebooks keep theirs.
|
|
"""
|
|
try:
|
|
record = store.get(family, check_id(record_id))
|
|
except (Invalid, NotFound) as error:
|
|
raise NotFound(f"{family[:-1]} not found") from error
|
|
return record if family == NOTEBOOKS else {k: v for k, v in record.items() if k != "revision"}
|
|
|
|
|
|
def _id_list(body: dict[str, Any], key: str, maximum: int = 32) -> list[str]:
|
|
values = body.get(key, [])
|
|
if not isinstance(values, list) or not all(isinstance(v, str) for v in values):
|
|
raise Invalid(f"{key} must be a list of ids")
|
|
unique = list(dict.fromkeys(values))
|
|
if len(unique) > maximum:
|
|
raise TooLarge(f"{key} exceeds {maximum} ids")
|
|
return unique
|
|
|
|
|
|
def _emit(store: TenantStore, identity: Identity, conversation_id: str | None, kind: str, summary: str, detail: dict[str, Any], evidence: list[dict[str, Any]]) -> None:
|
|
if conversation_id is None:
|
|
return
|
|
try:
|
|
from hux.events import emit
|
|
except ModuleNotFoundError:
|
|
return
|
|
emit(store, identity, conversation_id, kind, summary, detail=detail, evidence=evidence)
|
|
|
|
|
|
# -- sources ---------------------------------------------------------------------
|
|
|
|
def normalise_uri(uri: str) -> str:
|
|
"""Canonical form used for dedupe: lowercase scheme and host, no fragment, no trailing slash."""
|
|
parts = urlsplit(uri.strip())
|
|
if parts.scheme.lower() not in SCHEMES:
|
|
raise Invalid("uri scheme must be one of http, https, artifact, memory, file")
|
|
path = parts.path.rstrip("/") or ("/" if parts.netloc else "")
|
|
return urlunsplit((parts.scheme.lower(), parts.netloc.lower(), path, parts.query, ""))
|
|
|
|
|
|
def create_source(request: Request) -> Response:
|
|
"""``POST /hux/v1/sources``: record where evidence came from; duplicates return the original."""
|
|
body = _body(request)
|
|
key = request.idempotency_key()
|
|
kind = _string(body, "kind", 40)
|
|
if kind not in SOURCE_KINDS:
|
|
raise Invalid("unknown source kind")
|
|
classification = body.get("classification", "unknown")
|
|
if classification not in CLASSIFICATIONS:
|
|
raise Invalid("unknown classification")
|
|
title = _clean(_string(body, "title", 300))
|
|
uri = _string(body, "uri", 2000, required=False)
|
|
dedupe = _sha("uri", normalise_uri(uri)) if uri is not None else _sha("title", kind, title.strip().lower())
|
|
stamp = now_iso()
|
|
with request.store.lock(FAMILY):
|
|
existing = replay(request.store, FAMILY, key) or _lookup(request.store, "source_keys", dedupe)
|
|
if existing is not None:
|
|
request.audit("research.source_create", existing, reason="deduplicated")
|
|
return Response(200, _resolve(request.store, SOURCES, existing), {"HUX-Replayed": "true"})
|
|
if request.store.count(SOURCES) >= MAX_SOURCES:
|
|
raise TooLarge("tenant has reached 10000 sources")
|
|
record: dict[str, Any] = {
|
|
"schema": "hux.source.v1", "id": new_id("src"), "kind": kind, "title": title,
|
|
"classification": classification, "retrieved_at": stamp, "dedupe_key": dedupe,
|
|
"provenance": {"surface": request.identity.surface, "actor": _actor(request.identity), "recorded_at": stamp},
|
|
}
|
|
if uri is not None:
|
|
record["uri"] = uri
|
|
for field, limit in (("publisher", 200), ("published_at", 40), ("content_hash", 71)):
|
|
if _string(body, field, limit, required=False) is not None:
|
|
record[field] = body[field]
|
|
conversation_id = body.get("conversation_id")
|
|
if conversation_id is not None:
|
|
record["provenance"]["conversation_id"] = check_id(conversation_id)
|
|
_check(record)
|
|
stored = record | {}
|
|
request.store.put(SOURCES, record)
|
|
request.store.append(FAMILY, "source_keys", {"key": dedupe, "id": stored["id"]})
|
|
remember(request.store, FAMILY, key, stored["id"])
|
|
request.audit("research.source_create", stored["id"])
|
|
return Response(201, stored)
|
|
|
|
|
|
def _actor(identity: Identity) -> dict[str, str]:
|
|
if identity.trust == "worker":
|
|
return {"type": "system", "id": "worker"}
|
|
return {"type": "user", "id": identity.subject}
|
|
|
|
|
|
def get_source(request: Request) -> Response:
|
|
"""``GET /hux/v1/sources/{id}``: one source record."""
|
|
record = _resolve(request.store, SOURCES, request.params["id"])
|
|
request.audit("research.source_get", record["id"])
|
|
return Response(200, record)
|
|
|
|
|
|
# -- passages --------------------------------------------------------------------
|
|
|
|
def create_passage(request: Request) -> Response:
|
|
"""``POST /hux/v1/passages``: an exact excerpt of a source; hash and dedupe key are server-computed."""
|
|
body = _body(request)
|
|
key = request.idempotency_key()
|
|
source = _resolve(request.store, SOURCES, body.get("source_id"))
|
|
text = _clean(_string(body, "text", 4000))
|
|
text_hash = "sha256:" + hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
dedupe = _sha("passage", source["id"], text_hash)
|
|
with request.store.lock(FAMILY):
|
|
existing = replay(request.store, FAMILY, key) or _lookup(request.store, "passage_keys", dedupe)
|
|
if existing is not None:
|
|
request.audit("research.passage_create", existing, reason="deduplicated")
|
|
return Response(200, _resolve(request.store, PASSAGES, existing), {"HUX-Replayed": "true"})
|
|
if request.store.count(PASSAGES) >= MAX_PASSAGES:
|
|
raise TooLarge("tenant has reached 20000 passages")
|
|
record: dict[str, Any] = {"schema": "hux.passage.v1", "id": new_id("psg"), "source_id": source["id"], "text": text, "hash": text_hash, "dedupe_key": dedupe}
|
|
if body.get("locator") is not None:
|
|
record["locator"] = body["locator"]
|
|
_check(record)
|
|
stored = record | {}
|
|
request.store.put(PASSAGES, record)
|
|
request.store.append(FAMILY, "passage_keys", {"key": dedupe, "id": stored["id"]})
|
|
remember(request.store, FAMILY, key, stored["id"])
|
|
request.audit("research.passage_create", stored["id"])
|
|
return Response(201, stored)
|
|
|
|
|
|
# -- citations -------------------------------------------------------------------
|
|
|
|
def _message_ledger(message_id: str) -> str:
|
|
return "msg_" + hashlib.sha256(message_id.encode("utf-8")).hexdigest()[:40]
|
|
|
|
|
|
def attach_citation(request: Request) -> Response:
|
|
"""``POST /hux/v1/messages/{id}/citations``: tie a claim to passages with a support verdict."""
|
|
body = _body(request)
|
|
key = request.idempotency_key()
|
|
message_id = request.params["id"]
|
|
if len(message_id) > 120:
|
|
raise Invalid("message id is too long")
|
|
claim = _clean(_string(body, "claim", 1000))
|
|
support = body.get("support", "unverified")
|
|
if support not in SUPPORT:
|
|
raise Invalid("unknown support verdict")
|
|
passage_ids = _id_list(body, "passage_ids")
|
|
if not passage_ids:
|
|
raise Invalid("at least one passage_id is required")
|
|
passages = [_resolve(request.store, PASSAGES, pid) for pid in passage_ids]
|
|
dedupe = _sha("citation", message_id, claim, *sorted(passage_ids))
|
|
with request.store.lock(FAMILY):
|
|
existing = replay(request.store, FAMILY, key) or _lookup(request.store, "citation_keys", dedupe)
|
|
if existing is not None:
|
|
request.audit("research.citation_attach", existing, reason="deduplicated")
|
|
return Response(200, _resolve(request.store, CITATIONS, existing), {"HUX-Replayed": "true"})
|
|
record: dict[str, Any] = {"schema": "hux.citation.v1", "id": new_id("cit"), "message_id": message_id, "claim": claim, "passage_ids": passage_ids, "support": support, "dedupe_key": dedupe}
|
|
if _string(body, "note", 500, required=False) is not None:
|
|
record["note"] = _clean(body["note"])
|
|
_check(record)
|
|
stored = record | {}
|
|
request.store.put(CITATIONS, record)
|
|
request.store.append(FAMILY, "citation_keys", {"key": dedupe, "id": stored["id"]})
|
|
request.store.append(FAMILY, _message_ledger(message_id), {"id": stored["id"]})
|
|
remember(request.store, FAMILY, key, stored["id"])
|
|
request.audit("research.citation_attach", stored["id"])
|
|
evidence = [{"kind": "passage", "id": p["id"], "hash": p["hash"]} for p in passages]
|
|
conversation_id = body.get("conversation_id")
|
|
_emit(request.store, request.identity, None if conversation_id is None else check_id(conversation_id), "citation.attached", f"Citation attached ({support})", {"message_id": message_id, "citation_id": stored["id"], "support": support}, evidence)
|
|
return Response(201, stored)
|
|
|
|
|
|
def list_citations(request: Request) -> Response:
|
|
"""``GET /hux/v1/messages/{id}/citations``: citations with passages and sources embedded."""
|
|
items = []
|
|
for row in request.store.read(FAMILY, _message_ledger(request.params["id"])):
|
|
citation = _resolve(request.store, CITATIONS, row["id"])
|
|
passages = [_resolve(request.store, PASSAGES, pid) for pid in citation["passage_ids"]]
|
|
sources = {p["source_id"]: _resolve(request.store, SOURCES, p["source_id"]) for p in passages}
|
|
items.append({"citation": citation, "passages": passages, "sources": list(sources.values())})
|
|
request.audit("research.citation_list", request.params["id"])
|
|
return page(items)
|
|
|
|
|
|
def validate_citations(store: TenantStore, citation_ids: list[str]) -> list[str]:
|
|
"""Integrity problems for a set of citations: dangling passages, passage/source mismatch, contradictions without a note."""
|
|
problems: list[str] = []
|
|
for citation_id in citation_ids:
|
|
try:
|
|
citation = _resolve(store, CITATIONS, citation_id)
|
|
except NotFound:
|
|
problems.append(f"{citation_id}: citation does not resolve")
|
|
continue
|
|
if citation["support"] == "contradicts" and not citation.get("note"):
|
|
problems.append(f"{citation_id}: contradicts without a note")
|
|
for passage_id in citation["passage_ids"]:
|
|
try:
|
|
passage = _resolve(store, PASSAGES, passage_id)
|
|
except NotFound:
|
|
problems.append(f"{citation_id}: passage {passage_id} does not resolve")
|
|
continue
|
|
if not store.exists(SOURCES, passage["source_id"]):
|
|
problems.append(f"{citation_id}: passage {passage_id} names missing source {passage['source_id']}")
|
|
return problems
|
|
|
|
|
|
# -- notebooks -------------------------------------------------------------------
|
|
|
|
def _text_list(body: dict[str, Any], key: str, current: list[str]) -> list[str]:
|
|
values = body.get(key)
|
|
if values is None:
|
|
return current
|
|
if not isinstance(values, list) or not all(isinstance(v, str) and v.strip() for v in values):
|
|
raise Invalid(f"{key} must be a list of non-empty strings")
|
|
return _clean(values)
|
|
|
|
|
|
def create_notebook(request: Request) -> Response:
|
|
"""``POST /hux/v1/notebooks``: open a research question for a conversation."""
|
|
body = _body(request)
|
|
key = request.idempotency_key()
|
|
with request.store.lock(FAMILY):
|
|
existing = replay(request.store, FAMILY, key)
|
|
if existing is not None:
|
|
request.audit("research.notebook_create", existing, reason="idempotent_replay")
|
|
return Response(200, request.store.get(NOTEBOOKS, existing), {"HUX-Replayed": "true"})
|
|
record: dict[str, Any] = {
|
|
"schema": "hux.research_notebook.v1", "id": new_id("nb"), "conversation_id": check_id(body.get("conversation_id")),
|
|
"question": _clean(_string(body, "question", 1000)), "status": "open", "source_ids": [], "passage_ids": [], "citation_ids": [],
|
|
"assumptions": _text_list(body, "assumptions", []), "unresolved_questions": _text_list(body, "unresolved_questions", []),
|
|
"updated_at": now_iso(), "notes": [], "revision": 1,
|
|
}
|
|
_check(record)
|
|
stored = request.store.put(NOTEBOOKS, record)
|
|
remember(request.store, FAMILY, key, stored["id"])
|
|
request.audit("research.notebook_create", stored["id"])
|
|
return Response(201, stored, {"ETag": "1"})
|
|
|
|
|
|
def get_notebook(request: Request) -> Response:
|
|
"""``GET /hux/v1/notebooks/{id}``: one notebook."""
|
|
record = _resolve(request.store, NOTEBOOKS, request.params["id"])
|
|
request.audit("research.notebook_get", record["id"])
|
|
return Response(200, record, {"ETag": str(record["revision"])})
|
|
|
|
|
|
def _merged_ids(store: TenantStore, family: str, current: list[str], body: dict[str, Any], key: str) -> list[str]:
|
|
added = _id_list(body, key, maximum=500)
|
|
for record_id in added:
|
|
_resolve(store, family, record_id)
|
|
return list(dict.fromkeys([*current, *added]))
|
|
|
|
|
|
def patch_notebook(request: Request) -> Response:
|
|
"""``PATCH /hux/v1/notebooks/{id}``: add references and notes, replace assumptions, move status (If-Match required)."""
|
|
body = _body(request)
|
|
expected = request.if_match()
|
|
if expected is None:
|
|
raise Invalid("If-Match is required")
|
|
with request.store.lock(FAMILY):
|
|
notebook = _resolve(request.store, NOTEBOOKS, request.params["id"])
|
|
if expected != notebook["revision"]:
|
|
raise Conflict(f"revision {expected} does not match current revision {notebook['revision']}", [str(notebook["revision"])])
|
|
stamp = now_iso()
|
|
updated = {
|
|
**notebook,
|
|
"source_ids": _merged_ids(request.store, SOURCES, notebook["source_ids"], body, "add_source_ids"),
|
|
"passage_ids": _merged_ids(request.store, PASSAGES, notebook["passage_ids"], body, "add_passage_ids"),
|
|
"citation_ids": _merged_ids(request.store, CITATIONS, notebook["citation_ids"], body, "add_citation_ids"),
|
|
"assumptions": _text_list(body, "assumptions", notebook["assumptions"]),
|
|
"unresolved_questions": _text_list(body, "unresolved_questions", notebook["unresolved_questions"]),
|
|
"updated_at": stamp,
|
|
}
|
|
notes = body.get("add_notes", [])
|
|
if not isinstance(notes, list) or not all(isinstance(n, dict) for n in notes):
|
|
raise Invalid("add_notes must be a list of objects")
|
|
for note in notes:
|
|
entry = {"at": stamp, "text": _clean(_string(note, "text", 2000))}
|
|
if note.get("source_id") is not None:
|
|
entry["source_id"] = _resolve(request.store, SOURCES, note["source_id"])["id"]
|
|
updated["notes"] = [*updated["notes"], entry]
|
|
status = body.get("status")
|
|
if status is not None and status != notebook["status"]:
|
|
if status not in NOTEBOOK_TRANSITIONS or status not in NOTEBOOK_TRANSITIONS[notebook["status"]]:
|
|
raise Conflict(f"notebook cannot move from {notebook['status']} to {status}")
|
|
updated["status"] = status
|
|
_check(updated)
|
|
stored = request.store.put(NOTEBOOKS, updated, expected_revision=expected)
|
|
request.audit("research.notebook_patch", stored["id"])
|
|
return Response(200, stored, {"ETag": str(stored["revision"])})
|
|
|
|
|
|
def register(router: Router) -> None:
|
|
"""Attach HUX-08 routes."""
|
|
card = "HUX-08"
|
|
router.add("POST", "/hux/v1/sources", card, "research.source_create", create_source)
|
|
router.add("GET", "/hux/v1/sources/{id}", card, "research.source_get", get_source)
|
|
router.add("POST", "/hux/v1/passages", card, "research.passage_create", create_passage)
|
|
router.add("POST", "/hux/v1/messages/{id}/citations", card, "research.citation_attach", attach_citation)
|
|
router.add("GET", "/hux/v1/messages/{id}/citations", card, "research.citation_list", list_citations)
|
|
router.add("POST", "/hux/v1/notebooks", card, "research.notebook_create", create_notebook)
|
|
router.add("GET", "/hux/v1/notebooks/{id}", card, "research.notebook_get", get_notebook)
|
|
router.add("PATCH", "/hux/v1/notebooks/{id}", card, "research.notebook_patch", patch_notebook)
|