"""Credential-free extraction of bounded public HTML and text pages.""" from __future__ import annotations from html.parser import HTMLParser from typing import Any from urllib.parse import urljoin import httpx from agent.web_search_provider import WebSearchProvider from tools.url_safety import is_safe_url from tools.website_policy import check_website_access MAX_RESPONSE_BYTES = 2 * 1024 * 1024 MAX_REDIRECTS = 5 class _VisibleTextParser(HTMLParser): """Collect readable text while discarding scripts, styles, and chrome.""" _ignored = {"script", "style", "noscript", "svg", "nav", "footer"} def __init__(self) -> None: super().__init__(convert_charrefs=True) self._ignore_depth = 0 self._title_depth = 0 self.title: list[str] = [] self.text: list[str] = [] def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: tag = tag.lower() if tag in self._ignored: self._ignore_depth += 1 if tag == "title": self._title_depth += 1 if not self._ignore_depth and tag in {"p", "div", "section", "article", "li", "br", "h1", "h2", "h3", "h4"}: self.text.append("\n") def handle_endtag(self, tag: str) -> None: tag = tag.lower() if tag == "title" and self._title_depth: self._title_depth -= 1 if tag in self._ignored and self._ignore_depth: self._ignore_depth -= 1 if not self._ignore_depth and tag in {"p", "div", "section", "article", "li", "h1", "h2", "h3", "h4"}: self.text.append("\n") def handle_data(self, data: str) -> None: value = " ".join(data.split()) if not value: return if self._title_depth: self.title.append(value) if not self._ignore_depth: self.text.append(value) def readable_text(self) -> str: """Return normalized paragraphs from collected visible text.""" lines = [" ".join(line.split()) for line in " ".join(self.text).splitlines()] return "\n\n".join(line for line in lines if line) def _fetch_public(url: str) -> tuple[str, str, str]: """Fetch one public URL with redirect, size, MIME, and policy checks.""" current = url headers = { "User-Agent": "HermesPrivateChat/1.0 (+https://chat.bstein.dev)", "Accept": "text/html, text/plain;q=0.9, application/xhtml+xml;q=0.8", } with httpx.Client(follow_redirects=False, timeout=15.0, headers=headers) as client: for _ in range(MAX_REDIRECTS + 1): if not is_safe_url(current): raise ValueError("URL targets a private or internal network address") blocked = check_website_access(current) if blocked: raise ValueError(blocked.get("message", "URL is blocked by website policy")) response = client.get(current) if response.status_code in {301, 302, 303, 307, 308}: location = response.headers.get("location") if not location: raise ValueError("redirect response omitted Location") current = urljoin(current, location) continue response.raise_for_status() content_type = response.headers.get("content-type", "").lower() if not any(kind in content_type for kind in ("text/html", "text/plain", "application/xhtml+xml")): raise ValueError(f"unsupported content type: {content_type or 'unknown'}") raw = response.content if len(raw) > MAX_RESPONSE_BYTES: raise ValueError("page exceeds the 2 MiB extraction limit") return current, content_type, response.text raise ValueError("too many redirects") class PublicExtractProvider(WebSearchProvider): """Extract bounded content directly from public pages without credentials.""" @property def name(self) -> str: return "public-extract" @property def display_name(self) -> str: return "Public page extractor" def is_available(self) -> bool: return True def supports_search(self) -> bool: return False def supports_extract(self) -> bool: return True def extract(self, urls: list[str], **kwargs: Any) -> list[dict[str, Any]]: results: list[dict[str, Any]] = [] for url in urls[:20]: try: final_url, content_type, body = _fetch_public(url) if "html" in content_type: parser = _VisibleTextParser() parser.feed(body) title = " ".join(parser.title).strip() content = parser.readable_text() else: title = "" content = body results.append( { "url": final_url, "title": title, "content": content, "raw_content": content, "metadata": {"source": "public-extract"}, } ) except Exception as exc: results.append( {"url": url, "title": "", "content": "", "error": str(exc)} ) return results