"""Small keyless web-search adapter for the local research endpoint.""" from __future__ import annotations from html.parser import HTMLParser from urllib.parse import parse_qs, urlsplit class _DuckDuckGoResults(HTMLParser): def __init__(self, limit: int) -> None: super().__init__(convert_charrefs=True) self.limit = limit self.items: list[dict[str, str]] = [] self._active: str | None = None self._href = "" self._text: list[str] = [] self._snippet: list[str] = [] self._in_snippet = False def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: values = dict(attrs) classes = set((values.get("class") or "").split()) if tag == "a" and "result__a" in classes: self._active = "title" self._href = values.get("href") or "" self._text = [] elif "result__snippet" in classes: self._in_snippet = True self._snippet = [] def handle_endtag(self, tag: str) -> None: if tag == "a" and self._active == "title": title = " ".join("".join(self._text).split()) url = self._target_url(self._href) if title and url and len(self.items) < self.limit: self.items.append({"title": title[:300], "url": url, "snippet": ""}) self._active = None if self._in_snippet and tag in {"a", "div", "td"}: snippet = " ".join("".join(self._snippet).split()) if snippet: for item in reversed(self.items): if not item["snippet"]: item["snippet"] = snippet[:1200] break self._in_snippet = False def handle_data(self, data: str) -> None: if self._active == "title": self._text.append(data) if self._in_snippet: self._snippet.append(data) @staticmethod def _target_url(href: str) -> str: if not href: return "" parsed = urlsplit(href) if parsed.hostname and parsed.hostname.endswith("duckduckgo.com"): target = parse_qs(parsed.query).get("uddg", [""])[0] return target if parsed.scheme in {"http", "https"}: return href return "" def parse_duckduckgo_results(document: str, limit: int) -> list[dict[str, str]]: parser = _DuckDuckGoResults(limit) parser.feed(document) # Search cards may repeat the same destination under tracking variants. unique: list[dict[str, str]] = [] seen: set[str] = set() for item in parser.items: parsed = urlsplit(item["url"]) canonical = f"{parsed.scheme.lower()}://{(parsed.hostname or '').lower()}{parsed.path.rstrip('/') or '/'}" if canonical in seen: continue seen.add(canonical) unique.append(item) return unique[:limit]