Add web research and source file reading
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
"""Small keyless web-search adapter for the local research endpoint."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from html.parser import HTMLParser
|
||||
from urllib.parse import parse_qs, urlsplit
|
||||
|
||||
|
||||
class _DuckDuckGoResults(HTMLParser):
|
||||
def __init__(self, limit: int) -> None:
|
||||
super().__init__(convert_charrefs=True)
|
||||
self.limit = limit
|
||||
self.items: list[dict[str, str]] = []
|
||||
self._active: str | None = None
|
||||
self._href = ""
|
||||
self._text: list[str] = []
|
||||
self._snippet: list[str] = []
|
||||
self._in_snippet = False
|
||||
|
||||
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
||||
values = dict(attrs)
|
||||
classes = set((values.get("class") or "").split())
|
||||
if tag == "a" and "result__a" in classes:
|
||||
self._active = "title"
|
||||
self._href = values.get("href") or ""
|
||||
self._text = []
|
||||
elif "result__snippet" in classes:
|
||||
self._in_snippet = True
|
||||
self._snippet = []
|
||||
|
||||
def handle_endtag(self, tag: str) -> None:
|
||||
if tag == "a" and self._active == "title":
|
||||
title = " ".join("".join(self._text).split())
|
||||
url = self._target_url(self._href)
|
||||
if title and url and len(self.items) < self.limit:
|
||||
self.items.append({"title": title[:300], "url": url, "snippet": ""})
|
||||
self._active = None
|
||||
if self._in_snippet and tag in {"a", "div", "td"}:
|
||||
snippet = " ".join("".join(self._snippet).split())
|
||||
if snippet:
|
||||
for item in reversed(self.items):
|
||||
if not item["snippet"]:
|
||||
item["snippet"] = snippet[:1200]
|
||||
break
|
||||
self._in_snippet = False
|
||||
|
||||
def handle_data(self, data: str) -> None:
|
||||
if self._active == "title":
|
||||
self._text.append(data)
|
||||
if self._in_snippet:
|
||||
self._snippet.append(data)
|
||||
|
||||
@staticmethod
|
||||
def _target_url(href: str) -> str:
|
||||
if not href:
|
||||
return ""
|
||||
parsed = urlsplit(href)
|
||||
if parsed.hostname and parsed.hostname.endswith("duckduckgo.com"):
|
||||
target = parse_qs(parsed.query).get("uddg", [""])[0]
|
||||
return target
|
||||
if parsed.scheme in {"http", "https"}:
|
||||
return href
|
||||
return ""
|
||||
|
||||
|
||||
def parse_duckduckgo_results(document: str, limit: int) -> list[dict[str, str]]:
|
||||
parser = _DuckDuckGoResults(limit)
|
||||
parser.feed(document)
|
||||
# Search cards may repeat the same destination under tracking variants.
|
||||
unique: list[dict[str, str]] = []
|
||||
seen: set[str] = set()
|
||||
for item in parser.items:
|
||||
parsed = urlsplit(item["url"])
|
||||
canonical = f"{parsed.scheme.lower()}://{(parsed.hostname or '').lower()}{parsed.path.rstrip('/') or '/'}"
|
||||
if canonical in seen:
|
||||
continue
|
||||
seen.add(canonical)
|
||||
unique.append(item)
|
||||
return unique[:limit]
|
||||
Reference in New Issue
Block a user