from __future__ import annotations import json import re import urllib.parse import urllib.request from dataclasses import dataclass from html import unescape from typing import Any from control_plane.resources.models import Resource @dataclass class SearxngSearchClient: endpoint_url: str timeout_seconds: int = 20 @classmethod def from_resources(cls) -> "SearxngSearchClient | None": resource = Resource.objects.filter(is_active=True, provider="searxng").first() if resource is None: return None return cls(endpoint_url=str(resource.config.get("endpoint_url", "http://127.0.0.1:8080")), timeout_seconds=int(resource.config.get("timeout_seconds", 20))) def health(self) -> str: try: with urllib.request.urlopen(self.endpoint_url.rstrip("/") + "/", timeout=5) as response: return "AVAILABLE" if 200 <= response.status < 500 else "DEGRADED" except Exception: return "UNAVAILABLE" def search(self, query: str, *, category: str = "general", limit: int = 5) -> list[dict[str, Any]]: payload = self.search_payload(query) results = [] for item in payload.get("results", [])[:limit]: if not isinstance(item, dict) or not item.get("url"): continue results.append( { "type": "public_web", "source": "searxng", "url": str(item["url"]), "title": str(item.get("title", "")), "category": category, "summary": str(item.get("content", item.get("snippet", ""))), "fallback_evidence": False, } ) return results def search_payload(self, query: str) -> dict[str, Any]: params = urllib.parse.urlencode({"q": query, "format": "json", "categories": "web", "language": "en"}) url = self.endpoint_url.rstrip("/") + "/search?" + params request = urllib.request.Request(url, headers={"Accept": "application/json"}, method="GET") with urllib.request.urlopen(request, timeout=self.timeout_seconds) as response: return json.loads(response.read().decode("utf-8")) @dataclass class WebPageFetcher: timeout_seconds: int = 12 max_bytes: int = 200_000 max_text_chars: int = 4_000 def fetch_many(self, sources: list[dict[str, Any]], *, max_pages: int = 8) -> list[dict[str, Any]]: pages = [] seen = set() for source in sources: url = str(source.get("url", "")) if url in seen or not self._allowed_url(url): continue seen.add(url) page = self.fetch(url) if page is None: continue pages.append({**page, "category": source.get("category", ""), "source_title": source.get("title", "")}) if len(pages) >= max_pages: break return pages def fetch(self, url: str) -> dict[str, Any] | None: if not self._allowed_url(url): return None request = urllib.request.Request(url, headers={"User-Agent": "ArtifexResearchBot/0.1 (+local bounded research)"}, method="GET") try: with urllib.request.urlopen(request, timeout=self.timeout_seconds) as response: content_type = response.headers.get("Content-Type", "") raw = response.read(self.max_bytes) except Exception: return None text = raw.decode("utf-8", errors="ignore") if "html" in content_type.lower() or " bool: parsed = urllib.parse.urlparse(url) return parsed.scheme in {"http", "https"} and bool(parsed.netloc) def _html_to_text(self, html: str) -> str: html = re.sub(r"(?is)<(script|style|noscript|svg).*?", " ", html) html = re.sub(r"(?s)", " ", html) html = re.sub(r"(?is)", "\n", html) html = re.sub(r"(?is)", "\n", html) text = re.sub(r"(?s)<[^>]+>", " ", html) return self._clean_text(unescape(text)) def _clean_text(self, text: str) -> str: lines = [re.sub(r"\s+", " ", line).strip() for line in text.splitlines()] return "\n".join(line for line in lines if line)