247 lines
9.2 KiB
Python
247 lines
9.2 KiB
Python
"""Web-Connector: deckt die Strategien 'sitemap' und 'crawler' ab.
|
|
|
|
- sitemap: liest eine (gz-)Sitemap und holt viele Detailseiten (neueste zuerst,
|
|
inkrementell). So werden z.B. die Kundeninfos aufgenommen.
|
|
- crawler: laedt eine Index-Seite, sammelt alle Detail-Links (per
|
|
detail_pattern) und holt jede Detailseite als eigenes Dokument.
|
|
|
|
Funktioniert lokal ohne Credentials gegen oeffentliche Seiten.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
from urllib.parse import urljoin, urlparse
|
|
|
|
from ..model import ComponentType, Document, Source, Strategy, Tool, source_meta_fingerprint
|
|
from ..store import reconcile_meta
|
|
from .base import Connector
|
|
|
|
USER_AGENT = "wissensdatenbank-etl/0.1 (+local-test)"
|
|
TIMEOUT = 20
|
|
|
|
|
|
class WebCrawlerConnector(Connector):
|
|
name = "web"
|
|
|
|
def __init__(self, data_dir: str | None = None):
|
|
self.data_dir = data_dir
|
|
|
|
def fetch(self, source: Source, tool: Tool | None) -> list[Document]:
|
|
try:
|
|
import requests # noqa: F401 (lazy Verfuegbarkeitspruefung)
|
|
from bs4 import BeautifulSoup # noqa: F401
|
|
except ImportError as e: # pragma: no cover
|
|
print(f" [web] Abhaengigkeit fehlt: {e} -> uebersprungen")
|
|
return []
|
|
|
|
if source.strategy == Strategy.SITEMAP:
|
|
return self._crawl_sitemap(source, tool)
|
|
return self._crawl_index(source, tool) # CRAWLER (Default)
|
|
|
|
# -- intern -------------------------------------------------------------
|
|
|
|
def _get(self, url: str):
|
|
import requests
|
|
|
|
resp = requests.get(url, headers={"User-Agent": USER_AGENT}, timeout=TIMEOUT)
|
|
resp.raise_for_status()
|
|
return resp.text
|
|
|
|
def _crawl_index(self, source: Source, tool: Tool | None) -> list[Document]:
|
|
try:
|
|
html = self._get(source.url)
|
|
except Exception as e:
|
|
print(f" [web] Index nicht erreichbar {source.url}: {e}")
|
|
return []
|
|
|
|
from bs4 import BeautifulSoup
|
|
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
pattern = source.options.get("detail_pattern")
|
|
max_pages = int(source.options.get("max_pages", 25))
|
|
|
|
links: list[str] = []
|
|
seen: set[str] = set()
|
|
for a in soup.find_all("a", href=True):
|
|
full = urljoin(source.url, a["href"])
|
|
if pattern and not re.search(pattern, full):
|
|
continue
|
|
if full in seen:
|
|
continue
|
|
seen.add(full)
|
|
links.append(full)
|
|
if len(links) >= max_pages:
|
|
break
|
|
|
|
print(f" [web] Index {source.url}: {len(links)} Detailseiten gefunden")
|
|
docs: list[Document] = []
|
|
for link in links:
|
|
docs.extend(self._fetch_detail(link, source, tool))
|
|
return docs
|
|
|
|
def _crawl_sitemap(self, source: Source, tool: Tool | None) -> list[Document]:
|
|
"""Holt Detailseiten aus einer (ggf. gzip-)Sitemap.
|
|
|
|
Options:
|
|
sitemap_url : URL der Sitemap oder des Sitemap-Index
|
|
url_pattern : Regex, das Detail-URLs matcht
|
|
limit : optional, max. Anzahl (neueste zuerst, nach Trailing-ID).
|
|
Fehlt limit -> ALLE Treffer des Patterns werden geholt.
|
|
selector : CSS-Selector fuer den Hauptinhalt
|
|
"""
|
|
sitemap_url = source.options.get("sitemap_url", source.url)
|
|
pattern = source.options.get("url_pattern", "")
|
|
limit_opt = source.options.get("limit")
|
|
limit = int(limit_opt) if limit_opt else None # None = alle
|
|
|
|
locs = self._read_sitemap(sitemap_url, depth=2)
|
|
details = self._select_details(locs, pattern, limit)
|
|
|
|
# Inkrementell: bekannte Detailseiten nicht erneut crawlen. Scope-Wechsel
|
|
# erzwingt Neuverarbeitung; reine Metadaten-Aenderung -> guenstiges In-Place-Re-Tagging
|
|
if source.options.get("incremental") and self.data_dir:
|
|
known = self._known_urls(source.domain)
|
|
current_fp = source_meta_fingerprint(source)
|
|
desired = {s.value for s in source.scopes}
|
|
before = len(details)
|
|
fresh: list[str] = []
|
|
retagged = 0
|
|
for u in details:
|
|
entry = known.get(u)
|
|
if entry is None:
|
|
fresh.append(u)
|
|
continue
|
|
existing = {f["scope"] for f in entry["files"]}
|
|
if existing != desired:
|
|
fresh.append(u) # Scope-Wechsel -> neu holen
|
|
continue
|
|
stale = [f["path"] for f in entry["files"] if f["fp"] != current_fp]
|
|
if stale:
|
|
retagged += reconcile_meta(stale, source, current_fp)
|
|
details = fresh
|
|
print(f" [web] inkrementell: {before - len(details)} bekannt "
|
|
f"({retagged} re-tagged), {len(details)} neu")
|
|
|
|
print(f" [web] Sitemap {sitemap_url}: {len(details)} Detailseiten "
|
|
f"(limit {limit if limit is not None else 'alle'})")
|
|
|
|
docs: list[Document] = []
|
|
for link in details:
|
|
docs.extend(self._fetch_detail(link, source, tool))
|
|
return docs
|
|
|
|
def _known_urls(self, domain: str) -> dict[str, dict]:
|
|
"""{url: {"files":[{"scope","path","fp"}]}} bereits verarbeiteter Dokumente."""
|
|
from pathlib import Path
|
|
|
|
from ..store import parse_frontmatter
|
|
|
|
known: dict[str, dict] = {}
|
|
base = Path(self.data_dir) / "processed"
|
|
if not base.exists():
|
|
return known
|
|
for md in base.rglob("*.md"):
|
|
try:
|
|
meta, _ = parse_frontmatter(md.read_text(encoding="utf-8"))
|
|
except Exception:
|
|
continue
|
|
if meta.get("domain") == domain and meta.get("url"):
|
|
entry = known.setdefault(meta["url"], {"files": []})
|
|
entry["files"].append({
|
|
"scope": meta.get("scope", ""),
|
|
"path": str(md),
|
|
"fp": meta.get("meta_fingerprint", ""),
|
|
})
|
|
return known
|
|
|
|
@staticmethod
|
|
def _select_details(locs: list[str], pattern: str, limit: int | None = None) -> list[str]:
|
|
"""Filtert nach pattern und liefert die neuesten (Trailing-ID desc); limit=None => alle."""
|
|
matched = [u for u in locs if not pattern or re.search(pattern, u)]
|
|
|
|
def _key(u: str) -> int:
|
|
m = re.search(r"(\d+)\s*$", u)
|
|
return int(m.group(1)) if m else 0
|
|
|
|
# Tiebreaker nach URL -> bei gleicher Trailing-ID deterministische Reihenfolge
|
|
# (set()-Iteration ist sonst zufaellig -> Git-Churn bei gesetztem limit).
|
|
return sorted(set(matched), key=lambda u: (_key(u), u), reverse=True)[:limit]
|
|
|
|
def _read_sitemap(self, url: str, depth: int = 2) -> list[str]:
|
|
"""Liest <loc>-Eintraege; folgt Sitemap-Indizes rekursiv und entpackt .gz."""
|
|
import gzip
|
|
|
|
import requests
|
|
|
|
try:
|
|
raw = requests.get(url, headers={"User-Agent": USER_AGENT}, timeout=TIMEOUT).content
|
|
except Exception as e:
|
|
print(f" [web] Sitemap nicht erreichbar {url}: {e}")
|
|
return []
|
|
if url.endswith(".gz") or raw[:2] == b"\x1f\x8b":
|
|
try:
|
|
raw = gzip.decompress(raw)
|
|
except Exception:
|
|
pass
|
|
txt = raw.decode("utf-8", "ignore")
|
|
locs = re.findall(r"<loc>(.*?)</loc>", txt)
|
|
|
|
out: list[str] = []
|
|
for loc in locs:
|
|
if depth > 0 and (loc.endswith(".xml") or loc.endswith(".xml.gz")):
|
|
out.extend(self._read_sitemap(loc, depth - 1))
|
|
else:
|
|
out.append(loc)
|
|
return out
|
|
|
|
def _fetch_detail(self, url: str, source: Source, tool: Tool | None) -> list[Document]:
|
|
try:
|
|
html = self._get(url)
|
|
except Exception as e:
|
|
print(f" [web] Seite nicht erreichbar {url}: {e}")
|
|
return []
|
|
|
|
import html2text
|
|
from bs4 import BeautifulSoup
|
|
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
|
|
title = ""
|
|
if soup.title and soup.title.string:
|
|
title = soup.title.string.strip()
|
|
h1 = soup.find("h1")
|
|
if h1:
|
|
title = h1.get_text(strip=True) or title
|
|
|
|
selector = source.options.get("selector")
|
|
node = soup.select_one(selector) if selector else None
|
|
if node is None:
|
|
node = soup.find("main") or soup.find("article") or soup.body or soup
|
|
|
|
# Stoer-Elemente entfernen
|
|
for tag in node.find_all(["script", "style", "nav", "footer", "header"]):
|
|
tag.decompose()
|
|
|
|
h = html2text.HTML2Text()
|
|
h.body_width = 0
|
|
h.ignore_images = True
|
|
body_md = h.handle(str(node)).strip()
|
|
|
|
host = urlparse(url).netloc
|
|
return [
|
|
Document(
|
|
tool=tool.id if tool else "allgemein",
|
|
scope=sc,
|
|
domain=source.domain,
|
|
title=title or url,
|
|
body_md=body_md,
|
|
url=url,
|
|
component_type=ComponentType.PAGE,
|
|
tags=list(source.tags),
|
|
owners=list(source.owners),
|
|
contact=source.contact_address,
|
|
source_system=host or "web",
|
|
)
|
|
for sc in source.scopes
|
|
]
|