- AsyncFetcher: Sicheres HTTP-Fetching mit SSRF-Schutz, Connection Pooling (50/10), Timeout, Redirect Limit, Rate Limiting, User-Agent - Content-Extraction: trafilatura für HTML→Text, BeautifulSoup4 Fallback - PDF-Extraction: pdfminer.six mit Error-Handling - NormalizedDocument: Schema (url, title, text, metadata, links, content_hash, extraction_tool, extracted_at, word_count) - Crawler-Manager: SSRF-Check → robots.txt → HTTP-Fetch → Extraction → Normalization (Batch-fähig, Error-Isolation pro Fetch) - Security-Policy: SSRF-Schutz (RFC1918, Cloud Metadata, file://, ftp://, localhost), URL-Validation (nur http/https) - Crawler-Endpoints: POST /crawler/fetch, /crawler/fetch/batch, /crawler/validate-url - Test-Cases: SSRF-Schutz, Content Extraction, NormalizedDocument, Error Handling, Content Hash Determinismus
131 lines
3.6 KiB
Python
131 lines
3.6 KiB
Python
"""Main content extraction for HTML pages.
|
|
|
|
Uses trafilatura for fast, deterministic HTML-to-text extraction
|
|
with BeautifulSoup4 as fallback.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import os
|
|
import re
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Configuration
|
|
# ---------------------------------------------------------------------------
|
|
|
|
_MAX_OUTPUT_LENGTH = int(
|
|
os.environ.get("NSCT_CRAWLER_MAX_OUTPUT_LENGTH", "500000")
|
|
)
|
|
|
|
try:
|
|
import trafilatura
|
|
|
|
_HAS_TRAFILATURA = True
|
|
except ImportError:
|
|
_HAS_TRAFILATURA = False
|
|
|
|
try:
|
|
from bs4 import BeautifulSoup
|
|
|
|
_HAS_BS4 = True
|
|
except ImportError:
|
|
_HAS_BS4 = False
|
|
|
|
|
|
def extract_main_content(html: str, url: str = "") -> str:
|
|
"""Extract main content from HTML.
|
|
|
|
Priority: trafilatura -> BeautifulSoup4 -> raw text extraction.
|
|
|
|
Args:
|
|
html: HTML content as string.
|
|
url: Source URL (used for metadata hints).
|
|
|
|
Returns:
|
|
Cleaned main content text (stripped, truncated to max length).
|
|
"""
|
|
if not html:
|
|
return ""
|
|
|
|
text = ""
|
|
|
|
if _HAS_TRAFILATURA:
|
|
try:
|
|
text = trafilatura.extract(
|
|
html,
|
|
include_comments=False,
|
|
include_tables=False,
|
|
include_formatting=False,
|
|
include_links=True,
|
|
output_format="txt",
|
|
)
|
|
if text is not None:
|
|
text = text.strip()
|
|
except Exception:
|
|
logger.debug("trafilatura extraction failed, falling back to bs4")
|
|
text = ""
|
|
|
|
if not text and _HAS_BS4:
|
|
try:
|
|
text = _extract_with_bs4(html)
|
|
except Exception:
|
|
logger.debug("bs4 extraction failed, using minimal fallback")
|
|
text = _minimal_extract(html)
|
|
|
|
if not text:
|
|
text = _minimal_extract(html)
|
|
|
|
# Truncate to max output length
|
|
if len(text) > _MAX_OUTPUT_LENGTH:
|
|
text = text[:_MAX_OUTPUT_LENGTH]
|
|
|
|
return text.strip()
|
|
|
|
|
|
def _extract_with_bs4(html: str) -> str:
|
|
"""Extract main content using BeautifulSoup4.
|
|
|
|
Heuristic: look for article, main, content, or body tags
|
|
and extract text, stripping nav, header, footer, sidebar.
|
|
"""
|
|
soup = BeautifulSoup(html, "html.parser")
|
|
|
|
# Remove script, style, noscript, iframe
|
|
for tag in soup.find_all(["script", "style", "noscript", "iframe"]):
|
|
tag.decompose()
|
|
|
|
# Try specific content containers
|
|
for selector in ["article", "main", "[class*=content]", "[id*=content]",
|
|
"[class*=main]", "[id*=main]"]:
|
|
tag = soup.select_one(selector)
|
|
if tag:
|
|
return tag.get_text(separator="\n", strip=True)[:_MAX_OUTPUT_LENGTH]
|
|
|
|
# Fallback: body content, excluding nav
|
|
body = soup.find("body") or soup
|
|
for tag in body.find_all(["nav", "header", "footer", "aside", "sidebar"]):
|
|
tag.decompose()
|
|
|
|
text = body.get_text(separator="\n", strip=True)
|
|
# Collapse excessive newlines
|
|
text = re.sub(r"\n{3,}", "\n\n", text)
|
|
return text[:_MAX_OUTPUT_LENGTH]
|
|
|
|
|
|
def _minimal_extract(html: str) -> str:
|
|
"""Fallback: strip all tags and return plain text."""
|
|
# Remove HTML tags
|
|
text = re.sub(r"<[^>]+>", " ", html)
|
|
# Decode common entities
|
|
text = text.replace(" ", " ")
|
|
text = text.replace("&", "&")
|
|
text = text.replace("<", "<")
|
|
text = text.replace(">", ">")
|
|
text = text.replace(""", '"')
|
|
text = text.replace("'", "'")
|
|
# Collapse whitespace
|
|
text = re.sub(r"\s+", " ", text)
|
|
return text[:_MAX_OUTPUT_LENGTH].strip() |