url reader: follow html meta-refresh redirects (getnews shortlinks) with ssrf re-guard
This commit is contained in:
+30
-11
@@ -38,6 +38,9 @@ FETCH_URL_TOOL = {
|
||||
}
|
||||
|
||||
|
||||
_META_REFRESH_URL = re.compile(r"url\s*=\s*['\"]?([^'\";\s]+)", re.I)
|
||||
|
||||
|
||||
class _Extractor(HTMLParser):
|
||||
def __init__(self) -> None:
|
||||
super().__init__()
|
||||
@@ -45,6 +48,7 @@ class _Extractor(HTMLParser):
|
||||
self.parts: List[str] = []
|
||||
self.images: List[str] = []
|
||||
self.og_image: Optional[str] = None
|
||||
self.refresh_url: Optional[str] = None
|
||||
|
||||
def handle_starttag(self, tag: str, attrs) -> None:
|
||||
if tag in ("script", "style", "noscript", "svg"):
|
||||
@@ -55,6 +59,12 @@ class _Extractor(HTMLParser):
|
||||
self.images.append(src)
|
||||
if tag == "meta" and attr.get("property") == "og:image" and attr.get("content"):
|
||||
self.og_image = attr["content"]
|
||||
# meta-refresh redirect (link shorteners, getnews stubs) — URL-04
|
||||
content = attr.get("content")
|
||||
if tag == "meta" and (attr.get("http-equiv") or "").lower() == "refresh" and content:
|
||||
match = _META_REFRESH_URL.search(content)
|
||||
if match and self.refresh_url is None:
|
||||
self.refresh_url = match.group(1)
|
||||
|
||||
def handle_endtag(self, tag: str) -> None:
|
||||
if tag in ("script", "style", "noscript", "svg") and self._skip > 0:
|
||||
@@ -126,29 +136,38 @@ class URLReader:
|
||||
try:
|
||||
async with aiohttp.ClientSession(timeout=timeout, headers={"User-Agent": "FjerkroaBot/1.0"}) as session:
|
||||
final_url, body = await self._get(session, url, max_bytes)
|
||||
# follow a meta-refresh redirect (link shorteners / getnews stubs), re-guarded — URL-04
|
||||
for _ in range(2):
|
||||
extractor = self._extract(body.decode("utf-8", "ignore"))
|
||||
if not extractor.refresh_url:
|
||||
break
|
||||
target = urljoin(final_url, extractor.refresh_url)
|
||||
if guard_url(target) is not None or target == final_url:
|
||||
break
|
||||
logging.info(f"url reader: following meta-refresh -> {target}")
|
||||
final_url, body = await self._get(session, target, max_bytes)
|
||||
except Exception as err:
|
||||
return {"error": str(err)}
|
||||
text = self._to_text(body.decode("utf-8", "ignore"))
|
||||
clean = sanitize_external_text(text, int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
|
||||
images = await self._ingest_images(body.decode("utf-8", "ignore"), final_url, channel, user)
|
||||
html = body.decode("utf-8", "ignore")
|
||||
clean = sanitize_external_text(self._to_text(html), int(config.get("url-max-chars", DEFAULT_MAX_CHARS)))
|
||||
images = await self._ingest_images(html, final_url, channel, user)
|
||||
return {"url": final_url, "text": clean, "images_cached": images}
|
||||
|
||||
def _to_text(self, html: str) -> str:
|
||||
def _extract(self, html: str) -> "_Extractor":
|
||||
extractor = _Extractor()
|
||||
try:
|
||||
extractor.feed(html)
|
||||
except Exception as err:
|
||||
logging.debug(f"html parse (text) failed: {err!r}")
|
||||
return re.sub(r"\s+\n", "\n", " ".join(extractor.parts))
|
||||
logging.debug(f"html parse failed: {err!r}")
|
||||
return extractor
|
||||
|
||||
def _to_text(self, html: str) -> str:
|
||||
return re.sub(r"\s+\n", "\n", " ".join(self._extract(html).parts))
|
||||
|
||||
async def _ingest_images(self, html: str, base_url: str, channel: str, user: str) -> int:
|
||||
if self.image_cache is None:
|
||||
return 0
|
||||
extractor = _Extractor()
|
||||
try:
|
||||
extractor.feed(html)
|
||||
except Exception as err:
|
||||
logging.debug(f"html parse (images) failed: {err!r}")
|
||||
extractor = self._extract(html)
|
||||
candidates = ([extractor.og_image] if extractor.og_image else []) + extractor.images
|
||||
limit = int(self._config().get("url-max-images", DEFAULT_MAX_IMAGES))
|
||||
cached = 0
|
||||
|
||||
Reference in New Issue
Block a user