news parser: handle rss 1.0/rdf (4gamer jp feed parsed 0 items -> #newsjp near-silent)

This commit is contained in:
Oleksandr Kozachuk
2026-07-14 12:21:36 +02:00
parent cdd5a4cd48
commit 80000528d3
3 changed files with 34 additions and 10 deletions
+16 -9
View File
@@ -27,6 +27,7 @@ DEFAULT_SUMMARY_CHARS = 200
DEFAULT_NEWS_KEEP = 400
FETCH_TIMEOUT_S = 15
_ATOM = "{http://www.w3.org/2005/Atom}"
_RSS1 = "{http://purl.org/rss/1.0/}" # RSS 1.0 / RDF (e.g. 4gamer.net) namespaces <item>/<title>/<link>
_TAG_RE = re.compile(r"<[^>]+>")
@@ -36,22 +37,28 @@ def _clean_summary(raw: str, max_len: int = 300) -> str:
return re.sub(r"\s+", " ", text).strip()[:max_len]
def _rss_items(root: Any, ns: str, source: str) -> List[Dict[str, str]]:
"""RSS 2.0 (ns='') and RSS 1.0/RDF (ns=_RSS1) both use <item><title><link><description>."""
out: List[Dict[str, str]] = []
for item in root.iter(f"{ns}item"):
title = (item.findtext(f"{ns}title") or "").strip()
link = (item.findtext(f"{ns}link") or "").strip()
summary = _clean_summary(item.findtext(f"{ns}description") or "")
if title:
out.append({"title": title, "link": link, "source": source, "summary": summary})
return out
def parse_feed(data: bytes, source: str = "") -> List[Dict[str, str]]:
"""Parse RSS or Atom bytes into [{title, link, source}] (tolerant)."""
"""Parse RSS 2.0, RSS 1.0/RDF, or Atom bytes into [{title, link, source, summary}] (tolerant)."""
try:
root = ElementTree.fromstring(data)
except Exception as err:
# malformed XML or a blocked entity/DTD attack — tolerate, never raise (NEWS-01)
logging.warning(f"news: unparseable/unsafe feed {source!r}: {err!r}")
return []
items: List[Dict[str, str]] = []
# RSS: <rss><channel><item><title/><link/><description/>
for item in root.iter("item"):
title = (item.findtext("title") or "").strip()
link = (item.findtext("link") or "").strip()
summary = _clean_summary(item.findtext("description") or "")
if title:
items.append({"title": title, "link": link, "source": source, "summary": summary})
# RSS 2.0 (unqualified) + RSS 1.0/RDF (namespaced, e.g. 4gamer) share <item><title><link><description>
items: List[Dict[str, str]] = _rss_items(root, "", source) + _rss_items(root, _RSS1, source)
# Atom: <feed><entry><title/><link href=/><summary|content/>
for entry in root.iter(f"{_ATOM}entry"):
title = (entry.findtext(f"{_ATOM}title") or "").strip()