197 lines
6 KiB
Python
197 lines
6 KiB
Python
"""RSS feed parsing + normalization.
|
|
|
|
Thin wrapper over feedparser that returns a clean list of FeedItem records.
|
|
We deliberately keep field extraction conservative — every downstream task
|
|
works with the same minimal shape.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timedelta, timezone
|
|
from typing import Any
|
|
|
|
import feedparser
|
|
import httpx
|
|
|
|
from shared.logging import get_logger
|
|
|
|
log = get_logger(__name__)
|
|
|
|
FETCH_TIMEOUT_S = 30.0
|
|
USER_AGENT = "didibrain-scheduler/0.1 (+https://didi365.eu)"
|
|
|
|
|
|
@dataclass(slots=True, frozen=True)
|
|
class FeedItem:
|
|
"""One article retrieved from an RSS feed.
|
|
|
|
Attributes:
|
|
title: Article title (cleaned).
|
|
url: Canonical URL (link tag).
|
|
summary: Short summary or description, may be empty.
|
|
full_text: Full article body if the feed exposes it (RSS rarely does;
|
|
most feeds only have summaries — caller may fetch the URL
|
|
separately to enrich).
|
|
publisher: Hostname of the source URL.
|
|
published_at: When the article was published. UTC.
|
|
feed_url: Source feed URL (for traceability).
|
|
"""
|
|
|
|
title: str
|
|
url: str
|
|
summary: str
|
|
full_text: str
|
|
publisher: str
|
|
published_at: datetime
|
|
feed_url: str
|
|
|
|
|
|
def _parse_published(entry: Any) -> datetime | None:
|
|
"""Best-effort parser for feedparser's various date fields.
|
|
|
|
Falls back to None if the entry has no usable date.
|
|
"""
|
|
for field in ("published_parsed", "updated_parsed", "created_parsed"):
|
|
struct = getattr(entry, field, None) or entry.get(field)
|
|
if struct:
|
|
try:
|
|
# struct_time is naive; treat as UTC (most feeds are).
|
|
return datetime(*struct[:6], tzinfo=timezone.utc)
|
|
except (TypeError, ValueError):
|
|
continue
|
|
return None
|
|
|
|
|
|
def _publisher_of(url: str) -> str:
|
|
"""Extract host from URL — used as the EvidenceItem.publisher field."""
|
|
try:
|
|
from urllib.parse import urlparse
|
|
|
|
host = urlparse(url).hostname or ""
|
|
return host.lower().lstrip("www.")
|
|
except Exception: # noqa: BLE001
|
|
return ""
|
|
|
|
|
|
async def fetch_feed(feed_url: str) -> list[FeedItem]:
|
|
"""Fetch and parse a single RSS feed.
|
|
|
|
Returns an empty list on any error (logged) — the caller iterates over
|
|
many feeds and shouldn't be derailed by one bad source.
|
|
"""
|
|
try:
|
|
async with httpx.AsyncClient(
|
|
timeout=FETCH_TIMEOUT_S,
|
|
headers={"User-Agent": USER_AGENT},
|
|
follow_redirects=True,
|
|
) as client:
|
|
resp = await client.get(feed_url)
|
|
if resp.status_code >= 400:
|
|
log.warning(
|
|
"feed_fetch_http_error",
|
|
feed=feed_url,
|
|
status=resp.status_code,
|
|
)
|
|
return []
|
|
body = resp.text
|
|
except httpx.HTTPError as e:
|
|
log.warning("feed_fetch_failed", feed=feed_url, error=str(e))
|
|
return []
|
|
except Exception as e: # noqa: BLE001
|
|
log.warning(
|
|
"feed_fetch_unexpected",
|
|
feed=feed_url,
|
|
error=f"{type(e).__name__}:{e}",
|
|
)
|
|
return []
|
|
|
|
# feedparser is synchronous + CPU-bound on parse; offload to a thread so
|
|
# we don't block the event loop.
|
|
parsed = await asyncio.to_thread(feedparser.parse, body)
|
|
if parsed.bozo and not parsed.entries:
|
|
log.debug(
|
|
"feed_bozo",
|
|
feed=feed_url,
|
|
error=str(parsed.bozo_exception)[:120],
|
|
)
|
|
return []
|
|
|
|
items: list[FeedItem] = []
|
|
for entry in parsed.entries:
|
|
url = (entry.get("link") or "").strip()
|
|
if not url:
|
|
continue
|
|
title = (entry.get("title") or "").strip()
|
|
summary = (entry.get("summary") or entry.get("description") or "").strip()
|
|
# full_text rarely present — feedparser exposes 'content' on some
|
|
# feeds. Take the first content block when available.
|
|
full_text = ""
|
|
contents = entry.get("content") or []
|
|
if contents and isinstance(contents, list):
|
|
first = contents[0]
|
|
if isinstance(first, dict):
|
|
full_text = (first.get("value") or "").strip()
|
|
published_at = _parse_published(entry) or datetime.now(tz=timezone.utc)
|
|
|
|
items.append(
|
|
FeedItem(
|
|
title=title,
|
|
url=url,
|
|
summary=summary,
|
|
full_text=full_text,
|
|
publisher=_publisher_of(url),
|
|
published_at=published_at,
|
|
feed_url=feed_url,
|
|
)
|
|
)
|
|
return items
|
|
|
|
|
|
async def fetch_feeds(feed_urls: list[str]) -> list[FeedItem]:
|
|
"""Fetch many feeds in parallel, flatten results."""
|
|
if not feed_urls:
|
|
return []
|
|
tasks = [fetch_feed(u) for u in feed_urls]
|
|
results = await asyncio.gather(*tasks, return_exceptions=True)
|
|
flat: list[FeedItem] = []
|
|
for r in results:
|
|
if isinstance(r, list):
|
|
flat.extend(r)
|
|
return flat
|
|
|
|
|
|
def filter_by_age(
|
|
items: list[FeedItem],
|
|
*,
|
|
min_age_s: int,
|
|
max_age_s: int,
|
|
) -> list[FeedItem]:
|
|
"""Keep items whose age is within ``[min_age_s, max_age_s]``.
|
|
|
|
The min bound exists because some feeds publish before the article
|
|
body is fully crawlable; we'd rather wait a bit. The max bound prevents
|
|
re-ingesting old items that were already in the corpus.
|
|
"""
|
|
now = datetime.now(tz=timezone.utc)
|
|
out: list[FeedItem] = []
|
|
for item in items:
|
|
age = (now - item.published_at).total_seconds()
|
|
if age < min_age_s or age > max_age_s:
|
|
continue
|
|
out.append(item)
|
|
return out
|
|
|
|
|
|
def dedup_by_url(items: list[FeedItem]) -> list[FeedItem]:
|
|
"""Drop duplicates within a batch (same URL across multiple feeds)."""
|
|
seen: set[str] = set()
|
|
out: list[FeedItem] = []
|
|
for item in items:
|
|
key = item.url.split("?")[0].rstrip("/").lower()
|
|
if key in seen:
|
|
continue
|
|
seen.add(key)
|
|
out.append(item)
|
|
return out
|