didi-lot1-ai/ai_platform/modules/didi_brain/scheduler/feed_parser.py

197 lines
6 KiB
Python

"""RSS feed parsing + normalization.
Thin wrapper over feedparser that returns a clean list of FeedItem records.
We deliberately keep field extraction conservative — every downstream task
works with the same minimal shape.
"""
from __future__ import annotations
import asyncio
from dataclasses import dataclass
from datetime import datetime, timedelta, timezone
from typing import Any
import feedparser
import httpx
from shared.logging import get_logger
log = get_logger(__name__)
FETCH_TIMEOUT_S = 30.0
USER_AGENT = "didibrain-scheduler/0.1 (+https://didi365.eu)"
@dataclass(slots=True, frozen=True)
class FeedItem:
"""One article retrieved from an RSS feed.
Attributes:
title: Article title (cleaned).
url: Canonical URL (link tag).
summary: Short summary or description, may be empty.
full_text: Full article body if the feed exposes it (RSS rarely does;
most feeds only have summaries — caller may fetch the URL
separately to enrich).
publisher: Hostname of the source URL.
published_at: When the article was published. UTC.
feed_url: Source feed URL (for traceability).
"""
title: str
url: str
summary: str
full_text: str
publisher: str
published_at: datetime
feed_url: str
def _parse_published(entry: Any) -> datetime | None:
"""Best-effort parser for feedparser's various date fields.
Falls back to None if the entry has no usable date.
"""
for field in ("published_parsed", "updated_parsed", "created_parsed"):
struct = getattr(entry, field, None) or entry.get(field)
if struct:
try:
# struct_time is naive; treat as UTC (most feeds are).
return datetime(*struct[:6], tzinfo=timezone.utc)
except (TypeError, ValueError):
continue
return None
def _publisher_of(url: str) -> str:
"""Extract host from URL — used as the EvidenceItem.publisher field."""
try:
from urllib.parse import urlparse
host = urlparse(url).hostname or ""
return host.lower().lstrip("www.")
except Exception: # noqa: BLE001
return ""
async def fetch_feed(feed_url: str) -> list[FeedItem]:
"""Fetch and parse a single RSS feed.
Returns an empty list on any error (logged) — the caller iterates over
many feeds and shouldn't be derailed by one bad source.
"""
try:
async with httpx.AsyncClient(
timeout=FETCH_TIMEOUT_S,
headers={"User-Agent": USER_AGENT},
follow_redirects=True,
) as client:
resp = await client.get(feed_url)
if resp.status_code >= 400:
log.warning(
"feed_fetch_http_error",
feed=feed_url,
status=resp.status_code,
)
return []
body = resp.text
except httpx.HTTPError as e:
log.warning("feed_fetch_failed", feed=feed_url, error=str(e))
return []
except Exception as e: # noqa: BLE001
log.warning(
"feed_fetch_unexpected",
feed=feed_url,
error=f"{type(e).__name__}:{e}",
)
return []
# feedparser is synchronous + CPU-bound on parse; offload to a thread so
# we don't block the event loop.
parsed = await asyncio.to_thread(feedparser.parse, body)
if parsed.bozo and not parsed.entries:
log.debug(
"feed_bozo",
feed=feed_url,
error=str(parsed.bozo_exception)[:120],
)
return []
items: list[FeedItem] = []
for entry in parsed.entries:
url = (entry.get("link") or "").strip()
if not url:
continue
title = (entry.get("title") or "").strip()
summary = (entry.get("summary") or entry.get("description") or "").strip()
# full_text rarely present — feedparser exposes 'content' on some
# feeds. Take the first content block when available.
full_text = ""
contents = entry.get("content") or []
if contents and isinstance(contents, list):
first = contents[0]
if isinstance(first, dict):
full_text = (first.get("value") or "").strip()
published_at = _parse_published(entry) or datetime.now(tz=timezone.utc)
items.append(
FeedItem(
title=title,
url=url,
summary=summary,
full_text=full_text,
publisher=_publisher_of(url),
published_at=published_at,
feed_url=feed_url,
)
)
return items
async def fetch_feeds(feed_urls: list[str]) -> list[FeedItem]:
"""Fetch many feeds in parallel, flatten results."""
if not feed_urls:
return []
tasks = [fetch_feed(u) for u in feed_urls]
results = await asyncio.gather(*tasks, return_exceptions=True)
flat: list[FeedItem] = []
for r in results:
if isinstance(r, list):
flat.extend(r)
return flat
def filter_by_age(
items: list[FeedItem],
*,
min_age_s: int,
max_age_s: int,
) -> list[FeedItem]:
"""Keep items whose age is within ``[min_age_s, max_age_s]``.
The min bound exists because some feeds publish before the article
body is fully crawlable; we'd rather wait a bit. The max bound prevents
re-ingesting old items that were already in the corpus.
"""
now = datetime.now(tz=timezone.utc)
out: list[FeedItem] = []
for item in items:
age = (now - item.published_at).total_seconds()
if age < min_age_s or age > max_age_s:
continue
out.append(item)
return out
def dedup_by_url(items: list[FeedItem]) -> list[FeedItem]:
"""Drop duplicates within a batch (same URL across multiple feeds)."""
seen: set[str] = set()
out: list[FeedItem] = []
for item in items:
key = item.url.split("?")[0].rstrip("/").lower()
if key in seen:
continue
seen.add(key)
out.append(item)
return out