262 lines
8.5 KiB
Python
262 lines
8.5 KiB
Python
"""Canonical tag taxonomy for DidiBrain.
|
|
|
|
This module is the SINGLE SOURCE OF TRUTH for what tags exist in the brain.
|
|
All scrapers, extractors, and importers reference these names — never make up
|
|
ad-hoc tags.
|
|
|
|
The structure is a nested dict where each key is either:
|
|
- a leaf (value is None), or
|
|
- a subtree (value is another dict)
|
|
|
|
The seeder script (scripts/04_seed_taxonomy.py) walks this tree and creates
|
|
any missing tags in Atomic. It is idempotent.
|
|
|
|
Atomic creates 5 default root tags at first boot: Topics, People, Locations,
|
|
Organizations, Events. We REUSE the "Topics" root by extending it with our
|
|
own children — the rest stay as-is for compatibility with Atomic's optional
|
|
auto-tagging (which we keep disabled but won't fight).
|
|
|
|
After seeding, scripts/04 writes a flat map of "Path/Like/This → tag_uuid"
|
|
to shared/_tag_ids.json, which downstream code reads via TagResolver.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
from collections.abc import Iterator
|
|
from pathlib import Path
|
|
from typing import Any
|
|
|
|
# A node is either a dict (subtree) or None (leaf).
|
|
TaxonomyTree = dict[str, "TaxonomyTree | None"]
|
|
|
|
# ---------------------------------------------------------------- the spec
|
|
# Order in this dict is the order tags will be created in Atomic.
|
|
TAXONOMY: TaxonomyTree = {
|
|
# Reuse Atomic's existing "Topics" root and extend it with our hierarchy.
|
|
"Topics": {
|
|
"Health": {
|
|
"Vaccines": None,
|
|
"COVID": None,
|
|
"Disease": None,
|
|
"Medicine": None,
|
|
"PublicHealth": None,
|
|
},
|
|
"Politics": {
|
|
"Elections": None,
|
|
"Diplomacy": None,
|
|
"War": None,
|
|
"Government": None,
|
|
},
|
|
"Climate": None,
|
|
"Technology": None,
|
|
"Economy": None,
|
|
"Society": None,
|
|
},
|
|
|
|
# Country of origin for the source / event.
|
|
"Country": {
|
|
"Romania": None,
|
|
"USA": None,
|
|
"Russia": None,
|
|
"Ukraine": None,
|
|
"UK": None,
|
|
"France": None,
|
|
"Germany": None,
|
|
"Spain": None,
|
|
"Italy": None,
|
|
"Poland": None,
|
|
"Moldova": None,
|
|
"Global": None, # for transnational / multi-country items
|
|
},
|
|
|
|
# What kind of source the atom came from.
|
|
"SourceType": {
|
|
"Wikipedia": None,
|
|
"MainstreamMedia": None,
|
|
"StateMedia": None, # state-affiliated outlets (TASS, RT, Sputnik, Xinhua...)
|
|
"TabloidMedia": None,
|
|
"FactCheck": None, # Snopes, PolitiFact, AFP FC, Veridica, Funky...
|
|
"Government": None, # gov.ro, whitehouse.gov, who.int...
|
|
"ScientificJournal": None, # peer-reviewed
|
|
"SocialMedia": None,
|
|
"Blog": None,
|
|
"Forum": None,
|
|
},
|
|
|
|
# Editorial credibility tier — applied by scraper from a static registry.
|
|
"Credibility": {
|
|
"Tier1": None, # Reuters/AP/BBC class
|
|
"Tier2": None, # major mainstream
|
|
"Tier3": None, # weaker mainstream / tabloid
|
|
"StateAffiliated": None,
|
|
"KnownDisinfo": None, # known disinfo outlets (we do still ingest these)
|
|
"Unknown": None,
|
|
},
|
|
|
|
# Primary language of the atom content.
|
|
"Language": {
|
|
"RO": None,
|
|
"EN": None,
|
|
"RU": None,
|
|
"UA": None,
|
|
"FR": None,
|
|
"DE": None,
|
|
"ES": None,
|
|
"IT": None,
|
|
"PL": None,
|
|
},
|
|
|
|
# What KIND of atom this is (vs "what topic" — that's Topics).
|
|
"Type": {
|
|
"Document": None, # full scraped article
|
|
"Claim": None, # extracted atomic claim
|
|
"Quote": None, # verbatim quote/excerpt
|
|
"Summary": None, # synthesized summary
|
|
"Annotation": None, # human/AI annotation about another atom
|
|
},
|
|
|
|
# Stance the source itself takes toward the central claim of the document.
|
|
"Stance": {
|
|
"Asserts": None, # source presents it as fact
|
|
"Reports": None, # source describes it as someone else's claim
|
|
"Refutes": None, # source disagrees / debunks
|
|
"Questions": None, # source raises doubts but doesn't refute
|
|
"Neutral": None, # purely informational, no stance
|
|
},
|
|
|
|
# Verification status of a Claim atom — populated by Didi after analysis,
|
|
# not at ingest. Documents stay un-tagged here.
|
|
"ClaimStatus": {
|
|
"Confirmed": None,
|
|
"Disputed": None,
|
|
"Debunked": None,
|
|
"Unverified": None,
|
|
"PartiallyTrue": None,
|
|
},
|
|
}
|
|
|
|
# ============================================================ flatten helpers
|
|
|
|
|
|
def walk(tree: TaxonomyTree, parent_path: str = "") -> Iterator[tuple[str, str | None, str]]:
|
|
"""Yield (full_path, parent_path_or_None, name) for every node, depth-first.
|
|
|
|
Example output for {"A": {"B": None}}:
|
|
("A", None, "A")
|
|
("A/B", "A", "B")
|
|
"""
|
|
for name, children in tree.items():
|
|
path = f"{parent_path}/{name}" if parent_path else name
|
|
yield (path, parent_path or None, name)
|
|
if children:
|
|
yield from walk(children, path)
|
|
|
|
|
|
def all_paths(tree: TaxonomyTree | None = None) -> list[str]:
|
|
"""All canonical paths in the taxonomy, in creation order."""
|
|
return [p for p, _, _ in walk(tree if tree is not None else TAXONOMY)]
|
|
|
|
|
|
# ============================================================ tag id resolver
|
|
|
|
_DEFAULT_CACHE = Path(__file__).resolve().parent / "_tag_ids.json"
|
|
|
|
|
|
class TagResolver:
|
|
"""Resolves canonical tag paths to Atomic UUIDs.
|
|
|
|
Loads from a JSON cache file written by the seeder. If the file is
|
|
missing or stale, callers should re-run scripts/04_seed_taxonomy.py.
|
|
"""
|
|
|
|
def __init__(self, cache_path: Path | None = None):
|
|
self._cache_path = cache_path or _DEFAULT_CACHE
|
|
self._map: dict[str, str] = {}
|
|
if self._cache_path.exists():
|
|
self._map = json.loads(self._cache_path.read_text(encoding="utf-8"))
|
|
|
|
def __contains__(self, path: str) -> bool:
|
|
return path in self._map
|
|
|
|
def get(self, path: str) -> str | None:
|
|
return self._map.get(path)
|
|
|
|
def require(self, path: str) -> str:
|
|
v = self._map.get(path)
|
|
if not v:
|
|
raise KeyError(
|
|
f"Tag path {path!r} not in resolver cache at {self._cache_path}. "
|
|
f"Run scripts/04_seed_taxonomy.py."
|
|
)
|
|
return v
|
|
|
|
def ids_for(self, paths: list[str], *, ignore_missing: bool = False) -> list[str]:
|
|
ids: list[str] = []
|
|
missing: list[str] = []
|
|
for p in paths:
|
|
v = self._map.get(p)
|
|
if v:
|
|
ids.append(v)
|
|
else:
|
|
missing.append(p)
|
|
if missing and not ignore_missing:
|
|
raise KeyError(f"Missing tag paths: {missing}")
|
|
return ids
|
|
|
|
@property
|
|
def all(self) -> dict[str, str]:
|
|
return dict(self._map)
|
|
|
|
def save(self, mapping: dict[str, str]) -> None:
|
|
self._cache_path.write_text(
|
|
json.dumps(mapping, indent=2, ensure_ascii=False, sort_keys=True),
|
|
encoding="utf-8",
|
|
)
|
|
self._map = mapping
|
|
|
|
def load_from_mapping(self, mapping: dict[str, str]) -> None:
|
|
"""Replace the in-memory map without touching disk.
|
|
|
|
Used by long-running services (e.g. containerized brain_api) that
|
|
refresh the resolver from Atomic at startup, so they don't need
|
|
the _tag_ids.json file baked into the image.
|
|
"""
|
|
self._map = dict(mapping)
|
|
|
|
|
|
def build_path_map_from_tags(tags: list[dict[str, Any]]) -> dict[str, str]:
|
|
"""Convert Atomic's flat tag list (each with parent_id) into path → id map.
|
|
|
|
Atomic /api/tags returns each tag with id, name, parent_id, and a nested
|
|
children list. We don't trust the children list (depth may be limited)
|
|
and instead walk parent_id chains ourselves.
|
|
"""
|
|
by_id: dict[str, dict[str, Any]] = {}
|
|
|
|
def collect(items: list[dict[str, Any]]) -> None:
|
|
for t in items:
|
|
tid = t.get("id")
|
|
if not tid:
|
|
continue
|
|
by_id[tid] = t
|
|
kids = t.get("children") or []
|
|
if kids:
|
|
collect(kids)
|
|
|
|
collect(tags)
|
|
|
|
def path_for(tid: str) -> str:
|
|
parts: list[str] = []
|
|
cur: str | None = tid
|
|
seen: set[str] = set()
|
|
while cur and cur not in seen:
|
|
seen.add(cur)
|
|
t = by_id.get(cur)
|
|
if not t:
|
|
break
|
|
parts.append(t.get("name", "?"))
|
|
cur = t.get("parent_id")
|
|
return "/".join(reversed(parts))
|
|
|
|
return {path_for(tid): tid for tid in by_id}
|