Livrare LOT 1 - Didi

This commit is contained in:
Dezvoltari Evotech 2026-06-25 14:13:25 -07:00
commit 5380c3fc63
990 changed files with 133308 additions and 0 deletions

View file

@ -0,0 +1,6 @@
"""Claim extractor — turns Type/Document atoms into Type/Claim atoms.
Public API:
from extractor.extract import extract_claims_from_atom, ExtractedClaim
from extractor.batch import run_batch_extraction
"""

View file

@ -0,0 +1,398 @@
{
"1d7be08c-9fc9-4063-a13b-d20a01b1f24f": {
"atom_id": "1d7be08c-9fc9-4063-a13b-d20a01b1f24f",
"extracted_at": "2026-04-30T17:39:20.830644+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Moderna_COVID-19_vaccine",
"valid_count": 29
},
"221df1cf-8361-4a07-b720-b7660a062da7": {
"atom_id": "221df1cf-8361-4a07-b720-b7660a062da7",
"extracted_at": "2026-04-22T08:20:58.548047+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Plandemic",
"valid_count": 29
},
"3c332c68-8908-44a3-8b6f-3e76e83da111": {
"atom_id": "3c332c68-8908-44a3-8b6f-3e76e83da111",
"extracted_at": "2026-04-22T08:13:47.644503+00:00",
"prompt_version": "v1",
"pushed_count": 15,
"raw_count": 15,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Tiomersal",
"valid_count": 15
},
"3c90743e-04e2-4f2b-a099-8e4c8d7d83b6": {
"atom_id": "3c90743e-04e2-4f2b-a099-8e4c8d7d83b6",
"extracted_at": "2026-04-30T17:27:01.973084+00:00",
"prompt_version": "v1",
"pushed_count": 15,
"raw_count": 15,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Tiomersal",
"valid_count": 15
},
"5094dfc9-9b8b-42e9-a8f7-b3ad16c39713": {
"atom_id": "5094dfc9-9b8b-42e9-a8f7-b3ad16c39713",
"extracted_at": "2026-04-30T17:32:40.173050+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Children%27s_Health_Defense",
"valid_count": 30
},
"5857c5f6-e90e-4028-9643-efd0228d8bdc": {
"atom_id": "5857c5f6-e90e-4028-9643-efd0228d8bdc",
"extracted_at": "2026-04-22T08:27:25.131781+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Moderna_COVID-19_vaccine",
"valid_count": 29
},
"6169e3a9-3574-4e48-8318-3515ae35eb11": {
"atom_id": "6169e3a9-3574-4e48-8318-3515ae35eb11",
"extracted_at": "2026-04-22T08:34:28.188203+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/MMR_vaccine_and_autism",
"valid_count": 29
},
"61f690f3-0b6a-45a5-8074-798d4afa1453": {
"atom_id": "61f690f3-0b6a-45a5-8074-798d4afa1453",
"extracted_at": "2026-04-22T08:30:34.956251+00:00",
"prompt_version": "v1",
"pushed_count": 28,
"raw_count": 28,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Vaccine_adverse_event",
"valid_count": 28
},
"685eedb4-62e9-437d-887f-ec81743f6bb8": {
"atom_id": "685eedb4-62e9-437d-887f-ec81743f6bb8",
"extracted_at": "2026-04-22T08:24:36.031727+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Robert_F._Kennedy_Jr.",
"valid_count": 29
},
"69f16450-60e7-40fd-ae50-dd607514352b": {
"atom_id": "69f16450-60e7-40fd-ae50-dd607514352b",
"extracted_at": "2026-04-22T08:37:13.179879+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Vaccination",
"valid_count": 29
},
"6b86489d-fc88-4f40-8c81-371e82ace952": {
"atom_id": "6b86489d-fc88-4f40-8c81-371e82ace952",
"extracted_at": "2026-04-30T17:51:07.975837+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Vaccine",
"valid_count": 30
},
"6e13b816-50c9-4cdd-a9e4-a67b2d65dca0": {
"atom_id": "6e13b816-50c9-4cdd-a9e4-a67b2d65dca0",
"extracted_at": "2026-04-22T08:38:16.576393+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Vaccine",
"valid_count": 30
},
"6e791e83-8bfd-4aaf-9111-9e77072517d8": {
"atom_id": "6e791e83-8bfd-4aaf-9111-9e77072517d8",
"extracted_at": "2026-04-22T08:14:35.928834+00:00",
"prompt_version": "v1",
"pushed_count": 27,
"raw_count": 27,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Andrew_Wakefield",
"valid_count": 27
},
"79c1010d-1406-4b37-8ab9-c8e3e6d054ef": {
"atom_id": "79c1010d-1406-4b37-8ab9-c8e3e6d054ef",
"extracted_at": "2026-04-30T17:41:23.814310+00:00",
"prompt_version": "v1",
"pushed_count": 28,
"raw_count": 28,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Vaccine_adverse_event",
"valid_count": 28
},
"8182687d-9350-4870-818f-5e6e226def88": {
"atom_id": "8182687d-9350-4870-818f-5e6e226def88",
"extracted_at": "2026-04-22T08:16:30.831673+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://ro.wikipedia.org/wiki/Pandemia_de_COVID-19_%C3%AEn_Rom%C3%A2nia",
"valid_count": 29
},
"8516163b-fe3e-448c-9fda-f48155a05327": {
"atom_id": "8516163b-fe3e-448c-9fda-f48155a05327",
"extracted_at": "2026-04-30T17:37:46.813140+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/COVID-19_vaccine_misinformation_and_hesitancy",
"valid_count": 30
},
"86de36c3-7b60-4120-ab0f-018dfa1c8ba9": {
"atom_id": "86de36c3-7b60-4120-ab0f-018dfa1c8ba9",
"extracted_at": "2026-04-22T08:32:07.787823+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Anti-vaccine_activism",
"valid_count": 30
},
"8d87dd49-6a8e-4d6d-b15a-1631db0c7153": {
"atom_id": "8d87dd49-6a8e-4d6d-b15a-1631db0c7153",
"extracted_at": "2026-04-30T17:31:49.206015+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://ro.wikipedia.org/wiki/Vaccin",
"valid_count": 29
},
"9272802d-211c-42e4-b064-cb8acc143e1c": {
"atom_id": "9272802d-211c-42e4-b064-cb8acc143e1c",
"extracted_at": "2026-04-30T17:46:26.293983+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/MMR_vaccine_and_autism",
"valid_count": 29
},
"9c12182c-f2fe-487f-b43d-f97fd8f84984": {
"atom_id": "9c12182c-f2fe-487f-b43d-f97fd8f84984",
"extracted_at": "2026-04-30T17:43:18.929395+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Anti-vaccine_activism",
"valid_count": 30
},
"9c2c5fbb-cb32-40ab-b6fd-6819c8e49fd2": {
"atom_id": "9c2c5fbb-cb32-40ab-b6fd-6819c8e49fd2",
"extracted_at": "2026-04-30T17:30:19.630562+00:00",
"prompt_version": "v1",
"pushed_count": 6,
"raw_count": 6,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Vaccinare",
"valid_count": 6
},
"a5d2ce82-0249-4253-9c2d-bddd6853d311": {
"atom_id": "a5d2ce82-0249-4253-9c2d-bddd6853d311",
"extracted_at": "2026-04-30T17:44:43.113762+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Andrew_Wakefield",
"valid_count": 30
},
"ab56b518-9d13-4de0-bf01-1c89abd087e0": {
"atom_id": "ab56b518-9d13-4de0-bf01-1c89abd087e0",
"extracted_at": "2026-04-22T08:12:44.236406+00:00",
"prompt_version": "v1",
"pushed_count": 21,
"raw_count": 21,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Variol%C4%83",
"valid_count": 21
},
"abc36af6-e4fe-4c75-b98d-57738f9406ca": {
"atom_id": "abc36af6-e4fe-4c75-b98d-57738f9406ca",
"extracted_at": "2026-04-22T08:26:19.603769+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/COVID-19_vaccine_misinformation_and_hesitancy",
"valid_count": 30
},
"aef7f81f-c6a8-44e0-859d-812b144380e2": {
"atom_id": "aef7f81f-c6a8-44e0-859d-812b144380e2",
"extracted_at": "2026-04-30T17:34:27.380347+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Plandemic",
"valid_count": 29
},
"b09275c4-d325-4aec-b730-8fa23e374284": {
"atom_id": "b09275c4-d325-4aec-b730-8fa23e374284",
"extracted_at": "2026-04-30T17:28:03.929162+00:00",
"prompt_version": "v1",
"pushed_count": 27,
"raw_count": 27,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Andrew_Wakefield",
"valid_count": 27
},
"b569c570-6d19-42a7-b13d-41039c8b4391": {
"atom_id": "b569c570-6d19-42a7-b13d-41039c8b4391",
"extracted_at": "2026-04-30T17:36:11.186401+00:00",
"prompt_version": "v1",
"pushed_count": 25,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 5
},
"source_url": "https://en.wikipedia.org/wiki/Robert_F._Kennedy_Jr.",
"valid_count": 25
},
"b7b39518-eb7b-4d7d-afb4-d6e9ef815094": {
"atom_id": "b7b39518-eb7b-4d7d-afb4-d6e9ef815094",
"extracted_at": "2026-04-22T08:35:55.179684+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Vaccine_hesitancy",
"valid_count": 30
},
"c7651882-6731-438a-933c-efb827344495": {
"atom_id": "c7651882-6731-438a-933c-efb827344495",
"extracted_at": "2026-04-22T08:29:18.331921+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Pfizer%E2%80%93BioNTech_COVID-19_vaccine",
"valid_count": 30
},
"c97398f4-819b-46a9-b6fc-f25f6e192340": {
"atom_id": "c97398f4-819b-46a9-b6fc-f25f6e192340",
"extracted_at": "2026-04-22T08:16:44.511818+00:00",
"prompt_version": "v1",
"pushed_count": 6,
"raw_count": 6,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Vaccinare",
"valid_count": 6
},
"d933d1f5-03a5-4238-b0ea-352d72f0699c": {
"atom_id": "d933d1f5-03a5-4238-b0ea-352d72f0699c",
"extracted_at": "2026-04-30T17:49:08.873214+00:00",
"prompt_version": "v1",
"pushed_count": 29,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 1
},
"source_url": "https://en.wikipedia.org/wiki/Vaccination",
"valid_count": 29
},
"e5e5aac1-addb-4586-a18b-2e004ac8f50f": {
"atom_id": "e5e5aac1-addb-4586-a18b-2e004ac8f50f",
"extracted_at": "2026-04-30T17:48:08.258151+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Vaccine_hesitancy",
"valid_count": 30
},
"e7e388e8-6b1c-4d48-8565-635fde082c75": {
"atom_id": "e7e388e8-6b1c-4d48-8565-635fde082c75",
"extracted_at": "2026-04-22T08:33:25.352110+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Andrew_Wakefield",
"valid_count": 30
},
"e9a1c988-01ac-48c8-8da3-a5f3827e0594": {
"atom_id": "e9a1c988-01ac-48c8-8da3-a5f3827e0594",
"extracted_at": "2026-04-22T08:19:10.267888+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Children%27s_Health_Defense",
"valid_count": 30
},
"e9e72b6f-909c-400a-8d59-0c103c2fac47": {
"atom_id": "e9e72b6f-909c-400a-8d59-0c103c2fac47",
"extracted_at": "2026-04-30T17:40:36.770904+00:00",
"prompt_version": "v1",
"pushed_count": 30,
"raw_count": 30,
"rejected": {},
"source_url": "https://en.wikipedia.org/wiki/Pfizer%E2%80%93BioNTech_COVID-19_vaccine",
"valid_count": 30
},
"ecb20b8f-4141-4831-b88a-10ad4c415305": {
"atom_id": "ecb20b8f-4141-4831-b88a-10ad4c415305",
"extracted_at": "2026-04-22T08:18:19.056060+00:00",
"prompt_version": "v1",
"pushed_count": 28,
"raw_count": 30,
"rejected": {
"quote_not_in_source": 2
},
"source_url": "https://ro.wikipedia.org/wiki/Vaccin",
"valid_count": 28
},
"f0434eae-41f4-42bf-8459-54bbc32e9b29": {
"atom_id": "f0434eae-41f4-42bf-8459-54bbc32e9b29",
"extracted_at": "2026-04-30T17:26:38.082088+00:00",
"prompt_version": "v1",
"pushed_count": 22,
"raw_count": 22,
"rejected": {},
"source_url": "https://ro.wikipedia.org/wiki/Variol%C4%83",
"valid_count": 22
}
}

View file

@ -0,0 +1,67 @@
"""Persistent state for the claim extractor.
We track which document atoms have already been processed so re-runs are
idempotent. Stored as a single JSON file under extractor/_extracted.json.
Each entry records the run timestamp, prompt version, and a small summary
of what came out, so we can audit later or selectively re-extract if a
prompt version changes.
"""
from __future__ import annotations
import json
from dataclasses import asdict, dataclass, field
from datetime import datetime, timezone
from pathlib import Path
_STATE_FILE = Path(__file__).resolve().parent / "_extracted.json"
@dataclass(slots=True)
class DocExtractionRecord:
atom_id: str
source_url: str
extracted_at: str
prompt_version: str
raw_count: int # how many claims the LLM returned
valid_count: int # how many passed validation
pushed_count: int # how many were created in Atomic (excludes dedup hits)
rejected: dict[str, int] = field(default_factory=dict)
class ExtractionState:
"""Loads / saves the extraction log file."""
def __init__(self, path: Path | None = None):
self._path = path or _STATE_FILE
self._records: dict[str, DocExtractionRecord] = {}
if self._path.exists():
data = json.loads(self._path.read_text(encoding="utf-8"))
for atom_id, raw in data.items():
self._records[atom_id] = DocExtractionRecord(**raw)
def has(self, atom_id: str, prompt_version: str) -> bool:
rec = self._records.get(atom_id)
return rec is not None and rec.prompt_version == prompt_version
def get(self, atom_id: str) -> DocExtractionRecord | None:
return self._records.get(atom_id)
def upsert(self, record: DocExtractionRecord) -> None:
self._records[record.atom_id] = record
def save(self) -> None:
out = {k: asdict(v) for k, v in self._records.items()}
self._path.write_text(
json.dumps(out, indent=2, ensure_ascii=False, sort_keys=True),
encoding="utf-8",
)
@property
def all(self) -> dict[str, DocExtractionRecord]:
return dict(self._records)
def now_iso() -> str:
return datetime.now(timezone.utc).isoformat()

View file

@ -0,0 +1,233 @@
"""Batch orchestrator for claim extraction.
Walks all Type/Document atoms in Atomic, runs extraction on each, and pushes
the resulting claims as new Type/Claim atoms. Idempotent across runs via the
state file in extractor/_extracted.json.
"""
from __future__ import annotations
import asyncio
from collections.abc import AsyncIterator
from dataclasses import dataclass, field
from typing import Any
from urllib.parse import unquote
from shared.atomic_api import AtomicClient
from shared.config import settings
from shared.embedding_client import EmbeddingClient # noqa: F401 (future use)
from shared.llm_client import LlmClient, LlmError
from shared.logging import get_logger
from shared.taxonomy import TagResolver
from extractor._state import DocExtractionRecord, ExtractionState, now_iso
from extractor.extract import (
PROMPT_VERSION,
ExtractionResult,
extract_claims_from_atom,
)
from extractor.push import push_claim
log = get_logger(__name__)
@dataclass(slots=True)
class BatchStats:
docs_seen: int = 0
docs_skipped_already_done: int = 0
docs_processed: int = 0
docs_failed: int = 0
claims_raw: int = 0
claims_valid: int = 0
claims_created: int = 0
claims_duplicate: int = 0
claims_error: int = 0
rejected_reasons: dict[str, int] = field(default_factory=dict)
# ====================================================== document selection
async def _list_documents_to_process(
atomic: AtomicClient, resolver: TagResolver, *, limit: int = 1000
) -> list[dict[str, Any]]:
"""Return all atoms tagged Type/Document, with their tags inlined.
We page through /api/atoms?tag_id=<Type/Document> and pull metadata for
each, since the list endpoint already returns tags inline.
"""
type_doc_id = resolver.require("Type/Document")
page_size = 50
out: list[dict[str, Any]] = []
offset = 0
while True:
result = await atomic.list_atoms(
limit=page_size, offset=offset, tag_id=type_doc_id
)
atoms = result.get("atoms") or result.get("data") or (result if isinstance(result, list) else [])
if not atoms:
break
for a in atoms:
out.append(a)
if len(out) >= limit:
return out
if len(atoms) < page_size:
break
offset += page_size
return out
def _title_from_atom(atom: dict[str, Any]) -> str:
"""Best-effort title: prefer the Markdown H1 in content, fall back to URL slug."""
content = atom.get("content") or ""
# Look for the first '# ...' line at the start
for line in content.lstrip().splitlines():
line = line.strip()
if line.startswith("# "):
return line[2:].strip()
if line:
break # first non-empty isn't a header → fall through to URL
url = atom.get("source_url") or ""
if url:
last = url.rstrip("/").rsplit("/", 1)[-1]
return unquote(last).replace("_", " ")
return atom.get("id", "?")[:8]
def _language_from_atom(atom: dict[str, Any]) -> str:
"""Read the Language/<X> tag if present, default 'EN'."""
for tag in atom.get("tags") or []:
name = tag.get("name", "")
# The tag list returns just `name`, not the full path. Languages are
# short codes (RO/EN/RU/...) so direct match works.
if name in {"RO", "EN", "RU", "UA", "FR", "DE", "ES", "IT", "PL"}:
return name
return "EN"
# ============================================================== one document
async def process_one(
*,
llm: LlmClient,
atomic: AtomicClient,
resolver: TagResolver,
state: ExtractionState,
atom: dict[str, Any],
stats: BatchStats,
) -> None:
atom_id = atom["id"]
if state.has(atom_id, PROMPT_VERSION):
stats.docs_skipped_already_done += 1
return
# /api/atoms (list) returns summary objects WITHOUT full content. We have
# to fetch the full atom individually to get the body for extraction.
full_atom = await atomic.get_atom(atom_id)
content = full_atom.get("content") or ""
title = _title_from_atom(full_atom)
language = _language_from_atom(full_atom)
if not content:
log.warning("doc_no_content", atom_id=atom_id)
stats.docs_failed += 1
return
log.info("extracting", atom_id=atom_id[:8], title=title, lang=language, chars=len(content))
try:
result: ExtractionResult = await extract_claims_from_atom(
llm,
title=title,
language=language,
content=content,
)
except LlmError as e:
log.error("extraction_failed", atom_id=atom_id, error=str(e), body=(e.body or "")[:300])
stats.docs_failed += 1
return
except Exception as e: # noqa: BLE001
log.error("extraction_crashed", atom_id=atom_id, error=f"{type(e).__name__}: {e}")
stats.docs_failed += 1
return
stats.docs_processed += 1
stats.claims_raw += result.raw_count
stats.claims_valid += len(result.valid)
for k, v in result.rejected.items():
stats.rejected_reasons[k] = stats.rejected_reasons.get(k, 0) + v
pushed = 0
duplicates = 0
errors = 0
for c in result.valid:
_, status = await push_claim(
atomic,
parent_atom=full_atom,
claim=c,
parent_title=title,
resolver=resolver,
)
if status == "created":
pushed += 1
elif status == "duplicate":
duplicates += 1
else:
errors += 1
stats.claims_created += pushed
stats.claims_duplicate += duplicates
stats.claims_error += errors
state.upsert(
DocExtractionRecord(
atom_id=atom_id,
source_url=full_atom.get("source_url", ""),
extracted_at=now_iso(),
prompt_version=PROMPT_VERSION,
raw_count=result.raw_count,
valid_count=len(result.valid),
pushed_count=pushed,
rejected=dict(result.rejected),
)
)
state.save() # save after each doc so a crash doesn't lose progress
# ============================================================ batch entry
async def run_batch(
*,
limit: int | None = None,
only_atom_ids: set[str] | None = None,
) -> BatchStats:
if not settings.atomic_token:
raise RuntimeError("ATOMIC_TOKEN missing")
resolver = TagResolver()
if "Type/Claim" not in resolver:
raise RuntimeError("taxonomy not seeded — run scripts/04_seed_taxonomy.py")
state = ExtractionState()
stats = BatchStats()
async with AtomicClient() as atomic, LlmClient() as llm:
docs = await _list_documents_to_process(atomic, resolver, limit=limit or 1000)
if only_atom_ids:
docs = [d for d in docs if d["id"] in only_atom_ids]
stats.docs_seen = len(docs)
log.info("batch_start", docs=len(docs), prompt_version=PROMPT_VERSION)
for atom in docs:
await process_one(
llm=llm,
atomic=atomic,
resolver=resolver,
state=state,
atom=atom,
stats=stats,
)
return stats

View file

@ -0,0 +1,204 @@
"""Single-document claim extraction logic.
Given one Type/Document atom, this module:
1. Builds the extraction prompt from the versioned template
2. Calls Qwen 397B (REASONING role) for structured JSON output
3. Parses + validates each claim:
- Required fields present and right types
- Stance is in the allowed enum
- Confidence above threshold
- Quote is a verbatim substring of the source (programmatic check
this is the cheap defense against LLM hallucination)
4. Returns a list of ExtractedClaim dataclasses ready for the pusher.
The pusher (push.py) takes ExtractedClaim and creates a Type/Claim atom in
Atomic, with the proper tag inheritance and a stable hash-based source_url.
"""
from __future__ import annotations
import hashlib
import re
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from shared.config import LlmRole
from shared.llm_client import LlmClient, LlmError
from shared.logging import get_logger
log = get_logger(__name__)
PROMPT_VERSION = "v1"
PROMPT_PATH = Path(__file__).resolve().parent / "prompts" / f"claim_extraction_{PROMPT_VERSION}.md"
ALLOWED_STANCES = {"ASSERTS", "REPORTS", "REFUTES", "QUESTIONS", "NEUTRAL"}
MIN_CONFIDENCE = 0.7
MIN_CLAIM_CHARS = 20
MIN_QUOTE_CHARS = 10
MAX_QUOTE_CHARS = 600
MAX_INPUT_CHARS = 120_000 # ~30K tokens, well under Qwen's 262K context
_PROMPT_TEMPLATE: str | None = None
def _load_prompt() -> str:
global _PROMPT_TEMPLATE
if _PROMPT_TEMPLATE is None:
_PROMPT_TEMPLATE = PROMPT_PATH.read_text(encoding="utf-8")
return _PROMPT_TEMPLATE
# ============================================================ data structures
@dataclass(slots=True, frozen=True)
class ExtractedClaim:
claim: str
quote: str
stance: str # uppercase: ASSERTS / REPORTS / REFUTES / QUESTIONS / NEUTRAL
confidence: float
def stable_hash(self) -> str:
"""8-char SHA-1 of canonicalized claim text — used in source_url fragment."""
canonical = " ".join(self.claim.lower().strip().split())
return hashlib.sha1(canonical.encode("utf-8")).hexdigest()[:8]
@dataclass(slots=True)
class ExtractionResult:
"""What `extract_claims_from_atom` returns."""
raw_count: int # how many items the LLM returned
valid: list[ExtractedClaim]
rejected: dict[str, int] # reason → count
backend: str # which backend served the request
# ============================================================ extraction core
async def extract_claims_from_atom(
llm: LlmClient,
*,
title: str,
language: str,
content: str,
max_tokens_out: int = 6000,
) -> ExtractionResult:
"""Run the extraction LLM call and validate the results.
Raises LlmError on transport / JSON parse failure (caller decides whether
to retry or skip the document).
"""
prompt = _load_prompt().replace("{title}", title)
prompt = prompt.replace("{language}", language)
truncated = content[:MAX_INPUT_CHARS]
if len(content) > MAX_INPUT_CHARS:
truncated += "\n\n[document truncated for extraction]"
prompt = prompt.replace("{content}", truncated)
result, usage = await llm.chat_json(
role=LlmRole.REASONING,
system=(
"You are a precise information extractor. Output strictly valid "
"JSON, no commentary, no markdown fences."
),
user=prompt,
max_tokens=max_tokens_out,
temperature=0.0,
)
if not isinstance(result, dict) or "claims" not in result:
raise LlmError(
f"unexpected response shape: keys={list(result.keys()) if isinstance(result, dict) else type(result).__name__}"
)
raw_claims = result.get("claims") or []
if not isinstance(raw_claims, list):
raise LlmError(f"`claims` is not a list: {type(raw_claims).__name__}")
valid: list[ExtractedClaim] = []
rejected: dict[str, int] = {}
norm_source = _normalize_for_match(content)
for raw in raw_claims:
outcome = _parse_one(raw, norm_source)
if isinstance(outcome, ExtractedClaim):
valid.append(outcome)
else:
rejected[outcome] = rejected.get(outcome, 0) + 1
# Dedup within this batch by stable hash
seen: set[str] = set()
deduped: list[ExtractedClaim] = []
for c in valid:
h = c.stable_hash()
if h in seen:
rejected["intra_batch_duplicate"] = rejected.get("intra_batch_duplicate", 0) + 1
continue
seen.add(h)
deduped.append(c)
return ExtractionResult(
raw_count=len(raw_claims),
valid=deduped,
rejected=rejected,
backend=str(usage.get("backend", "")),
)
# ============================================================ validation
def _parse_one(raw: Any, norm_source: str) -> ExtractedClaim | str:
"""Validate one raw item from the LLM. Returns ExtractedClaim or error reason str."""
if not isinstance(raw, dict):
return "not_a_dict"
claim = (raw.get("claim") or "").strip()
quote = (raw.get("quote") or "").strip()
stance = (raw.get("stance") or "").strip().upper()
try:
confidence = float(raw.get("confidence", 0))
except (TypeError, ValueError):
return "bad_confidence_type"
if len(claim) < MIN_CLAIM_CHARS:
return "claim_too_short"
if len(quote) < MIN_QUOTE_CHARS:
return "quote_too_short"
if len(quote) > MAX_QUOTE_CHARS:
return "quote_too_long"
if stance not in ALLOWED_STANCES:
return "bad_stance"
if confidence < MIN_CONFIDENCE:
return "low_confidence"
# The critical anti-hallucination check: the quote must actually appear
# in the source document (after whitespace normalization).
if not _quote_in_source(quote, norm_source):
return "quote_not_in_source"
return ExtractedClaim(
claim=claim,
quote=quote,
stance=stance,
confidence=confidence,
)
_WHITESPACE_RE = re.compile(r"\s+")
def _normalize_for_match(s: str) -> str:
"""Collapse runs of whitespace and normalize quote chars for substring match."""
s = s.replace("\u2018", "'").replace("\u2019", "'")
s = s.replace("\u201c", '"').replace("\u201d", '"')
s = s.replace("\u2013", "-").replace("\u2014", "-")
return _WHITESPACE_RE.sub(" ", s).strip()
def _quote_in_source(quote: str, norm_source: str) -> bool:
return _normalize_for_match(quote) in norm_source

View file

@ -0,0 +1,65 @@
You are an expert fact extractor for a knowledge graph that supports
disinformation analysis. Your job is to read one source document and extract
every distinct, verifiable factual claim it makes.
# What counts as a "claim"
A claim is a **specific, self-contained, verifiable proposition**. It must be:
- Concrete enough that someone could check it against evidence
- Self-contained: understandable without reading surrounding text
- About facts, events, findings, or attributed statements — not opinions
A claim is **NOT**:
- An opinion or value judgment ("the policy was misguided")
- A vague generality ("many people worry", "some scientists think")
- A definition or terminology explanation
- A question or recommendation
- Pure background description without a specific assertion
# Stance classification
For each claim, decide how the SOURCE itself treats it:
- **ASSERTS** — the source presents the claim as factual / a finding it stands behind
- **REPORTS** — the source describes someone else's claim without endorsing or refuting it
- **REFUTES** — the source explicitly disagrees with, debunks, or corrects the claim
- **QUESTIONS** — the source raises doubts or uncertainty without fully refuting
- **NEUTRAL** — purely descriptive context with no editorial stance
# Output rules — STRICT
You MUST output ONLY a single JSON object with this exact shape:
```json
{
"claims": [
{
"claim": "<canonical claim text, in the SAME LANGUAGE as the source>",
"quote": "<EXACT verbatim substring from the source where this claim appears>",
"stance": "ASSERTS|REPORTS|REFUTES|QUESTIONS|NEUTRAL",
"confidence": 0.0-1.0
}
]
}
```
Hard rules:
1. The `quote` MUST be a character-perfect substring of the source. Copy-paste it. Do not paraphrase, translate, or add ellipses inside it.
2. The `claim` must be in the SAME language as the source document. If the source is Romanian, the claim must be in Romanian. Do NOT translate.
3. Quote at minimum one full sentence; quote at most ~300 characters.
4. Extract up to **30** claims, prioritizing the most consequential and contestable ones. Skip duplicates and trivia.
5. Confidence reflects how clean and verifiable the claim is. Use 0.7-0.85 for ordinary factual claims, 0.85-0.95 for well-supported specific findings, below 0.7 for claims you are unsure about (these will be filtered out).
6. Do NOT add any text before or after the JSON. No markdown fences. No commentary. Just the JSON object.
7. If the document has no extractable factual claims, return `{"claims": []}`.
# Source
Title: {title}
Language: {language}
# Source content
{content}

View file

@ -0,0 +1,124 @@
"""Push validated ExtractedClaim objects into Atomic as Type/Claim atoms.
Each claim becomes a tiny atom with:
- source_url = `{parent_url}#claim={hash8}` so it dedups idempotently and
so `parent_url = source_url.split("#")[0]` is trivial to recover later
- tag inheritance from the parent (Country, Topic, SourceType, Credibility,
Language) plus our two new tags: Type/Claim and Stance/<X>
"""
from __future__ import annotations
from typing import Any
from shared.atomic_api import AtomicApiError, AtomicClient
from shared.logging import get_logger
from shared.taxonomy import TagResolver
from extractor.extract import ExtractedClaim
log = get_logger(__name__)
_STANCE_PATH = {
"ASSERTS": "Stance/Asserts",
"REPORTS": "Stance/Reports",
"REFUTES": "Stance/Refutes",
"QUESTIONS": "Stance/Questions",
"NEUTRAL": "Stance/Neutral",
}
def build_claim_markdown(
claim: ExtractedClaim,
*,
parent_title: str,
parent_url: str,
parent_atom_id: str,
) -> str:
"""Render a claim atom's body as Markdown.
The structure is intentionally consistent so it can be parsed back later
by Didi or by re-indexing scripts.
"""
return (
f"# Claim\n\n"
f"{claim.claim}\n\n"
f"## Quote\n"
f"> {claim.quote}\n\n"
f"## Source\n"
f"- Document: [{parent_title}]({parent_url})\n"
f"- Parent atom: `{parent_atom_id}`\n"
f"- Stance in source: {claim.stance}\n"
f"- Extraction confidence: {claim.confidence:.2f}\n"
)
def build_claim_url(parent_url: str, claim: ExtractedClaim) -> str:
"""Stable hash-based URL fragment so re-extraction dedupes naturally."""
base = parent_url.split("#", 1)[0]
return f"{base}#claim={claim.stable_hash()}"
def inherit_tag_ids(
parent_atom: dict[str, Any],
claim: ExtractedClaim,
resolver: TagResolver,
) -> list[str]:
"""Build the tag-id list for a new claim atom.
Inherits all parent tags except Type/Document, and adds Type/Claim plus
the appropriate Stance/<X>.
"""
type_doc_id = resolver.require("Type/Document")
type_claim_id = resolver.require("Type/Claim")
stance_id = resolver.require(_STANCE_PATH[claim.stance])
parent_tag_ids = [
t["id"] for t in (parent_atom.get("tags") or []) if t.get("id") != type_doc_id
]
return parent_tag_ids + [type_claim_id, stance_id]
async def push_claim(
atomic: AtomicClient,
*,
parent_atom: dict[str, Any],
claim: ExtractedClaim,
parent_title: str,
resolver: TagResolver,
) -> tuple[dict[str, Any] | None, str]:
"""Create one claim atom in Atomic. Returns (atom_dict, status).
Status is one of:
- "created": new atom was created
- "duplicate": same hash already exists, skipped
- "error": creation failed (atom_dict is None)
"""
parent_url = parent_atom.get("source_url") or ""
parent_id = parent_atom.get("id") or ""
claim_url = build_claim_url(parent_url, claim)
# Idempotency: same canonical hash → skip
existing = await atomic.get_atom_by_source_url(claim_url)
if existing:
return existing, "duplicate"
md = build_claim_markdown(
claim,
parent_title=parent_title,
parent_url=parent_url,
parent_atom_id=parent_id,
)
tag_ids = inherit_tag_ids(parent_atom, claim, resolver)
try:
atom = await atomic.create_atom(
content=md,
source_url=claim_url,
tag_ids=tag_ids,
)
return atom, "created"
except AtomicApiError as e:
log.error("push_claim_failed", url=claim_url, status=e.status, body=e.body[:200])
return None, "error"