Livrare LOT 1 - Didi
This commit is contained in:
commit
5380c3fc63
990 changed files with 133308 additions and 0 deletions
6
ai_platform/modules/didi_brain/extractor/__init__.py
Normal file
6
ai_platform/modules/didi_brain/extractor/__init__.py
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
"""Claim extractor — turns Type/Document atoms into Type/Claim atoms.
|
||||
|
||||
Public API:
|
||||
from extractor.extract import extract_claims_from_atom, ExtractedClaim
|
||||
from extractor.batch import run_batch_extraction
|
||||
"""
|
||||
398
ai_platform/modules/didi_brain/extractor/_extracted.json
Normal file
398
ai_platform/modules/didi_brain/extractor/_extracted.json
Normal file
|
|
@ -0,0 +1,398 @@
|
|||
{
|
||||
"1d7be08c-9fc9-4063-a13b-d20a01b1f24f": {
|
||||
"atom_id": "1d7be08c-9fc9-4063-a13b-d20a01b1f24f",
|
||||
"extracted_at": "2026-04-30T17:39:20.830644+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Moderna_COVID-19_vaccine",
|
||||
"valid_count": 29
|
||||
},
|
||||
"221df1cf-8361-4a07-b720-b7660a062da7": {
|
||||
"atom_id": "221df1cf-8361-4a07-b720-b7660a062da7",
|
||||
"extracted_at": "2026-04-22T08:20:58.548047+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Plandemic",
|
||||
"valid_count": 29
|
||||
},
|
||||
"3c332c68-8908-44a3-8b6f-3e76e83da111": {
|
||||
"atom_id": "3c332c68-8908-44a3-8b6f-3e76e83da111",
|
||||
"extracted_at": "2026-04-22T08:13:47.644503+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 15,
|
||||
"raw_count": 15,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Tiomersal",
|
||||
"valid_count": 15
|
||||
},
|
||||
"3c90743e-04e2-4f2b-a099-8e4c8d7d83b6": {
|
||||
"atom_id": "3c90743e-04e2-4f2b-a099-8e4c8d7d83b6",
|
||||
"extracted_at": "2026-04-30T17:27:01.973084+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 15,
|
||||
"raw_count": 15,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Tiomersal",
|
||||
"valid_count": 15
|
||||
},
|
||||
"5094dfc9-9b8b-42e9-a8f7-b3ad16c39713": {
|
||||
"atom_id": "5094dfc9-9b8b-42e9-a8f7-b3ad16c39713",
|
||||
"extracted_at": "2026-04-30T17:32:40.173050+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Children%27s_Health_Defense",
|
||||
"valid_count": 30
|
||||
},
|
||||
"5857c5f6-e90e-4028-9643-efd0228d8bdc": {
|
||||
"atom_id": "5857c5f6-e90e-4028-9643-efd0228d8bdc",
|
||||
"extracted_at": "2026-04-22T08:27:25.131781+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Moderna_COVID-19_vaccine",
|
||||
"valid_count": 29
|
||||
},
|
||||
"6169e3a9-3574-4e48-8318-3515ae35eb11": {
|
||||
"atom_id": "6169e3a9-3574-4e48-8318-3515ae35eb11",
|
||||
"extracted_at": "2026-04-22T08:34:28.188203+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/MMR_vaccine_and_autism",
|
||||
"valid_count": 29
|
||||
},
|
||||
"61f690f3-0b6a-45a5-8074-798d4afa1453": {
|
||||
"atom_id": "61f690f3-0b6a-45a5-8074-798d4afa1453",
|
||||
"extracted_at": "2026-04-22T08:30:34.956251+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 28,
|
||||
"raw_count": 28,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccine_adverse_event",
|
||||
"valid_count": 28
|
||||
},
|
||||
"685eedb4-62e9-437d-887f-ec81743f6bb8": {
|
||||
"atom_id": "685eedb4-62e9-437d-887f-ec81743f6bb8",
|
||||
"extracted_at": "2026-04-22T08:24:36.031727+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Robert_F._Kennedy_Jr.",
|
||||
"valid_count": 29
|
||||
},
|
||||
"69f16450-60e7-40fd-ae50-dd607514352b": {
|
||||
"atom_id": "69f16450-60e7-40fd-ae50-dd607514352b",
|
||||
"extracted_at": "2026-04-22T08:37:13.179879+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccination",
|
||||
"valid_count": 29
|
||||
},
|
||||
"6b86489d-fc88-4f40-8c81-371e82ace952": {
|
||||
"atom_id": "6b86489d-fc88-4f40-8c81-371e82ace952",
|
||||
"extracted_at": "2026-04-30T17:51:07.975837+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccine",
|
||||
"valid_count": 30
|
||||
},
|
||||
"6e13b816-50c9-4cdd-a9e4-a67b2d65dca0": {
|
||||
"atom_id": "6e13b816-50c9-4cdd-a9e4-a67b2d65dca0",
|
||||
"extracted_at": "2026-04-22T08:38:16.576393+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccine",
|
||||
"valid_count": 30
|
||||
},
|
||||
"6e791e83-8bfd-4aaf-9111-9e77072517d8": {
|
||||
"atom_id": "6e791e83-8bfd-4aaf-9111-9e77072517d8",
|
||||
"extracted_at": "2026-04-22T08:14:35.928834+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 27,
|
||||
"raw_count": 27,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Andrew_Wakefield",
|
||||
"valid_count": 27
|
||||
},
|
||||
"79c1010d-1406-4b37-8ab9-c8e3e6d054ef": {
|
||||
"atom_id": "79c1010d-1406-4b37-8ab9-c8e3e6d054ef",
|
||||
"extracted_at": "2026-04-30T17:41:23.814310+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 28,
|
||||
"raw_count": 28,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccine_adverse_event",
|
||||
"valid_count": 28
|
||||
},
|
||||
"8182687d-9350-4870-818f-5e6e226def88": {
|
||||
"atom_id": "8182687d-9350-4870-818f-5e6e226def88",
|
||||
"extracted_at": "2026-04-22T08:16:30.831673+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Pandemia_de_COVID-19_%C3%AEn_Rom%C3%A2nia",
|
||||
"valid_count": 29
|
||||
},
|
||||
"8516163b-fe3e-448c-9fda-f48155a05327": {
|
||||
"atom_id": "8516163b-fe3e-448c-9fda-f48155a05327",
|
||||
"extracted_at": "2026-04-30T17:37:46.813140+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/COVID-19_vaccine_misinformation_and_hesitancy",
|
||||
"valid_count": 30
|
||||
},
|
||||
"86de36c3-7b60-4120-ab0f-018dfa1c8ba9": {
|
||||
"atom_id": "86de36c3-7b60-4120-ab0f-018dfa1c8ba9",
|
||||
"extracted_at": "2026-04-22T08:32:07.787823+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Anti-vaccine_activism",
|
||||
"valid_count": 30
|
||||
},
|
||||
"8d87dd49-6a8e-4d6d-b15a-1631db0c7153": {
|
||||
"atom_id": "8d87dd49-6a8e-4d6d-b15a-1631db0c7153",
|
||||
"extracted_at": "2026-04-30T17:31:49.206015+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Vaccin",
|
||||
"valid_count": 29
|
||||
},
|
||||
"9272802d-211c-42e4-b064-cb8acc143e1c": {
|
||||
"atom_id": "9272802d-211c-42e4-b064-cb8acc143e1c",
|
||||
"extracted_at": "2026-04-30T17:46:26.293983+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/MMR_vaccine_and_autism",
|
||||
"valid_count": 29
|
||||
},
|
||||
"9c12182c-f2fe-487f-b43d-f97fd8f84984": {
|
||||
"atom_id": "9c12182c-f2fe-487f-b43d-f97fd8f84984",
|
||||
"extracted_at": "2026-04-30T17:43:18.929395+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Anti-vaccine_activism",
|
||||
"valid_count": 30
|
||||
},
|
||||
"9c2c5fbb-cb32-40ab-b6fd-6819c8e49fd2": {
|
||||
"atom_id": "9c2c5fbb-cb32-40ab-b6fd-6819c8e49fd2",
|
||||
"extracted_at": "2026-04-30T17:30:19.630562+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 6,
|
||||
"raw_count": 6,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Vaccinare",
|
||||
"valid_count": 6
|
||||
},
|
||||
"a5d2ce82-0249-4253-9c2d-bddd6853d311": {
|
||||
"atom_id": "a5d2ce82-0249-4253-9c2d-bddd6853d311",
|
||||
"extracted_at": "2026-04-30T17:44:43.113762+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Andrew_Wakefield",
|
||||
"valid_count": 30
|
||||
},
|
||||
"ab56b518-9d13-4de0-bf01-1c89abd087e0": {
|
||||
"atom_id": "ab56b518-9d13-4de0-bf01-1c89abd087e0",
|
||||
"extracted_at": "2026-04-22T08:12:44.236406+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 21,
|
||||
"raw_count": 21,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Variol%C4%83",
|
||||
"valid_count": 21
|
||||
},
|
||||
"abc36af6-e4fe-4c75-b98d-57738f9406ca": {
|
||||
"atom_id": "abc36af6-e4fe-4c75-b98d-57738f9406ca",
|
||||
"extracted_at": "2026-04-22T08:26:19.603769+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/COVID-19_vaccine_misinformation_and_hesitancy",
|
||||
"valid_count": 30
|
||||
},
|
||||
"aef7f81f-c6a8-44e0-859d-812b144380e2": {
|
||||
"atom_id": "aef7f81f-c6a8-44e0-859d-812b144380e2",
|
||||
"extracted_at": "2026-04-30T17:34:27.380347+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Plandemic",
|
||||
"valid_count": 29
|
||||
},
|
||||
"b09275c4-d325-4aec-b730-8fa23e374284": {
|
||||
"atom_id": "b09275c4-d325-4aec-b730-8fa23e374284",
|
||||
"extracted_at": "2026-04-30T17:28:03.929162+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 27,
|
||||
"raw_count": 27,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Andrew_Wakefield",
|
||||
"valid_count": 27
|
||||
},
|
||||
"b569c570-6d19-42a7-b13d-41039c8b4391": {
|
||||
"atom_id": "b569c570-6d19-42a7-b13d-41039c8b4391",
|
||||
"extracted_at": "2026-04-30T17:36:11.186401+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 25,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 5
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Robert_F._Kennedy_Jr.",
|
||||
"valid_count": 25
|
||||
},
|
||||
"b7b39518-eb7b-4d7d-afb4-d6e9ef815094": {
|
||||
"atom_id": "b7b39518-eb7b-4d7d-afb4-d6e9ef815094",
|
||||
"extracted_at": "2026-04-22T08:35:55.179684+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccine_hesitancy",
|
||||
"valid_count": 30
|
||||
},
|
||||
"c7651882-6731-438a-933c-efb827344495": {
|
||||
"atom_id": "c7651882-6731-438a-933c-efb827344495",
|
||||
"extracted_at": "2026-04-22T08:29:18.331921+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Pfizer%E2%80%93BioNTech_COVID-19_vaccine",
|
||||
"valid_count": 30
|
||||
},
|
||||
"c97398f4-819b-46a9-b6fc-f25f6e192340": {
|
||||
"atom_id": "c97398f4-819b-46a9-b6fc-f25f6e192340",
|
||||
"extracted_at": "2026-04-22T08:16:44.511818+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 6,
|
||||
"raw_count": 6,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Vaccinare",
|
||||
"valid_count": 6
|
||||
},
|
||||
"d933d1f5-03a5-4238-b0ea-352d72f0699c": {
|
||||
"atom_id": "d933d1f5-03a5-4238-b0ea-352d72f0699c",
|
||||
"extracted_at": "2026-04-30T17:49:08.873214+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 29,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 1
|
||||
},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccination",
|
||||
"valid_count": 29
|
||||
},
|
||||
"e5e5aac1-addb-4586-a18b-2e004ac8f50f": {
|
||||
"atom_id": "e5e5aac1-addb-4586-a18b-2e004ac8f50f",
|
||||
"extracted_at": "2026-04-30T17:48:08.258151+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Vaccine_hesitancy",
|
||||
"valid_count": 30
|
||||
},
|
||||
"e7e388e8-6b1c-4d48-8565-635fde082c75": {
|
||||
"atom_id": "e7e388e8-6b1c-4d48-8565-635fde082c75",
|
||||
"extracted_at": "2026-04-22T08:33:25.352110+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Andrew_Wakefield",
|
||||
"valid_count": 30
|
||||
},
|
||||
"e9a1c988-01ac-48c8-8da3-a5f3827e0594": {
|
||||
"atom_id": "e9a1c988-01ac-48c8-8da3-a5f3827e0594",
|
||||
"extracted_at": "2026-04-22T08:19:10.267888+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Children%27s_Health_Defense",
|
||||
"valid_count": 30
|
||||
},
|
||||
"e9e72b6f-909c-400a-8d59-0c103c2fac47": {
|
||||
"atom_id": "e9e72b6f-909c-400a-8d59-0c103c2fac47",
|
||||
"extracted_at": "2026-04-30T17:40:36.770904+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 30,
|
||||
"raw_count": 30,
|
||||
"rejected": {},
|
||||
"source_url": "https://en.wikipedia.org/wiki/Pfizer%E2%80%93BioNTech_COVID-19_vaccine",
|
||||
"valid_count": 30
|
||||
},
|
||||
"ecb20b8f-4141-4831-b88a-10ad4c415305": {
|
||||
"atom_id": "ecb20b8f-4141-4831-b88a-10ad4c415305",
|
||||
"extracted_at": "2026-04-22T08:18:19.056060+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 28,
|
||||
"raw_count": 30,
|
||||
"rejected": {
|
||||
"quote_not_in_source": 2
|
||||
},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Vaccin",
|
||||
"valid_count": 28
|
||||
},
|
||||
"f0434eae-41f4-42bf-8459-54bbc32e9b29": {
|
||||
"atom_id": "f0434eae-41f4-42bf-8459-54bbc32e9b29",
|
||||
"extracted_at": "2026-04-30T17:26:38.082088+00:00",
|
||||
"prompt_version": "v1",
|
||||
"pushed_count": 22,
|
||||
"raw_count": 22,
|
||||
"rejected": {},
|
||||
"source_url": "https://ro.wikipedia.org/wiki/Variol%C4%83",
|
||||
"valid_count": 22
|
||||
}
|
||||
}
|
||||
67
ai_platform/modules/didi_brain/extractor/_state.py
Normal file
67
ai_platform/modules/didi_brain/extractor/_state.py
Normal file
|
|
@ -0,0 +1,67 @@
|
|||
"""Persistent state for the claim extractor.
|
||||
|
||||
We track which document atoms have already been processed so re-runs are
|
||||
idempotent. Stored as a single JSON file under extractor/_extracted.json.
|
||||
|
||||
Each entry records the run timestamp, prompt version, and a small summary
|
||||
of what came out, so we can audit later or selectively re-extract if a
|
||||
prompt version changes.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import asdict, dataclass, field
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
_STATE_FILE = Path(__file__).resolve().parent / "_extracted.json"
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class DocExtractionRecord:
|
||||
atom_id: str
|
||||
source_url: str
|
||||
extracted_at: str
|
||||
prompt_version: str
|
||||
raw_count: int # how many claims the LLM returned
|
||||
valid_count: int # how many passed validation
|
||||
pushed_count: int # how many were created in Atomic (excludes dedup hits)
|
||||
rejected: dict[str, int] = field(default_factory=dict)
|
||||
|
||||
|
||||
class ExtractionState:
|
||||
"""Loads / saves the extraction log file."""
|
||||
|
||||
def __init__(self, path: Path | None = None):
|
||||
self._path = path or _STATE_FILE
|
||||
self._records: dict[str, DocExtractionRecord] = {}
|
||||
if self._path.exists():
|
||||
data = json.loads(self._path.read_text(encoding="utf-8"))
|
||||
for atom_id, raw in data.items():
|
||||
self._records[atom_id] = DocExtractionRecord(**raw)
|
||||
|
||||
def has(self, atom_id: str, prompt_version: str) -> bool:
|
||||
rec = self._records.get(atom_id)
|
||||
return rec is not None and rec.prompt_version == prompt_version
|
||||
|
||||
def get(self, atom_id: str) -> DocExtractionRecord | None:
|
||||
return self._records.get(atom_id)
|
||||
|
||||
def upsert(self, record: DocExtractionRecord) -> None:
|
||||
self._records[record.atom_id] = record
|
||||
|
||||
def save(self) -> None:
|
||||
out = {k: asdict(v) for k, v in self._records.items()}
|
||||
self._path.write_text(
|
||||
json.dumps(out, indent=2, ensure_ascii=False, sort_keys=True),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
@property
|
||||
def all(self) -> dict[str, DocExtractionRecord]:
|
||||
return dict(self._records)
|
||||
|
||||
|
||||
def now_iso() -> str:
|
||||
return datetime.now(timezone.utc).isoformat()
|
||||
233
ai_platform/modules/didi_brain/extractor/batch.py
Normal file
233
ai_platform/modules/didi_brain/extractor/batch.py
Normal file
|
|
@ -0,0 +1,233 @@
|
|||
"""Batch orchestrator for claim extraction.
|
||||
|
||||
Walks all Type/Document atoms in Atomic, runs extraction on each, and pushes
|
||||
the resulting claims as new Type/Claim atoms. Idempotent across runs via the
|
||||
state file in extractor/_extracted.json.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from collections.abc import AsyncIterator
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
from urllib.parse import unquote
|
||||
|
||||
from shared.atomic_api import AtomicClient
|
||||
from shared.config import settings
|
||||
from shared.embedding_client import EmbeddingClient # noqa: F401 (future use)
|
||||
from shared.llm_client import LlmClient, LlmError
|
||||
from shared.logging import get_logger
|
||||
from shared.taxonomy import TagResolver
|
||||
|
||||
from extractor._state import DocExtractionRecord, ExtractionState, now_iso
|
||||
from extractor.extract import (
|
||||
PROMPT_VERSION,
|
||||
ExtractionResult,
|
||||
extract_claims_from_atom,
|
||||
)
|
||||
from extractor.push import push_claim
|
||||
|
||||
log = get_logger(__name__)
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class BatchStats:
|
||||
docs_seen: int = 0
|
||||
docs_skipped_already_done: int = 0
|
||||
docs_processed: int = 0
|
||||
docs_failed: int = 0
|
||||
claims_raw: int = 0
|
||||
claims_valid: int = 0
|
||||
claims_created: int = 0
|
||||
claims_duplicate: int = 0
|
||||
claims_error: int = 0
|
||||
rejected_reasons: dict[str, int] = field(default_factory=dict)
|
||||
|
||||
|
||||
# ====================================================== document selection
|
||||
|
||||
|
||||
async def _list_documents_to_process(
|
||||
atomic: AtomicClient, resolver: TagResolver, *, limit: int = 1000
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Return all atoms tagged Type/Document, with their tags inlined.
|
||||
|
||||
We page through /api/atoms?tag_id=<Type/Document> and pull metadata for
|
||||
each, since the list endpoint already returns tags inline.
|
||||
"""
|
||||
type_doc_id = resolver.require("Type/Document")
|
||||
page_size = 50
|
||||
out: list[dict[str, Any]] = []
|
||||
offset = 0
|
||||
while True:
|
||||
result = await atomic.list_atoms(
|
||||
limit=page_size, offset=offset, tag_id=type_doc_id
|
||||
)
|
||||
atoms = result.get("atoms") or result.get("data") or (result if isinstance(result, list) else [])
|
||||
if not atoms:
|
||||
break
|
||||
for a in atoms:
|
||||
out.append(a)
|
||||
if len(out) >= limit:
|
||||
return out
|
||||
if len(atoms) < page_size:
|
||||
break
|
||||
offset += page_size
|
||||
return out
|
||||
|
||||
|
||||
def _title_from_atom(atom: dict[str, Any]) -> str:
|
||||
"""Best-effort title: prefer the Markdown H1 in content, fall back to URL slug."""
|
||||
content = atom.get("content") or ""
|
||||
# Look for the first '# ...' line at the start
|
||||
for line in content.lstrip().splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith("# "):
|
||||
return line[2:].strip()
|
||||
if line:
|
||||
break # first non-empty isn't a header → fall through to URL
|
||||
url = atom.get("source_url") or ""
|
||||
if url:
|
||||
last = url.rstrip("/").rsplit("/", 1)[-1]
|
||||
return unquote(last).replace("_", " ")
|
||||
return atom.get("id", "?")[:8]
|
||||
|
||||
|
||||
def _language_from_atom(atom: dict[str, Any]) -> str:
|
||||
"""Read the Language/<X> tag if present, default 'EN'."""
|
||||
for tag in atom.get("tags") or []:
|
||||
name = tag.get("name", "")
|
||||
# The tag list returns just `name`, not the full path. Languages are
|
||||
# short codes (RO/EN/RU/...) so direct match works.
|
||||
if name in {"RO", "EN", "RU", "UA", "FR", "DE", "ES", "IT", "PL"}:
|
||||
return name
|
||||
return "EN"
|
||||
|
||||
|
||||
# ============================================================== one document
|
||||
|
||||
|
||||
async def process_one(
|
||||
*,
|
||||
llm: LlmClient,
|
||||
atomic: AtomicClient,
|
||||
resolver: TagResolver,
|
||||
state: ExtractionState,
|
||||
atom: dict[str, Any],
|
||||
stats: BatchStats,
|
||||
) -> None:
|
||||
atom_id = atom["id"]
|
||||
if state.has(atom_id, PROMPT_VERSION):
|
||||
stats.docs_skipped_already_done += 1
|
||||
return
|
||||
|
||||
# /api/atoms (list) returns summary objects WITHOUT full content. We have
|
||||
# to fetch the full atom individually to get the body for extraction.
|
||||
full_atom = await atomic.get_atom(atom_id)
|
||||
content = full_atom.get("content") or ""
|
||||
title = _title_from_atom(full_atom)
|
||||
language = _language_from_atom(full_atom)
|
||||
|
||||
if not content:
|
||||
log.warning("doc_no_content", atom_id=atom_id)
|
||||
stats.docs_failed += 1
|
||||
return
|
||||
|
||||
log.info("extracting", atom_id=atom_id[:8], title=title, lang=language, chars=len(content))
|
||||
|
||||
try:
|
||||
result: ExtractionResult = await extract_claims_from_atom(
|
||||
llm,
|
||||
title=title,
|
||||
language=language,
|
||||
content=content,
|
||||
)
|
||||
except LlmError as e:
|
||||
log.error("extraction_failed", atom_id=atom_id, error=str(e), body=(e.body or "")[:300])
|
||||
stats.docs_failed += 1
|
||||
return
|
||||
except Exception as e: # noqa: BLE001
|
||||
log.error("extraction_crashed", atom_id=atom_id, error=f"{type(e).__name__}: {e}")
|
||||
stats.docs_failed += 1
|
||||
return
|
||||
|
||||
stats.docs_processed += 1
|
||||
stats.claims_raw += result.raw_count
|
||||
stats.claims_valid += len(result.valid)
|
||||
for k, v in result.rejected.items():
|
||||
stats.rejected_reasons[k] = stats.rejected_reasons.get(k, 0) + v
|
||||
|
||||
pushed = 0
|
||||
duplicates = 0
|
||||
errors = 0
|
||||
for c in result.valid:
|
||||
_, status = await push_claim(
|
||||
atomic,
|
||||
parent_atom=full_atom,
|
||||
claim=c,
|
||||
parent_title=title,
|
||||
resolver=resolver,
|
||||
)
|
||||
if status == "created":
|
||||
pushed += 1
|
||||
elif status == "duplicate":
|
||||
duplicates += 1
|
||||
else:
|
||||
errors += 1
|
||||
|
||||
stats.claims_created += pushed
|
||||
stats.claims_duplicate += duplicates
|
||||
stats.claims_error += errors
|
||||
|
||||
state.upsert(
|
||||
DocExtractionRecord(
|
||||
atom_id=atom_id,
|
||||
source_url=full_atom.get("source_url", ""),
|
||||
extracted_at=now_iso(),
|
||||
prompt_version=PROMPT_VERSION,
|
||||
raw_count=result.raw_count,
|
||||
valid_count=len(result.valid),
|
||||
pushed_count=pushed,
|
||||
rejected=dict(result.rejected),
|
||||
)
|
||||
)
|
||||
state.save() # save after each doc so a crash doesn't lose progress
|
||||
|
||||
|
||||
# ============================================================ batch entry
|
||||
|
||||
|
||||
async def run_batch(
|
||||
*,
|
||||
limit: int | None = None,
|
||||
only_atom_ids: set[str] | None = None,
|
||||
) -> BatchStats:
|
||||
if not settings.atomic_token:
|
||||
raise RuntimeError("ATOMIC_TOKEN missing")
|
||||
|
||||
resolver = TagResolver()
|
||||
if "Type/Claim" not in resolver:
|
||||
raise RuntimeError("taxonomy not seeded — run scripts/04_seed_taxonomy.py")
|
||||
|
||||
state = ExtractionState()
|
||||
stats = BatchStats()
|
||||
|
||||
async with AtomicClient() as atomic, LlmClient() as llm:
|
||||
docs = await _list_documents_to_process(atomic, resolver, limit=limit or 1000)
|
||||
if only_atom_ids:
|
||||
docs = [d for d in docs if d["id"] in only_atom_ids]
|
||||
stats.docs_seen = len(docs)
|
||||
log.info("batch_start", docs=len(docs), prompt_version=PROMPT_VERSION)
|
||||
|
||||
for atom in docs:
|
||||
await process_one(
|
||||
llm=llm,
|
||||
atomic=atomic,
|
||||
resolver=resolver,
|
||||
state=state,
|
||||
atom=atom,
|
||||
stats=stats,
|
||||
)
|
||||
|
||||
return stats
|
||||
204
ai_platform/modules/didi_brain/extractor/extract.py
Normal file
204
ai_platform/modules/didi_brain/extractor/extract.py
Normal file
|
|
@ -0,0 +1,204 @@
|
|||
"""Single-document claim extraction logic.
|
||||
|
||||
Given one Type/Document atom, this module:
|
||||
|
||||
1. Builds the extraction prompt from the versioned template
|
||||
2. Calls Qwen 397B (REASONING role) for structured JSON output
|
||||
3. Parses + validates each claim:
|
||||
- Required fields present and right types
|
||||
- Stance is in the allowed enum
|
||||
- Confidence above threshold
|
||||
- Quote is a verbatim substring of the source (programmatic check —
|
||||
this is the cheap defense against LLM hallucination)
|
||||
4. Returns a list of ExtractedClaim dataclasses ready for the pusher.
|
||||
|
||||
The pusher (push.py) takes ExtractedClaim and creates a Type/Claim atom in
|
||||
Atomic, with the proper tag inheritance and a stable hash-based source_url.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from shared.config import LlmRole
|
||||
from shared.llm_client import LlmClient, LlmError
|
||||
from shared.logging import get_logger
|
||||
|
||||
log = get_logger(__name__)
|
||||
|
||||
PROMPT_VERSION = "v1"
|
||||
PROMPT_PATH = Path(__file__).resolve().parent / "prompts" / f"claim_extraction_{PROMPT_VERSION}.md"
|
||||
ALLOWED_STANCES = {"ASSERTS", "REPORTS", "REFUTES", "QUESTIONS", "NEUTRAL"}
|
||||
MIN_CONFIDENCE = 0.7
|
||||
MIN_CLAIM_CHARS = 20
|
||||
MIN_QUOTE_CHARS = 10
|
||||
MAX_QUOTE_CHARS = 600
|
||||
MAX_INPUT_CHARS = 120_000 # ~30K tokens, well under Qwen's 262K context
|
||||
|
||||
_PROMPT_TEMPLATE: str | None = None
|
||||
|
||||
|
||||
def _load_prompt() -> str:
|
||||
global _PROMPT_TEMPLATE
|
||||
if _PROMPT_TEMPLATE is None:
|
||||
_PROMPT_TEMPLATE = PROMPT_PATH.read_text(encoding="utf-8")
|
||||
return _PROMPT_TEMPLATE
|
||||
|
||||
|
||||
# ============================================================ data structures
|
||||
|
||||
|
||||
@dataclass(slots=True, frozen=True)
|
||||
class ExtractedClaim:
|
||||
claim: str
|
||||
quote: str
|
||||
stance: str # uppercase: ASSERTS / REPORTS / REFUTES / QUESTIONS / NEUTRAL
|
||||
confidence: float
|
||||
|
||||
def stable_hash(self) -> str:
|
||||
"""8-char SHA-1 of canonicalized claim text — used in source_url fragment."""
|
||||
canonical = " ".join(self.claim.lower().strip().split())
|
||||
return hashlib.sha1(canonical.encode("utf-8")).hexdigest()[:8]
|
||||
|
||||
|
||||
@dataclass(slots=True)
|
||||
class ExtractionResult:
|
||||
"""What `extract_claims_from_atom` returns."""
|
||||
|
||||
raw_count: int # how many items the LLM returned
|
||||
valid: list[ExtractedClaim]
|
||||
rejected: dict[str, int] # reason → count
|
||||
backend: str # which backend served the request
|
||||
|
||||
|
||||
# ============================================================ extraction core
|
||||
|
||||
|
||||
async def extract_claims_from_atom(
|
||||
llm: LlmClient,
|
||||
*,
|
||||
title: str,
|
||||
language: str,
|
||||
content: str,
|
||||
max_tokens_out: int = 6000,
|
||||
) -> ExtractionResult:
|
||||
"""Run the extraction LLM call and validate the results.
|
||||
|
||||
Raises LlmError on transport / JSON parse failure (caller decides whether
|
||||
to retry or skip the document).
|
||||
"""
|
||||
prompt = _load_prompt().replace("{title}", title)
|
||||
prompt = prompt.replace("{language}", language)
|
||||
truncated = content[:MAX_INPUT_CHARS]
|
||||
if len(content) > MAX_INPUT_CHARS:
|
||||
truncated += "\n\n[document truncated for extraction]"
|
||||
prompt = prompt.replace("{content}", truncated)
|
||||
|
||||
result, usage = await llm.chat_json(
|
||||
role=LlmRole.REASONING,
|
||||
system=(
|
||||
"You are a precise information extractor. Output strictly valid "
|
||||
"JSON, no commentary, no markdown fences."
|
||||
),
|
||||
user=prompt,
|
||||
max_tokens=max_tokens_out,
|
||||
temperature=0.0,
|
||||
)
|
||||
|
||||
if not isinstance(result, dict) or "claims" not in result:
|
||||
raise LlmError(
|
||||
f"unexpected response shape: keys={list(result.keys()) if isinstance(result, dict) else type(result).__name__}"
|
||||
)
|
||||
|
||||
raw_claims = result.get("claims") or []
|
||||
if not isinstance(raw_claims, list):
|
||||
raise LlmError(f"`claims` is not a list: {type(raw_claims).__name__}")
|
||||
|
||||
valid: list[ExtractedClaim] = []
|
||||
rejected: dict[str, int] = {}
|
||||
|
||||
norm_source = _normalize_for_match(content)
|
||||
|
||||
for raw in raw_claims:
|
||||
outcome = _parse_one(raw, norm_source)
|
||||
if isinstance(outcome, ExtractedClaim):
|
||||
valid.append(outcome)
|
||||
else:
|
||||
rejected[outcome] = rejected.get(outcome, 0) + 1
|
||||
|
||||
# Dedup within this batch by stable hash
|
||||
seen: set[str] = set()
|
||||
deduped: list[ExtractedClaim] = []
|
||||
for c in valid:
|
||||
h = c.stable_hash()
|
||||
if h in seen:
|
||||
rejected["intra_batch_duplicate"] = rejected.get("intra_batch_duplicate", 0) + 1
|
||||
continue
|
||||
seen.add(h)
|
||||
deduped.append(c)
|
||||
|
||||
return ExtractionResult(
|
||||
raw_count=len(raw_claims),
|
||||
valid=deduped,
|
||||
rejected=rejected,
|
||||
backend=str(usage.get("backend", "")),
|
||||
)
|
||||
|
||||
|
||||
# ============================================================ validation
|
||||
|
||||
|
||||
def _parse_one(raw: Any, norm_source: str) -> ExtractedClaim | str:
|
||||
"""Validate one raw item from the LLM. Returns ExtractedClaim or error reason str."""
|
||||
if not isinstance(raw, dict):
|
||||
return "not_a_dict"
|
||||
|
||||
claim = (raw.get("claim") or "").strip()
|
||||
quote = (raw.get("quote") or "").strip()
|
||||
stance = (raw.get("stance") or "").strip().upper()
|
||||
try:
|
||||
confidence = float(raw.get("confidence", 0))
|
||||
except (TypeError, ValueError):
|
||||
return "bad_confidence_type"
|
||||
|
||||
if len(claim) < MIN_CLAIM_CHARS:
|
||||
return "claim_too_short"
|
||||
if len(quote) < MIN_QUOTE_CHARS:
|
||||
return "quote_too_short"
|
||||
if len(quote) > MAX_QUOTE_CHARS:
|
||||
return "quote_too_long"
|
||||
if stance not in ALLOWED_STANCES:
|
||||
return "bad_stance"
|
||||
if confidence < MIN_CONFIDENCE:
|
||||
return "low_confidence"
|
||||
|
||||
# The critical anti-hallucination check: the quote must actually appear
|
||||
# in the source document (after whitespace normalization).
|
||||
if not _quote_in_source(quote, norm_source):
|
||||
return "quote_not_in_source"
|
||||
|
||||
return ExtractedClaim(
|
||||
claim=claim,
|
||||
quote=quote,
|
||||
stance=stance,
|
||||
confidence=confidence,
|
||||
)
|
||||
|
||||
|
||||
_WHITESPACE_RE = re.compile(r"\s+")
|
||||
|
||||
|
||||
def _normalize_for_match(s: str) -> str:
|
||||
"""Collapse runs of whitespace and normalize quote chars for substring match."""
|
||||
s = s.replace("\u2018", "'").replace("\u2019", "'")
|
||||
s = s.replace("\u201c", '"').replace("\u201d", '"')
|
||||
s = s.replace("\u2013", "-").replace("\u2014", "-")
|
||||
return _WHITESPACE_RE.sub(" ", s).strip()
|
||||
|
||||
|
||||
def _quote_in_source(quote: str, norm_source: str) -> bool:
|
||||
return _normalize_for_match(quote) in norm_source
|
||||
|
|
@ -0,0 +1,65 @@
|
|||
You are an expert fact extractor for a knowledge graph that supports
|
||||
disinformation analysis. Your job is to read one source document and extract
|
||||
every distinct, verifiable factual claim it makes.
|
||||
|
||||
# What counts as a "claim"
|
||||
|
||||
A claim is a **specific, self-contained, verifiable proposition**. It must be:
|
||||
|
||||
- Concrete enough that someone could check it against evidence
|
||||
- Self-contained: understandable without reading surrounding text
|
||||
- About facts, events, findings, or attributed statements — not opinions
|
||||
|
||||
A claim is **NOT**:
|
||||
|
||||
- An opinion or value judgment ("the policy was misguided")
|
||||
- A vague generality ("many people worry", "some scientists think")
|
||||
- A definition or terminology explanation
|
||||
- A question or recommendation
|
||||
- Pure background description without a specific assertion
|
||||
|
||||
# Stance classification
|
||||
|
||||
For each claim, decide how the SOURCE itself treats it:
|
||||
|
||||
- **ASSERTS** — the source presents the claim as factual / a finding it stands behind
|
||||
- **REPORTS** — the source describes someone else's claim without endorsing or refuting it
|
||||
- **REFUTES** — the source explicitly disagrees with, debunks, or corrects the claim
|
||||
- **QUESTIONS** — the source raises doubts or uncertainty without fully refuting
|
||||
- **NEUTRAL** — purely descriptive context with no editorial stance
|
||||
|
||||
# Output rules — STRICT
|
||||
|
||||
You MUST output ONLY a single JSON object with this exact shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"claims": [
|
||||
{
|
||||
"claim": "<canonical claim text, in the SAME LANGUAGE as the source>",
|
||||
"quote": "<EXACT verbatim substring from the source where this claim appears>",
|
||||
"stance": "ASSERTS|REPORTS|REFUTES|QUESTIONS|NEUTRAL",
|
||||
"confidence": 0.0-1.0
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
Hard rules:
|
||||
|
||||
1. The `quote` MUST be a character-perfect substring of the source. Copy-paste it. Do not paraphrase, translate, or add ellipses inside it.
|
||||
2. The `claim` must be in the SAME language as the source document. If the source is Romanian, the claim must be in Romanian. Do NOT translate.
|
||||
3. Quote at minimum one full sentence; quote at most ~300 characters.
|
||||
4. Extract up to **30** claims, prioritizing the most consequential and contestable ones. Skip duplicates and trivia.
|
||||
5. Confidence reflects how clean and verifiable the claim is. Use 0.7-0.85 for ordinary factual claims, 0.85-0.95 for well-supported specific findings, below 0.7 for claims you are unsure about (these will be filtered out).
|
||||
6. Do NOT add any text before or after the JSON. No markdown fences. No commentary. Just the JSON object.
|
||||
7. If the document has no extractable factual claims, return `{"claims": []}`.
|
||||
|
||||
# Source
|
||||
|
||||
Title: {title}
|
||||
Language: {language}
|
||||
|
||||
# Source content
|
||||
|
||||
{content}
|
||||
124
ai_platform/modules/didi_brain/extractor/push.py
Normal file
124
ai_platform/modules/didi_brain/extractor/push.py
Normal file
|
|
@ -0,0 +1,124 @@
|
|||
"""Push validated ExtractedClaim objects into Atomic as Type/Claim atoms.
|
||||
|
||||
Each claim becomes a tiny atom with:
|
||||
- source_url = `{parent_url}#claim={hash8}` so it dedups idempotently and
|
||||
so `parent_url = source_url.split("#")[0]` is trivial to recover later
|
||||
- tag inheritance from the parent (Country, Topic, SourceType, Credibility,
|
||||
Language) plus our two new tags: Type/Claim and Stance/<X>
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from shared.atomic_api import AtomicApiError, AtomicClient
|
||||
from shared.logging import get_logger
|
||||
from shared.taxonomy import TagResolver
|
||||
|
||||
from extractor.extract import ExtractedClaim
|
||||
|
||||
log = get_logger(__name__)
|
||||
|
||||
_STANCE_PATH = {
|
||||
"ASSERTS": "Stance/Asserts",
|
||||
"REPORTS": "Stance/Reports",
|
||||
"REFUTES": "Stance/Refutes",
|
||||
"QUESTIONS": "Stance/Questions",
|
||||
"NEUTRAL": "Stance/Neutral",
|
||||
}
|
||||
|
||||
|
||||
def build_claim_markdown(
|
||||
claim: ExtractedClaim,
|
||||
*,
|
||||
parent_title: str,
|
||||
parent_url: str,
|
||||
parent_atom_id: str,
|
||||
) -> str:
|
||||
"""Render a claim atom's body as Markdown.
|
||||
|
||||
The structure is intentionally consistent so it can be parsed back later
|
||||
by Didi or by re-indexing scripts.
|
||||
"""
|
||||
return (
|
||||
f"# Claim\n\n"
|
||||
f"{claim.claim}\n\n"
|
||||
f"## Quote\n"
|
||||
f"> {claim.quote}\n\n"
|
||||
f"## Source\n"
|
||||
f"- Document: [{parent_title}]({parent_url})\n"
|
||||
f"- Parent atom: `{parent_atom_id}`\n"
|
||||
f"- Stance in source: {claim.stance}\n"
|
||||
f"- Extraction confidence: {claim.confidence:.2f}\n"
|
||||
)
|
||||
|
||||
|
||||
def build_claim_url(parent_url: str, claim: ExtractedClaim) -> str:
|
||||
"""Stable hash-based URL fragment so re-extraction dedupes naturally."""
|
||||
base = parent_url.split("#", 1)[0]
|
||||
return f"{base}#claim={claim.stable_hash()}"
|
||||
|
||||
|
||||
def inherit_tag_ids(
|
||||
parent_atom: dict[str, Any],
|
||||
claim: ExtractedClaim,
|
||||
resolver: TagResolver,
|
||||
) -> list[str]:
|
||||
"""Build the tag-id list for a new claim atom.
|
||||
|
||||
Inherits all parent tags except Type/Document, and adds Type/Claim plus
|
||||
the appropriate Stance/<X>.
|
||||
"""
|
||||
type_doc_id = resolver.require("Type/Document")
|
||||
type_claim_id = resolver.require("Type/Claim")
|
||||
stance_id = resolver.require(_STANCE_PATH[claim.stance])
|
||||
|
||||
parent_tag_ids = [
|
||||
t["id"] for t in (parent_atom.get("tags") or []) if t.get("id") != type_doc_id
|
||||
]
|
||||
return parent_tag_ids + [type_claim_id, stance_id]
|
||||
|
||||
|
||||
async def push_claim(
|
||||
atomic: AtomicClient,
|
||||
*,
|
||||
parent_atom: dict[str, Any],
|
||||
claim: ExtractedClaim,
|
||||
parent_title: str,
|
||||
resolver: TagResolver,
|
||||
) -> tuple[dict[str, Any] | None, str]:
|
||||
"""Create one claim atom in Atomic. Returns (atom_dict, status).
|
||||
|
||||
Status is one of:
|
||||
- "created": new atom was created
|
||||
- "duplicate": same hash already exists, skipped
|
||||
- "error": creation failed (atom_dict is None)
|
||||
"""
|
||||
parent_url = parent_atom.get("source_url") or ""
|
||||
parent_id = parent_atom.get("id") or ""
|
||||
|
||||
claim_url = build_claim_url(parent_url, claim)
|
||||
|
||||
# Idempotency: same canonical hash → skip
|
||||
existing = await atomic.get_atom_by_source_url(claim_url)
|
||||
if existing:
|
||||
return existing, "duplicate"
|
||||
|
||||
md = build_claim_markdown(
|
||||
claim,
|
||||
parent_title=parent_title,
|
||||
parent_url=parent_url,
|
||||
parent_atom_id=parent_id,
|
||||
)
|
||||
tag_ids = inherit_tag_ids(parent_atom, claim, resolver)
|
||||
|
||||
try:
|
||||
atom = await atomic.create_atom(
|
||||
content=md,
|
||||
source_url=claim_url,
|
||||
tag_ids=tag_ids,
|
||||
)
|
||||
return atom, "created"
|
||||
except AtomicApiError as e:
|
||||
log.error("push_claim_failed", url=claim_url, status=e.status, body=e.body[:200])
|
||||
return None, "error"
|
||||
Loading…
Add table
Add a link
Reference in a new issue