didi-lot1-ai/ai_platform/modules/didi_brain/brain_api/db.py

352 lines
15 KiB
Python

"""PostgreSQL connection pool + schema bootstrap for brain-side caches.
Brain owns a small relational layer alongside the Atomic knowledge graph for
things that don't fit as atoms (verification caches keyed by multiple fields
with TTL and per-column indexes). We connect directly to the same Postgres
instance atomic-server uses, but keep our tables in the `public` schema with
a `brain_` prefix so they're easy to spot and never collide with Atomic's.
"""
from __future__ import annotations
import asyncpg
from shared.config import settings
from shared.logging import get_logger
log = get_logger(__name__)
_SCHEMA_SQL = """
CREATE TABLE IF NOT EXISTS brain_verification_cache (
id uuid PRIMARY KEY DEFAULT gen_random_uuid(),
claim_hash text NOT NULL,
tier text NOT NULL CHECK (tier IN ('free', 'premium')),
-- evidence_hash and evidence_urls are METADATA only. The URLs brain
-- returns at gather time may differ from what backend wrote cache with
-- (corpus drift, ranker tie-breaks). We still store them so backend can
-- compare overlap and decide if cache is applicable to the current
-- evidence set. Lookup key is (claim_hash, tier) — one row per claim-
-- tier pair, UPSERT last-writer-wins.
evidence_hash text NOT NULL,
evidence_urls jsonb NOT NULL,
model text,
prompt_hash text NOT NULL,
framework_version text,
schema_name text NOT NULL DEFAULT 'didi-v1',
verification_raw jsonb,
verification_processed jsonb NOT NULL,
created_at timestamptz NOT NULL DEFAULT now(),
updated_at timestamptz NOT NULL DEFAULT now(),
expires_at timestamptz NOT NULL
);
-- v1 schema had UNIQUE (claim_hash, evidence_hash, tier). That made the
-- cache unreachable because evidence URLs at read time rarely match write
-- time. v2 = UNIQUE (claim_hash, tier) + new evidence_urls jsonb column.
-- Migrations are idempotent.
ALTER TABLE brain_verification_cache
ADD COLUMN IF NOT EXISTS evidence_urls jsonb NOT NULL DEFAULT '[]'::jsonb;
DO $$
BEGIN
IF EXISTS (
SELECT 1 FROM pg_constraint
WHERE conname = 'brain_verification_cache_claim_hash_evidence_hash_tier_key'
) THEN
ALTER TABLE brain_verification_cache
DROP CONSTRAINT brain_verification_cache_claim_hash_evidence_hash_tier_key;
END IF;
IF NOT EXISTS (
SELECT 1 FROM pg_constraint
WHERE conname = 'brain_verification_cache_claim_hash_tier_key'
) THEN
ALTER TABLE brain_verification_cache
ADD CONSTRAINT brain_verification_cache_claim_hash_tier_key
UNIQUE (claim_hash, tier);
END IF;
END$$;
CREATE INDEX IF NOT EXISTS idx_bvc_lookup
ON brain_verification_cache (claim_hash, tier);
CREATE INDEX IF NOT EXISTS idx_bvc_expires
ON brain_verification_cache (expires_at);
CREATE INDEX IF NOT EXISTS idx_bvc_prompt
ON brain_verification_cache (prompt_hash);
-- Drop the old 3-column lookup index if it exists.
DROP INDEX IF EXISTS idx_bvc_lookup_v1;
-- ============================================================================
-- brain_analysis_atom — cache for full-component LLM results (techniques, ai_tampered).
-- One row per (content_hash, component, prompt_hash). Tier is stored on the row
-- but NOT part of the unique key: write only happens for tier='premium', read
-- is tier-agnostic so free users benefit from premium-cached results.
--
-- 3 cache tiers:
-- gold — human_validated=true (set by didi moderation HIL flow). Survives
-- prompt change. Returned at maximum confidence.
-- silver — LLM result, llm_confidence >= threshold. Default for fresh writes.
-- bronze — LLM result, llm_confidence < threshold. Stored for audit but
-- NEVER served on lookup.
-- ============================================================================
CREATE TABLE IF NOT EXISTS brain_analysis_atom (
atom_id bigserial PRIMARY KEY,
content_hash text NOT NULL,
content_preview text,
component text NOT NULL CHECK (component IN ('techniques', 'ai_tampered', 'claims')),
tier text NOT NULL CHECK (tier IN ('free', 'premium')),
prompt_hash text NOT NULL,
framework_version text,
model_used text,
result_processed jsonb NOT NULL,
result_raw jsonb,
llm_confidence numeric,
cache_tier text NOT NULL DEFAULT 'silver'
CHECK (cache_tier IN ('gold', 'silver', 'bronze')),
human_validated boolean NOT NULL DEFAULT false,
human_corrections jsonb,
validator_user_id text,
validated_at timestamptz,
hit_count integer NOT NULL DEFAULT 0,
last_hit_at timestamptz,
created_at timestamptz NOT NULL DEFAULT now(),
updated_at timestamptz NOT NULL DEFAULT now(),
expires_at timestamptz, -- NULL = never expires (gold)
UNIQUE (content_hash, component, prompt_hash)
);
CREATE INDEX IF NOT EXISTS idx_baa_lookup
ON brain_analysis_atom (content_hash, component);
CREATE INDEX IF NOT EXISTS idx_baa_gold
ON brain_analysis_atom (component, cache_tier)
WHERE cache_tier = 'gold';
CREATE INDEX IF NOT EXISTS idx_baa_expires
ON brain_analysis_atom (expires_at)
WHERE expires_at IS NOT NULL;
CREATE INDEX IF NOT EXISTS idx_baa_prompt
ON brain_analysis_atom (prompt_hash);
-- ============================================================================
-- Volatility-aware caching extension (2026-05-04)
-- ============================================================================
-- Per-row volatility classification + topic tagging + audit history. Driven by
-- LLM classifier at write time (services/classifier.py) and used by:
-- - lookup paths (services/cache_judge.py) → confidence decay, NLI judge
-- - daily auditor (didibrain-auditor) → consecutive_audit_passes tracking
-- - breaking news watcher (didibrain-breaking-watcher) → topic-based mass invalidation
-- All ALTERs are idempotent — safe on every connect.
-- ============================================================================
ALTER TABLE brain_analysis_atom
ADD COLUMN IF NOT EXISTS volatility text
CHECK (volatility IN ('volatile', 'evolving', 'stable')),
ADD COLUMN IF NOT EXISTS topic_codes text[] DEFAULT ARRAY[]::text[],
ADD COLUMN IF NOT EXISTS entity_bindings jsonb DEFAULT '[]'::jsonb,
ADD COLUMN IF NOT EXISTS ttl_hours_used integer,
ADD COLUMN IF NOT EXISTS last_audited_at timestamptz,
ADD COLUMN IF NOT EXISTS audit_history jsonb DEFAULT '[]'::jsonb,
ADD COLUMN IF NOT EXISTS consecutive_audit_passes integer DEFAULT 0;
ALTER TABLE brain_verification_cache
ADD COLUMN IF NOT EXISTS volatility text
CHECK (volatility IN ('volatile', 'evolving', 'stable')),
ADD COLUMN IF NOT EXISTS topic_codes text[] DEFAULT ARRAY[]::text[],
ADD COLUMN IF NOT EXISTS entity_bindings jsonb DEFAULT '[]'::jsonb,
ADD COLUMN IF NOT EXISTS ttl_hours_used integer,
ADD COLUMN IF NOT EXISTS last_audited_at timestamptz,
ADD COLUMN IF NOT EXISTS audit_history jsonb DEFAULT '[]'::jsonb,
ADD COLUMN IF NOT EXISTS consecutive_audit_passes integer DEFAULT 0;
-- GIN indexes on topic_codes for fast topic-scoped invalidation/audit queries.
CREATE INDEX IF NOT EXISTS idx_baa_topics
ON brain_analysis_atom USING GIN (topic_codes);
CREATE INDEX IF NOT EXISTS idx_bvc_topics
ON brain_verification_cache USING GIN (topic_codes);
-- Partial indexes targeting audit-eligible rows (gold+silver, non-stable).
-- Auditor cron queries these to find atoms due for re-verification.
CREATE INDEX IF NOT EXISTS idx_baa_audit_due
ON brain_analysis_atom (last_audited_at NULLS FIRST, volatility)
WHERE cache_tier IN ('gold', 'silver') AND volatility IS NOT NULL AND volatility != 'stable';
CREATE INDEX IF NOT EXISTS idx_bvc_audit_due
ON brain_verification_cache (last_audited_at NULLS FIRST, volatility)
WHERE volatility IS NOT NULL AND volatility != 'stable';
-- ============================================================================
-- brain_fact_status — current truth value for entity-predicate-object triples.
-- Pilon 11 (versioned facts). Populated by:
-- - extractor pipeline (services/fact_status.py::extract_facts_from_claim)
-- when ingesting new claim atoms
-- - moderator overrides (admin endpoint PATCH /v1/fact_status/{id})
-- - breaking news watcher (when LLM detects fact change in fresh article)
-- Read by:
-- - gather lookup → if any bound fact is invalid, treat cache as stale_evidence
-- - dashboard fact browser (Phase D2)
-- ============================================================================
CREATE TABLE IF NOT EXISTS brain_fact_status (
fact_id bigserial PRIMARY KEY,
subject text NOT NULL, -- "Vladimir Putin"
predicate text NOT NULL, -- "is_president_of"
object text NOT NULL, -- "Russia"
canonical_form text NOT NULL, -- "Vladimir Putin is_president_of Russia"
canonical_form_hash text NOT NULL, -- sha256(canonical_form)[:32]
current_truth boolean, -- TRUE / FALSE / NULL=unknown
current_version_id bigint, -- non-FK pointer to brain_fact_version (avoid circular FK)
current_confidence numeric, -- 0-100, from latest LLM judgment
last_verified_at timestamptz,
last_evidence_urls jsonb DEFAULT '[]'::jsonb,
volatility text CHECK (volatility IN ('volatile', 'evolving', 'stable')),
topic_codes text[] DEFAULT ARRAY[]::text[],
-- Scheduling: auditor picks up rows where next_check_at < now()
next_check_at timestamptz NOT NULL DEFAULT now(),
check_interval_hours integer NOT NULL DEFAULT 24,
-- HIL trail
moderator_locked boolean NOT NULL DEFAULT false, -- true = audit cron must NOT auto-update
moderator_user_id text,
moderator_notes text,
created_at timestamptz NOT NULL DEFAULT now(),
updated_at timestamptz NOT NULL DEFAULT now(),
UNIQUE (canonical_form_hash)
);
CREATE INDEX IF NOT EXISTS idx_bfs_subject_predicate
ON brain_fact_status (subject, predicate);
CREATE INDEX IF NOT EXISTS idx_bfs_check_due
ON brain_fact_status (next_check_at)
WHERE moderator_locked = false;
CREATE INDEX IF NOT EXISTS idx_bfs_topics
ON brain_fact_status USING GIN (topic_codes);
CREATE INDEX IF NOT EXISTS idx_bfs_truth
ON brain_fact_status (current_truth)
WHERE current_truth IS NOT NULL;
-- ============================================================================
-- brain_fact_version — temporal versioning for facts.
-- Each row = one truth assertion valid in a [valid_from, valid_to) window.
-- valid_to IS NULL means "currently in force". When a fact changes, the active
-- version gets valid_to=now() and a new version is opened.
-- ============================================================================
CREATE TABLE IF NOT EXISTS brain_fact_version (
version_id bigserial PRIMARY KEY,
fact_id bigint NOT NULL REFERENCES brain_fact_status(fact_id) ON DELETE CASCADE,
truth_value boolean NOT NULL,
confidence numeric, -- 0-100
valid_from timestamptz NOT NULL,
valid_to timestamptz, -- NULL = current
source_atom_ids text[] DEFAULT ARRAY[]::text[], -- Atomic atom IDs supporting this version
evidence_urls jsonb DEFAULT '[]'::jsonb,
llm_reasoning text, -- LLM justification for this assertion
created_by text NOT NULL DEFAULT 'auto', -- 'auto' | 'moderator' | 'breaking_news_watcher'
moderator_user_id text, -- if created_by='moderator'
notes text,
created_at timestamptz NOT NULL DEFAULT now()
);
CREATE INDEX IF NOT EXISTS idx_bfv_fact
ON brain_fact_version (fact_id, valid_from DESC);
-- Partial index for currently-active versions (the most common lookup pattern).
CREATE INDEX IF NOT EXISTS idx_bfv_current
ON brain_fact_version (fact_id)
WHERE valid_to IS NULL;
-- ============================================================================
-- Audit log table (small, for invalidation/promotion/moderator actions).
-- Mirrors the lightweight audit pattern used by didi-admin's audit_log.
-- ============================================================================
CREATE TABLE IF NOT EXISTS brain_audit_log (
log_id bigserial PRIMARY KEY,
action text NOT NULL, -- 'invalidate' | 'promote_gold' | 'fact_override' | 'audit_demote'
target_table text NOT NULL, -- 'brain_analysis_atom' | 'brain_verification_cache' | 'brain_fact_status'
target_id text NOT NULL,
actor text, -- 'auditor' | 'breaking_watcher' | keycloak_id
payload jsonb DEFAULT '{}'::jsonb,
created_at timestamptz NOT NULL DEFAULT now()
);
CREATE INDEX IF NOT EXISTS idx_bal_target
ON brain_audit_log (target_table, target_id);
CREATE INDEX IF NOT EXISTS idx_bal_action_time
ON brain_audit_log (action, created_at DESC);
"""
class Database:
"""Thin wrapper over an asyncpg pool with a migration hook."""
def __init__(self, dsn: str) -> None:
self._dsn = dsn
self._pool: asyncpg.Pool | None = None
async def connect(self) -> None:
if self._pool is not None:
return
self._pool = await asyncpg.create_pool(
self._dsn,
min_size=1,
max_size=5,
command_timeout=10.0,
)
async with self._pool.acquire() as conn:
# pgcrypto for gen_random_uuid() — atomic-server usually enables it
# already via its own migrations, but we do it idempotently just
# in case brain is the first consumer on a fresh volume.
await conn.execute("CREATE EXTENSION IF NOT EXISTS pgcrypto")
await conn.execute(_SCHEMA_SQL)
log.info("brain_db_ready", dsn=self._redacted_dsn())
async def close(self) -> None:
if self._pool is not None:
await self._pool.close()
self._pool = None
@property
def pool(self) -> asyncpg.Pool:
if self._pool is None:
raise RuntimeError("brain_db not connected — call connect() first")
return self._pool
def _redacted_dsn(self) -> str:
# Hide password in logs
import re
return re.sub(r":([^@:/]+)@", ":***@", self._dsn)
# Singleton — wired up in app lifespan
db = Database(settings.postgres_dsn)