Livrare LOT 1 - Didi

This commit is contained in:
Dezvoltari Evotech 2026-06-25 14:13:25 -07:00
commit 5380c3fc63
990 changed files with 133308 additions and 0 deletions

View file

@ -0,0 +1,366 @@
# Integrare în aplicația ta LLM
Document care arată **EXACT** cum apelezi acest serviciu din aplicația ta
care folosește deja un LLM multimodal (Qwen Vision, GPT-4V, Claude Sonnet).
## Pattern de bază
```
┌─────────────────────────┐
│ User uploadează video │
│ în aplicația ta │
└───────────┬─────────────┘
├─ Pas 1: Trimite videoul la Forensic Features API
│ → primești evidence_text + base64 imagini
├─ Pas 2: Construiește prompt-ul TĂU existent
│ (cu tipologiile tale, instrucțiunile tale)
│ + APPEND evidence_text
├─ Pas 3: Atașează la apelul LLM:
│ - imaginile originale ale userului
│ - imaginile noastre (heatmap-uri)
└─ Pas 4: LLM-ul tău returnează verdictul
cu signal îmbogățit de la noi
```
## Exemple complete
### Python — Qwen Vision (compatible cu OpenAI ChatCompletions API)
```python
import requests
import base64
from pathlib import Path
FORENSIC_API = "http://localhost:8080"
QWEN_API = "http://your-qwen-host:14011/v1/chat/completions"
def analyze_video(video_path: str, your_typologies: list[str]) -> dict:
"""
Apelează Forensic Features API → construiește prompt → apelează Qwen.
"""
# ── Pas 1: Forensic Features API ──────────────────────────────────
with open(video_path, "rb") as f:
resp = requests.post(
f"{FORENSIC_API}/api/forensic-evidence",
files={"video": f},
data={
"encode_images": "1", # cu base64 pentru atașare directă
# Optional: "modules": "m25,m27,m28" pt doar 3 module
},
timeout=300,
)
resp.raise_for_status()
forensic = resp.json()
# ── Pas 2: Construiește mesajul multimodal pentru LLM ─────────────
# PROMPT-UL TĂU EXISTENT — așa cum îl ai acum
your_prompt = f"""
Ești un analist video forensic. Analizează acest video și verifică:
Tipologii de elemente vizuale de căutat:
{chr(10).join(f"- {t}" for t in your_typologies)}
Răspunde structurat...
"""
# Append evidence-ul nostru
augmented_prompt = your_prompt + "\n\n" + forensic["evidence_text"]
# ── Pas 3: Construiește content multimodal ───────────────────────
# Lista de imagini originale ale userului (din aplicația ta) +
# imaginile noastre forensice (heatmap-uri, plot-uri)
content = [
{"type": "text", "text": augmented_prompt},
]
# Imaginile TALE existente — pe care le aveai deja în pipeline
your_keyframes = extract_keyframes_yourself(video_path) # funcția ta existentă
for img_path in your_keyframes:
img_b64 = base64.b64encode(open(img_path, "rb").read()).decode()
content.append({
"type": "image_url",
"image_url": {"url": f"data:image/jpeg;base64,{img_b64}"},
})
# Imaginile NOASTRE forensice — heatmap-uri zone suspect, lighting arrows
for img in forensic["images"]:
if "data_url" in img:
content.append({
"type": "image_url",
"image_url": {"url": img["data_url"]},
})
# ── Pas 4: Apelează LLM-ul tău ───────────────────────────────────
qwen_payload = {
"model": "Qwen3.5-397B-A17B",
"messages": [
{"role": "system", "content": "Ești analist forensic. Răspunzi în JSON valid."},
{"role": "user", "content": content},
],
"max_tokens": 1500,
"temperature": 0.1,
}
qwen_resp = requests.post(QWEN_API, json=qwen_payload, timeout=120)
qwen_resp.raise_for_status()
# Parse JSON din răspuns
import json
llm_text = qwen_resp.json()["choices"][0]["message"]["content"]
verdict = json.loads(llm_text)
return {
"verdict": verdict, # ce decide LLM-ul tău
"forensic_evidence": forensic, # pentru debug / audit
"augmented_prompt": augmented_prompt, # pentru replicare
}
# Usage
result = analyze_video(
"user_uploaded.mp4",
your_typologies=["face_swap_visible", "background_anomaly", "logo_overlay"],
)
print(result["verdict"])
```
### JavaScript / Node — direct fetch
```javascript
async function analyzeWithForensic(videoFile) {
// Pas 1: Forensic Features
const fd = new FormData();
fd.append("video", videoFile);
fd.append("encode_images", "1");
const forensicResp = await fetch("http://localhost:8080/api/forensic-evidence", {
method: "POST",
body: fd,
});
const forensic = await forensicResp.json();
// Pas 2: Construiește content multimodal pentru LLM-ul tău
const content = [
{ type: "text", text: yourExistingPrompt + "\n\n" + forensic.evidence_text }
];
// Imaginile tale + ale noastre
for (const img of yourKeyframes) {
content.push({ type: "image_url", image_url: { url: img.dataUrl } });
}
for (const img of forensic.images) {
if (img.data_url) {
content.push({ type: "image_url", image_url: { url: img.data_url } });
}
}
// Pas 3: Apelează LLM-ul tău
const llmResp = await callYourLLM({ content });
return llmResp;
}
```
## Pattern async pentru video lung
Pe video >30 secunde, procesarea poate dura 1-5 minute. Folosește **async mode**:
```python
import time
def analyze_long_video_async(video_path):
# Submit
with open(video_path, "rb") as f:
resp = requests.post(
f"{FORENSIC_API}/api/forensic-evidence",
files={"video": f},
data={"async_mode": "1"},
)
job_id = resp.json()["job_id"]
print(f"Job submitted: {job_id}")
# Poll
while True:
status_resp = requests.get(f"{FORENSIC_API}/api/status/{job_id}")
status = status_resp.json()
print(f"Status: {status['status']} — {status['progress']}")
if status["status"] == "done":
break
if status["status"] == "error":
raise Exception(f"Forensic pipeline failed: {status}")
time.sleep(5)
# Retrieve
result_resp = requests.get(f"{FORENSIC_API}/api/result/{job_id}")
return result_resp.json()
```
## Cum interpretează LLM-ul tău evidence-ul
Când inserezi `evidence_text` în prompt, LLM-ul vede ceva de genul:
```
FORENSIC EVIDENCE (objective measurements you cannot recompute)
======================================================================
OVERALL VERDICT: FAKE (score=0.67, confidence=0.78)
Detectors active: 4, no signal: 0, disagreement: 0.005
Individual detectors:
----------------------------------------------------------------------
[m25 Physiology] score=0.63 (INCERT) confidence=0.41 contrib=+0.091
- Pulse: 82 BPM, SNR=1.1 dB (no plausible cardiac signal)
- Blink count: 1 over 5.1s (natural)
Visuals: m25_pulse_signal.png, m25_blink_timeline.png
[m27 AI-Generated Image Detector] score=0.78 (FAKE) confidence=1.00 contrib=+0.231
- JPEG-recon: 0.78 (above 0.65 = AI suspect)
- NPR: 0.48
Visuals: m27_score_timeline.png
[m28 Forgery Localization Heatmap] score=0.62 (INCERT) confidence=1.00 contrib=+0.237
- Forgery score: 0.62 (peak=0.90, boundary_mean=0.137) — intermediate
- Frames with face: 5/5
Visuals: m28_heatmap_0000.png, ..., m28_heatmap_0040.png
[m29 Lighting 3D Consistency] score=0.61 (INCERT) confidence=0.76
- Face vs scene lighting: 95° (mismatch >90°, suspect compus)
- Catchlights: nedetectabile (ochi închiși/ochelari/rezoluție mică)
Visuals: m29_lighting_0000.png, ..., m29_lighting_0030.png
======================================================================
HOW TO USE FORENSIC EVIDENCE ABOVE:
- These are objective numerical measurements that you CANNOT recompute from
images alone. They are produced by classical signal-processing detectors...
- For each detector that flags FAKE, search the keyframes for the visual
artifact that explains the score...
======================================================================
```
LLM-ul are toate aceste informații + imaginile reale + tipologiile tale.
Combinat cu reasoning-ul lui semantic, decide singur cu signal MULT mai bogat
decât doar din imagine.
## Best practices
### 1. Cache rezultatul forensic per video
Forensic features sunt **deterministe** pe același video. Cache prin
hash SHA256 al fișierului — economie de zeci de secunde per request repetat.
```python
import hashlib
def video_hash(path):
with open(path, "rb") as f:
return hashlib.sha256(f.read()).hexdigest()[:16]
# Cache in Redis sau SQLite cu key = video_hash
```
### 2. Selectează module relevante per tip conținut
Nu rula toate cele 5 mereu:
- **Imagine statică** (jpg/png) → m27 + m28 (rest sunt no-op temporale)
- **Talking head video** → m25 + m26 + m28 + m29
- **AI-generated landscape** (fără față) → m27 doar
- **Screen recording** → m24 (din pipeline vechi v3, NU în această versiune)
Setezi cu `-F "modules=m27,m28"`.
### 3. Truncare evidence_text pe LLM cu context mic
Dacă LLM-ul tău are context window mic (<8K), poți cere doar `summary`:
```python
# În prompt, în loc de evidence_text complet (1500-3000 chars), folosește:
short_evidence = f"""
Forensic signals on this video:
- Overall: {forensic['fusion']['label']} (score={forensic['fusion']['score']:.2f})
- m25 Physiology: {forensic['modules']['m25']['summary']['primary_label']}
- m27 AI Detector: {forensic['modules']['m27']['summary']['primary_label']}
- m28 Blending: {forensic['modules']['m28']['summary']['primary_label']}
"""
```
### 4. Atașează DOAR cele mai relevante PNG-uri
Pe LLM cu limite multimodal (4-6 imagini per call), nu trimite toate 12-15
din output. Filtrează:
```python
# Doar PNG-uri din module flagged FAKE
relevant_images = [
img for img in forensic["images"]
if forensic["modules"][img["tool_id"]]["summary"]["primary_label"] == "FAKE"
]
```
### 5. Loghează verdictul + evidence pentru audit
```python
# Salvează atât verdictul LLM cât și evidence-ul nostru
# pentru audit ulterior și calibrare
audit_log.write({
"video_id": video_hash(video_path),
"forensic_fusion": forensic["fusion"],
"llm_verdict": verdict,
"timestamp": datetime.utcnow().isoformat(),
})
```
## Health check și monitoring
```python
# Verifică serviciul e disponibil înainte de a procesa
def is_forensic_healthy():
try:
r = requests.get(f"{FORENSIC_API}/health", timeout=5)
return r.status_code == 200 and r.json().get("status") == "ok"
except Exception:
return False
# Fallback graceful dacă serviciul e down
if not is_forensic_healthy():
logger.warning("Forensic features API down, proceeding without augmentation")
# Doar apel LLM cu prompt-ul tău original, fără evidence
else:
# Apel complet cu augmentation
```
## Limitări la integrare
1. **Cost de timp**: +5-90s pe request (depinde de durata video). Pentru
UX, folosește async mode cu indicator de progres.
2. **Cost de tokens LLM**: evidence_text adaugă ~500-1000 tokens.
Imaginile noastre ~12-15 imagini × cost per image LLM.
3. **Determinism**: forensic features sunt deterministe, dar LLM-ul nu.
Pe același video, evidence e mereu același, dar verdict LLM poate varia.
4. **Limită upload**: 2 GB max. Video peste asta — split sau downscale înainte.
## Test live
```bash
# Verifică serviciu
curl http://localhost:8080/health
# Test cu un video real
curl -X POST http://localhost:8080/api/forensic-evidence \
-F "video=@your_test_video.mp4" \
-o response.json
# Extrage doar partea text pentru LLM
python -c "import json; print(json.load(open('response.json'))['evidence_text'])"
# Numără imaginile generate
python -c "import json; print(f'{len(json.load(open(\"response.json\"))[\"images\"])} PNG-uri pentru LLM')"
```