Add CP source adapter registry with multi-worker queues and LLM extract.

Replace legacy root backend/frontend with Telegram, Crawl4AI, and VIINA adapters routed by Redis job families.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-14 11:34:28 +03:00
co-authored by Cursor
parent 1492576fd9
commit 8fbabd3c11
54 changed files with 1625 additions and 2521 deletions
+5 -2
View File
@@ -2,13 +2,16 @@ FROM python:3.12-slim
WORKDIR /app
COPY requirements.txt .
COPY centers/analytics/api/requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY app/ ./app/
COPY contracts/ ./contracts/
COPY centers/analytics/api/app/ ./app/
RUN mkdir -p /data
ENV PYTHONPATH=/app
EXPOSE 8000
CMD ["uvicorn", "app.main:app", "--host", "0.0.0.0", "--port", "8000"]
+9
View File
@@ -1,8 +1,17 @@
from contextlib import asynccontextmanager
import sys
from pathlib import Path
from fastapi import FastAPI
from fastapi.middleware.cors import CORSMiddleware
# Monorepo / Docker: contracts live next to app or at repo root
_HERE = Path(__file__).resolve()
for _candidate in (_HERE.parent, *_HERE.parents):
if (_candidate / "contracts").is_dir() and str(_candidate) not in sys.path:
sys.path.insert(0, str(_candidate))
break
from .database import Base, engine, get_db
from .routers import admin, internal, map, objects, v1
from .seed import seed_objects, seed_test_consumer
+14 -2
View File
@@ -37,11 +37,23 @@ from ..services.jobs import enqueue_job
router = APIRouter(prefix="/admin", tags=["admin"])
def _validated_source_config(source_type: str, source_config: dict) -> dict:
try:
from contracts.queues import family_for_source
from contracts.sources import parse_source_config
family_for_source(source_type)
return parse_source_config(source_type, source_config or {}).model_dump()
except Exception as exc:
raise HTTPException(status_code=400, detail=str(exc)) from exc
@router.post("/jobs", response_model=ParseJobRead, status_code=201)
def create_parse_job(payload: ParseJobCreate, db: Session = Depends(get_db)):
source_config = _validated_source_config(payload.source_type, payload.source_config)
job = ParseJob(
source_type=payload.source_type,
source_config=payload.source_config,
source_config=source_config,
schedule=payload.schedule,
interval_seconds=payload.interval_seconds,
is_active=payload.is_active,
@@ -96,7 +108,7 @@ def update_parse_job(
raise HTTPException(status_code=404, detail="Job not found")
if payload.source_config is not None:
job.source_config = payload.source_config
job.source_config = _validated_source_config(job.source_type, payload.source_config)
if payload.interval_seconds is not None:
job.interval_seconds = payload.interval_seconds
if payload.is_active is not None:
+23 -3
View File
@@ -1,10 +1,22 @@
import json
import os
import sys
from pathlib import Path
import redis
# Allow importing shared contracts from monorepo root in local runs
_HERE = Path(__file__).resolve()
for _candidate in (_HERE.parent, *_HERE.parents):
if (_candidate / "contracts").is_dir():
if str(_candidate) not in sys.path:
sys.path.insert(0, str(_candidate))
break
from contracts.queues import queue_key_for_source # noqa: E402
from contracts.sources import parse_source_config # noqa: E402
REDIS_URL = os.getenv("REDIS_URL", "redis://redis:6379/0")
JOB_QUEUE_KEY = "cp:jobs"
def get_redis() -> redis.Redis:
@@ -12,9 +24,17 @@ def get_redis() -> redis.Redis:
def enqueue_job(job_id: int, source_type: str, source_config: dict) -> None:
# Validate known configs early; unknown types still raise from queue_key_for_source
try:
parse_source_config(source_type, source_config or {})
except Exception:
# Keep enqueue resilient for legacy rows; worker validates again
pass
payload = {
"job_id": job_id,
"source_type": source_type,
"source_config": source_config,
"source_config": source_config or {},
}
get_redis().rpush(JOB_QUEUE_KEY, json.dumps(payload))
key = queue_key_for_source(source_type)
get_redis().rpush(key, json.dumps(payload))
@@ -1,5 +1,5 @@
<script setup lang="ts">
import { onMounted, onUnmounted, ref } from "vue";
import { computed, onMounted, onUnmounted, ref } from "vue";
import { createJob, deleteJob, fetchJobs, retryJob, updateJob } from "../api/admin";
import type { ParseJob } from "../types/admin";
@@ -12,22 +12,39 @@ const INTERVAL_OPTIONS = [
{ value: 86400, label: "24 ч" },
];
const SOURCE_TYPES = [
{ value: "telegram", label: "Telegram" },
{ value: "crawl4ai", label: "Crawl4AI (web)" },
{ value: "viina", label: "VIINA (news NLP)" },
] as const;
const jobs = ref<ParseJob[]>([]);
const loading = ref(true);
const error = ref("");
const submitting = ref(false);
const editingJob = ref<ParseJob | null>(null);
const editForm = ref({
channel: "",
limit: 50,
urlsText: "",
textsText: "",
domain_profile: "generic_news",
input_mode: "urls" as "urls" | "texts" | "mixed",
extract_mode: "heuristic" as "heuristic" | "llm",
interval_seconds: 3600,
is_active: true,
});
const form = ref({
source_type: "telegram",
source_type: "telegram" as "telegram" | "crawl4ai" | "viina",
channel: "",
limit: 50,
urlsText: "",
textsText: "",
domain_profile: "generic_news",
input_mode: "urls" as "urls" | "texts" | "mixed",
extract_mode: "heuristic" as "heuristic" | "llm",
interval_seconds: 3600,
});
@@ -42,6 +59,38 @@ function limitFromConfig(config: Record<string, unknown>): number {
return typeof limit === "number" ? limit : 50;
}
function urlsFromConfig(config: Record<string, unknown>): string[] {
return Array.isArray(config.urls)
? config.urls.filter((u): u is string => typeof u === "string")
: [];
}
function textsFromConfig(config: Record<string, unknown>): string[] {
return Array.isArray(config.texts)
? config.texts.filter((t): t is string => typeof t === "string")
: [];
}
function parseLines(text: string): string[] {
return text
.split(/\r?\n/)
.map((line) => line.trim())
.filter(Boolean);
}
function sourceSummary(job: ParseJob): string {
const cfg = job.source_config;
const mode = cfg.extract_mode === "llm" ? " [LLM]" : "";
if (job.source_type === "telegram") {
return (channelFromConfig(cfg) || "—") + mode;
}
const urls = urlsFromConfig(cfg);
if (urls.length) return (urls.length === 1 ? urls[0] : `${urls.length} URL`) + mode;
const texts = textsFromConfig(cfg);
if (texts.length) return `${texts.length} текст(ов)`;
return "—";
}
function formatInterval(seconds: number): string {
const opt = INTERVAL_OPTIONS.find((item) => item.value === seconds);
if (opt) return opt.label;
@@ -50,6 +99,53 @@ function formatInterval(seconds: number): string {
return `${Math.round(seconds / 86400)} д`;
}
function buildSourceConfig(
sourceType: string,
data: {
channel: string;
limit: number;
urlsText: string;
textsText: string;
domain_profile: string;
input_mode: "urls" | "texts" | "mixed";
extract_mode: "heuristic" | "llm";
},
): Record<string, unknown> {
if (sourceType === "telegram") {
return {
channel: data.channel.trim(),
limit: data.limit,
extract_mode: data.extract_mode,
};
}
if (sourceType === "crawl4ai") {
return {
urls: parseLines(data.urlsText),
domain_profile: data.domain_profile.trim() || "generic_news",
extract_mode: data.extract_mode,
};
}
return {
urls: parseLines(data.urlsText),
texts: parseLines(data.textsText),
input_mode: data.input_mode,
};
}
const formHint = computed(() => {
if (form.value.source_type === "telegram") {
return form.value.extract_mode === "llm"
? "LLM (DeepSeek): неструктурированные посты канала разбираются в locality/date/coords/description."
: "Активные Telegram-парсеры подхватываются real-time listener; batch забирает последние N постов (эвристики).";
}
if (form.value.source_type === "crawl4ai") {
return form.value.extract_mode === "llm"
? "Crawl4AI + LLM (DeepSeek): страница → structured event. Нужен DEEPSEEK_API_KEY и cp-workers-web."
: "Crawl4AI обходит URL и нормализует страницы эвристиками. Обрабатывает cp-workers-web.";
}
return "VIINA-адаптер извлекает инциденты из новостных URL/текстов. Обрабатывает cp-workers-nlp.";
});
async function loadJobs() {
try {
jobs.value = await fetchJobs();
@@ -62,24 +158,43 @@ async function loadJobs() {
}
async function handleSubmit() {
if (!form.value.channel.trim()) {
if (form.value.source_type === "telegram" && !form.value.channel.trim()) {
error.value = "Укажите канал Telegram";
return;
}
if (form.value.source_type === "crawl4ai" && !parseLines(form.value.urlsText).length) {
error.value = "Укажите хотя бы один URL";
return;
}
if (form.value.source_type === "viina") {
const urls = parseLines(form.value.urlsText);
const texts = parseLines(form.value.textsText);
if (form.value.input_mode === "urls" && !urls.length) {
error.value = "Укажите URL для VIINA";
return;
}
if (form.value.input_mode === "texts" && !texts.length) {
error.value = "Укажите тексты для VIINA";
return;
}
if (form.value.input_mode === "mixed" && !urls.length && !texts.length) {
error.value = "Укажите URL или тексты";
return;
}
}
submitting.value = true;
error.value = "";
try {
await createJob({
source_type: form.value.source_type,
source_config: {
channel: form.value.channel.trim(),
limit: form.value.limit,
},
source_config: buildSourceConfig(form.value.source_type, form.value),
interval_seconds: form.value.interval_seconds,
is_active: true,
});
form.value.channel = "";
form.value.urlsText = "";
form.value.textsText = "";
await loadJobs();
} catch (err) {
error.value = err instanceof Error ? err.message : "Не удалось создать задание";
@@ -93,6 +208,17 @@ function openEdit(job: ParseJob) {
editForm.value = {
channel: channelFromConfig(job.source_config),
limit: limitFromConfig(job.source_config),
urlsText: urlsFromConfig(job.source_config).join("\n"),
textsText: textsFromConfig(job.source_config).join("\n"),
domain_profile:
typeof job.source_config.domain_profile === "string"
? job.source_config.domain_profile
: "generic_news",
input_mode:
job.source_config.input_mode === "texts" || job.source_config.input_mode === "mixed"
? job.source_config.input_mode
: "urls",
extract_mode: job.source_config.extract_mode === "llm" ? "llm" : "heuristic",
interval_seconds: job.interval_seconds,
is_active: job.is_active,
};
@@ -104,7 +230,9 @@ function closeEdit() {
async function handleSaveEdit() {
if (!editingJob.value) return;
if (!editForm.value.channel.trim()) {
const sourceType = editingJob.value.source_type;
if (sourceType === "telegram" && !editForm.value.channel.trim()) {
error.value = "Укажите канал Telegram";
return;
}
@@ -113,10 +241,7 @@ async function handleSaveEdit() {
error.value = "";
try {
await updateJob(editingJob.value.id, {
source_config: {
channel: editForm.value.channel.trim(),
limit: editForm.value.limit,
},
source_config: buildSourceConfig(sourceType, editForm.value),
interval_seconds: editForm.value.interval_seconds,
is_active: editForm.value.is_active,
});
@@ -130,7 +255,7 @@ async function handleSaveEdit() {
}
async function handleDelete(job: ParseJob) {
if (!window.confirm(`Удалить парсер #${job.id} (${channelFromConfig(job.source_config)})?`)) {
if (!window.confirm(`Удалить парсер #${job.id} (${sourceSummary(job)})?`)) {
return;
}
error.value = "";
@@ -181,26 +306,75 @@ onUnmounted(() => {
<section class="card">
<h3>Новый парсер</h3>
<p class="hint">
Активные парсеры подхватываются real-time listener (новые посты сразу в БД).
Дополнительно batch-прогон по интервалу забирает последние N постов.
{{ formHint }}
Дубликаты по <code>source_url</code> не записываются.
</p>
<form class="admin-form" @submit.prevent="handleSubmit">
<div class="form-row">
<label>
Тип источника
<select v-model="form.source_type" disabled>
<option value="telegram">Telegram</option>
<select v-model="form.source_type">
<option v-for="opt in SOURCE_TYPES" :key="opt.value" :value="opt.value">
{{ opt.label }}
</option>
</select>
</label>
<label>
Канал
<input v-model="form.channel" type="text" placeholder="creamy_caprice" required />
</label>
<label>
Лимит
<input v-model.number="form.limit" type="number" min="1" max="1000" />
</label>
<template v-if="form.source_type === 'telegram'">
<label>
Канал
<input v-model="form.channel" type="text" placeholder="creamy_caprice" required />
</label>
<label>
Лимит
<input v-model.number="form.limit" type="number" min="1" max="1000" />
</label>
<label>
Извлечение
<select v-model="form.extract_mode">
<option value="heuristic">Эвристики (структурированные посты)</option>
<option value="llm">LLM DeepSeek (неструктурированные)</option>
</select>
</label>
</template>
<template v-else-if="form.source_type === 'crawl4ai'">
<label class="span-2">
URL (по одному в строке)
<textarea v-model="form.urlsText" rows="3" placeholder="https://example.com/news/…" required />
</label>
<label>
Domain profile
<input v-model="form.domain_profile" type="text" placeholder="generic_news" />
</label>
<label>
Извлечение
<select v-model="form.extract_mode">
<option value="heuristic">Эвристики (regex)</option>
<option value="llm">LLM DeepSeek (Crawl4AI)</option>
</select>
</label>
</template>
<template v-else>
<label>
Режим ввода
<select v-model="form.input_mode">
<option value="urls">URLs</option>
<option value="texts">Texts</option>
<option value="mixed">Mixed</option>
</select>
</label>
<label v-if="form.input_mode !== 'texts'" class="span-2">
URL (по одному в строке)
<textarea v-model="form.urlsText" rows="3" placeholder="https://example.com/article…" />
</label>
<label v-if="form.input_mode !== 'urls'" class="span-2">
Тексты (по одному блоку в строке)
<textarea v-model="form.textsText" rows="3" placeholder="Текст статьи…" />
</label>
</template>
<label>
Интервал
<select v-model.number="form.interval_seconds">
@@ -226,8 +400,8 @@ onUnmounted(() => {
<thead>
<tr>
<th>ID</th>
<th>Канал</th>
<th>Лимит</th>
<th>Тип</th>
<th>Источник</th>
<th>Интервал</th>
<th>Активен</th>
<th>Статус</th>
@@ -242,8 +416,8 @@ onUnmounted(() => {
</tr>
<tr v-for="job in jobs" :key="job.id">
<td>{{ job.id }}</td>
<td>{{ channelFromConfig(job.source_config) }}</td>
<td>{{ limitFromConfig(job.source_config) }}</td>
<td>{{ job.source_type }}</td>
<td class="source-cell">{{ sourceSummary(job) }}</td>
<td>{{ formatInterval(job.interval_seconds) }}</td>
<td>{{ job.is_active ? "да" : "нет" }}</td>
<td>
@@ -278,17 +452,61 @@ onUnmounted(() => {
<div v-if="editingJob" class="modal-backdrop" @click.self="closeEdit">
<div class="modal card">
<h3>Редактировать парсер #{{ editingJob.id }}</h3>
<h3>Редактировать парсер #{{ editingJob.id }} ({{ editingJob.source_type }})</h3>
<form class="admin-form" @submit.prevent="handleSaveEdit">
<div class="form-row">
<label>
Канал
<input v-model="editForm.channel" type="text" required />
</label>
<label>
Лимит
<input v-model.number="editForm.limit" type="number" min="1" max="1000" />
</label>
<template v-if="editingJob.source_type === 'telegram'">
<label>
Канал
<input v-model="editForm.channel" type="text" required />
</label>
<label>
Лимит
<input v-model.number="editForm.limit" type="number" min="1" max="1000" />
</label>
<label>
Извлечение
<select v-model="editForm.extract_mode">
<option value="heuristic">Эвристики</option>
<option value="llm">LLM DeepSeek</option>
</select>
</label>
</template>
<template v-else-if="editingJob.source_type === 'crawl4ai'">
<label class="span-2">
URL (по одному в строке)
<textarea v-model="editForm.urlsText" rows="3" required />
</label>
<label>
Domain profile
<input v-model="editForm.domain_profile" type="text" />
</label>
<label>
Извлечение
<select v-model="editForm.extract_mode">
<option value="heuristic">Эвристики</option>
<option value="llm">LLM DeepSeek</option>
</select>
</label>
</template>
<template v-else>
<label>
Режим ввода
<select v-model="editForm.input_mode">
<option value="urls">URLs</option>
<option value="texts">Texts</option>
<option value="mixed">Mixed</option>
</select>
</label>
<label v-if="editForm.input_mode !== 'texts'" class="span-2">
URL
<textarea v-model="editForm.urlsText" rows="3" />
</label>
<label v-if="editForm.input_mode !== 'urls'" class="span-2">
Тексты
<textarea v-model="editForm.textsText" rows="3" />
</label>
</template>
<label>
Интервал
<select v-model.number="editForm.interval_seconds">
@@ -322,6 +540,26 @@ onUnmounted(() => {
line-height: 1.5;
}
.form-row .span-2 {
grid-column: span 2;
}
.form-row textarea {
width: 100%;
font: inherit;
padding: 0.4rem 0.5rem;
border: 1px solid #d1d5db;
border-radius: 4px;
resize: vertical;
}
.source-cell {
max-width: 280px;
overflow: hidden;
text-overflow: ellipsis;
white-space: nowrap;
}
.actions-cell {
white-space: nowrap;
}
@@ -346,8 +584,10 @@ onUnmounted(() => {
}
.modal {
width: min(520px, 92vw);
width: min(560px, 92vw);
margin: 0;
max-height: 90vh;
overflow: auto;
}
.checkbox-row {
+146
View File
@@ -0,0 +1,146 @@
# Архитектура парсинга (ЦП)
Центр парсинга (**ЦП**) — сервисы `cp-workers*` с **реестром адаптеров** источников. Собирают события и отправляют их в ЦА через internal API.
## Роль в платформе
```mermaid
flowchart LR
CA[ЦА ca-api]
Redis[(Redis cp:jobs:family)]
CP[ЦП adapters]
Src[Sources]
CA -->|"RPUSH JobPayload"| Redis
Redis -->|"BLPOP"| CP
CP --> Src
CP -->|"POST /internal/ingest"| CA
CP -->|"PATCH /internal/jobs/{id}"| CA
CP -->|"GET /internal/listener/subscriptions"| CA
```
| Направление | Механизм | Назначение |
|-------------|----------|------------|
| ЦА → ЦП | Redis `cp:jobs:{family}` | Batch-задания по семейству адаптеров |
| ЦП → источники | Адаптер (`telegram` / `crawl4ai` / `viina`) | Fetch + extract |
| ЦП → ЦА | `POST /internal/ingest` | Запись событий (`IngestEventItem`) |
| ЦП → ЦА | `PATCH /internal/jobs/{id}` | Статус batch-задания |
| ЦП ← ЦА | `GET /internal/listener/subscriptions` | Каналы real-time (только Telegram) |
Общие контракты: [`contracts/jobs.py`](../../contracts/jobs.py), [`contracts/ingest.py`](../../contracts/ingest.py), [`contracts/sources.py`](../../contracts/sources.py), [`contracts/queues.py`](../../contracts/queues.py).
## Адаптеры источников
Каждый `source_type` реализует `SourceAdapter`:
```python
async def run(job_id, source_config, *, ctx) -> tuple[list[dict], str | None]
```
Выход — список dict в форме `IngestEventItem`. ЦА **не** знает про Crawl4AI/VIINA.
| source_type | Семейство / очередь | Сервис | Зависимости |
|-------------|---------------------|--------|-------------|
| `telegram` | `telegram` → `cp:jobs:telegram` | `cp-workers` | Telethon |
| `crawl4ai` | `web` → `cp:jobs:web` | `cp-workers-web` | Crawl4AI + Playwright |
| `viina` | `nlp` → `cp:jobs:nlp` | `cp-workers-nlp` | httpx + BeautifulSoup |
Маршрутизация при enqueue в ЦА: [`queue_key_for_source`](../../contracts/queues.py).
Воркер слушает `WORKER_FAMILIES` / очереди своих `ENABLED_ADAPTERS`. Чужой job → requeue в нужную очередь.
Правила для всех адаптеров:
- стабильный `source_url` (дедуп в ЦА);
- `source_type` события = тип адаптера;
- `source_config` валидируется схемами из `contracts/sources.py`.
### LLM-режим (`extract_mode: llm`)
Для неструктурированных Telegram-постов и Crawl4AI:
- ключ `DEEPSEEK_API_KEY` в `.env` (воркеры `cp-workers` / `cp-workers-web`);
- Telegram: текст поста → DeepSeek JSON → `IngestEventItem`;
- Crawl4AI: страница → `LLMExtractionStrategy` (DeepSeek) с fallback на тот же DeepSeek по markdown;
- в UI «Парсеры»: поле **Извлечение** = LLM DeepSeek.
```mermaid
flowchart LR
Job[JobPayload]
Reg[AdapterRegistry]
TG[TelegramAdapter]
C4[Crawl4AIAdapter]
VI[ViinaAdapter]
Ingest[IngestEventItem]
Job --> Reg
Reg --> TG --> Ingest
Reg --> C4 --> Ingest
Reg --> VI --> Ingest
```
## Структура каталога
```
centers/parsing/
├── ARCHITECTURE.md
└── workers/
├── Dockerfile # context = repo root; ARG REQUIREMENTS_FILE
├── requirements.txt # telegram
├── requirements-web.txt # crawl4ai
├── requirements-nlp.txt # viina
├── worker.py
└── workers/
├── adapters/
│ ├── base.py # SourceAdapter, WorkerContext
│ ├── registry.py # ENABLED_ADAPTERS
│ ├── telegram.py
│ ├── crawl4ai_adapter.py
│ └── viina.py
├── converter.py
├── parsers/telegram_events.py
└── sources/ # Telethon session / listener / client
```
## Режимы работы
| Режим | Условие | Поведение |
|-------|---------|-----------|
| Listener + batch | `ENABLED_ADAPTERS` включает `telegram` и `TELEGRAM_LISTENER_ENABLED=true` | Shared Telethon + listener + `worker_loop` |
| Только batch | listener выключен или нет telegram | Только `BLPOP` по очередям семейства |
## Поток batch-заданий
1. ЦА `enqueue_job` → Redis `cp:jobs:{family}` с `{ job_id, source_type, source_config }`
2. Воркер семейства: `BLPOP` → `handle_job` → `registry.get(source_type).run(...)`
3. `POST /internal/ingest` + статус job
Legacy-ключ `cp:jobs` по-прежнему дренируется telegram-воркером (совместимость).
## Telegram real-time
`TelegramListener` без изменений: подписки из ЦА, ingest с `listener: true` (статус `ParseJob` не трогается).
## Переменные окружения
| Переменная | Назначение |
|------------|------------|
| `ENABLED_ADAPTERS` | Список адаптеров через запятую (`telegram`, `crawl4ai`, `viina`) |
| `WORKER_FAMILIES` | Какие семейства очередей слушать (`telegram`, `web`, `nlp`) |
| `REDIS_URL` / `CA_API_URL` / `INTERNAL_TOKEN` | Как раньше |
| `TELEGRAM_*` | Только для `cp-workers` |
## Как добавить новый источник
1. Схема `source_config` в [`contracts/sources.py`](../../contracts/sources.py) + запись в `SOURCE_FAMILY` ([`queues.py`](../../contracts/queues.py))
2. Класс адаптера в `workers/adapters/` + factory в `registry.py`
3. При тяжёлых deps — `requirements-*.txt` и сервис в `docker-compose.yml`
4. Поля формы в UI «Парсеры»
## Связанные части ЦА
| Файл ЦА | Роль |
|---------|------|
| `services/jobs.py` | `enqueue_job` → `cp:jobs:{family}` |
| `routers/admin.py` | CRUD + валидация `source_config` |
| `services/ingest.py` | Сохранение Event + карта |
| UI `/parsers` | Выбор `telegram` / `crawl4ai` / `viina` |
+12 -3
View File
@@ -2,11 +2,20 @@ FROM python:3.12-slim
WORKDIR /app
COPY requirements.txt .
ARG REQUIREMENTS_FILE=requirements.txt
COPY centers/parsing/workers/${REQUIREMENTS_FILE} ./requirements.txt
RUN pip install --no-cache-dir -r requirements.txt
COPY workers/ ./workers/
COPY worker.py .
# Optional Playwright browsers for Crawl4AI web workers
ARG INSTALL_PLAYWRIGHT=0
RUN if [ "$INSTALL_PLAYWRIGHT" = "1" ]; then \
python -m playwright install --with-deps chromium; \
fi
COPY contracts/ ./contracts/
COPY centers/parsing/workers/workers/ ./workers/
COPY centers/parsing/workers/worker.py .
ENV PYTHONPATH=/app
@@ -0,0 +1,5 @@
httpx==0.28.1
redis==5.2.1
beautifulsoup4==4.12.3
lxml==5.3.0
pydantic==2.10.3
@@ -0,0 +1,7 @@
httpx==0.28.1
redis==5.2.1
beautifulsoup4==4.12.3
lxml==5.3.0
pydantic==2.10.3
crawl4ai>=0.4.0,<0.7
playwright>=1.40.0
+1
View File
@@ -4,3 +4,4 @@ telethon==1.44.0
python-socks[asyncio]==2.7.1
beautifulsoup4==4.12.3
lxml==5.3.0
pydantic==2.10.3
+83 -44
View File
@@ -1,29 +1,38 @@
"""CP worker: Redis jobs + real-time Telethon listener."""
"""CP worker: Redis jobs via adapter registry + optional Telethon listener."""
from __future__ import annotations
import asyncio
import json
import logging
import os
import sys
from pathlib import Path
# Monorepo local runs / Docker: ensure contracts/ is importable
_HERE = Path(__file__).resolve()
for _candidate in (_HERE.parent, *_HERE.parents):
if (_candidate / "contracts").is_dir():
if str(_candidate) not in sys.path:
sys.path.insert(0, str(_candidate))
break
import httpx
import redis
from workers.converter import event_record_to_ingest
from workers.parsers.telegram_events import parse_event_posts
from workers.sources.telegram_client import (
TelegramAuthError,
TelegramConfigError,
fetch_channel_posts,
normalize_channel,
from contracts.queues import (
LEGACY_JOB_QUEUE_KEY,
family_for_source,
queue_key_for_family,
queue_key_for_source,
)
from workers.sources.telegram_listener import TelegramListener
from workers.sources.telegram_session import close_shared_client, get_shared_client
from workers.adapters import build_registry
from workers.adapters.base import WorkerContext
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
logger = logging.getLogger("cp-worker")
REDIS_URL = os.getenv("REDIS_URL", "redis://redis:6379/0")
JOB_QUEUE_KEY = "cp:jobs"
CA_API_URL = os.getenv("CA_API_URL", "http://ca-api:8000")
INTERNAL_TOKEN = os.getenv("INTERNAL_TOKEN", "dev-internal-token")
POLL_TIMEOUT = int(os.getenv("WORKER_POLL_TIMEOUT", "5"))
@@ -33,32 +42,31 @@ LISTENER_ENABLED = os.getenv("TELEGRAM_LISTENER_ENABLED", "true").lower() not in
"no",
"off",
)
# Comma-separated families this process polls (default: derived from adapters)
WORKER_FAMILIES = os.getenv("WORKER_FAMILIES", "").strip()
REGISTRY = build_registry()
def get_redis() -> redis.Redis:
return redis.from_url(REDIS_URL, decode_responses=True)
async def process_telegram_job(
job_id: int,
source_config: dict,
*,
client=None,
) -> tuple[list[dict], str | None]:
channel = source_config.get("channel", "creamy_caprice")
limit = int(source_config.get("limit", 100))
try:
username = normalize_channel(channel)
posts = await fetch_channel_posts(username, limit=limit, client=client)
except (TelegramConfigError, TelegramAuthError, ValueError) as exc:
return [], str(exc)
except Exception as exc:
return [], f"Telegram: {exc}"
records = parse_event_posts(posts)
events = [event_record_to_ingest(r) for r in records]
return events, None
def _poll_keys() -> list[str]:
if WORKER_FAMILIES:
families = [f.strip() for f in WORKER_FAMILIES.split(",") if f.strip()]
else:
families = sorted(
{
family_for_source(source_type)
for source_type in REGISTRY.enabled_types()
}
)
keys = [queue_key_for_family(f) for f in families]
# Backward compatible: telegram workers also drain legacy cp:jobs
if "telegram" in families and LEGACY_JOB_QUEUE_KEY not in keys:
keys.append(LEGACY_JOB_QUEUE_KEY)
return keys
async def post_ingest(job_id: int, events: list[dict]) -> None:
@@ -95,18 +103,35 @@ async def patch_job_status(job_id: int, status: str, error: str | None = None) -
response.raise_for_status()
async def handle_job(payload: dict, *, tg_client=None) -> None:
async def handle_job(payload: dict, *, ctx: WorkerContext) -> None:
job_id = payload["job_id"]
source_type = payload["source_type"]
source_config = payload.get("source_config", {})
logger.info("Processing job %s (%s)", job_id, source_type)
await patch_job_status(job_id, "running")
if source_type == "telegram":
events, error = await process_telegram_job(job_id, source_config, client=tg_client)
else:
events, error = [], f"Unsupported source_type: {source_type}"
adapter = REGISTRY.get(source_type)
if adapter is None:
try:
target = queue_key_for_source(source_type)
except ValueError:
await patch_job_status(
job_id,
"failed",
error=f"Unsupported source_type: {source_type}",
)
return
get_redis().rpush(target, json.dumps(payload))
logger.warning(
"Job %s (%s) not enabled here; requeued to %s",
job_id,
source_type,
target,
)
return
await patch_job_status(job_id, "running")
events, error = await adapter.run(job_id, source_config, ctx=ctx)
if error:
logger.error("Job %s failed: %s", job_id, error)
@@ -123,18 +148,23 @@ async def handle_job(payload: dict, *, tg_client=None) -> None:
await patch_job_status(job_id, "failed", error=str(exc))
async def worker_loop(*, tg_client=None) -> None:
async def worker_loop(*, ctx: WorkerContext) -> None:
r = get_redis()
logger.info("CP worker started, polling %s", JOB_QUEUE_KEY)
keys = _poll_keys()
logger.info(
"CP worker started, polling %s (adapters: %s)",
keys,
", ".join(REGISTRY.enabled_types()) or "(none)",
)
while True:
try:
item = await asyncio.to_thread(r.blpop, JOB_QUEUE_KEY, POLL_TIMEOUT)
item = await asyncio.to_thread(r.blpop, keys, POLL_TIMEOUT)
if not item:
continue
_, raw = item
payload = json.loads(raw)
await handle_job(payload, tg_client=tg_client)
await handle_job(payload, ctx=ctx)
except redis.RedisError as exc:
logger.error("Redis error: %s", exc)
await asyncio.sleep(3)
@@ -144,9 +174,18 @@ async def worker_loop(*, tg_client=None) -> None:
async def run_with_listener() -> None:
if "telegram" not in REGISTRY:
logger.warning("Listener requested but telegram adapter is not enabled")
await worker_loop(ctx=WorkerContext())
return
from workers.sources.telegram_listener import TelegramListener
from workers.sources.telegram_session import close_shared_client, get_shared_client
client = await get_shared_client()
listener = TelegramListener(client)
worker_task = asyncio.create_task(worker_loop(tg_client=client))
ctx = WorkerContext(tg_client=client)
worker_task = asyncio.create_task(worker_loop(ctx=ctx))
listener_task = asyncio.create_task(listener.run())
logger.info("Telegram listener enabled (shared session with batch worker)")
@@ -160,12 +199,12 @@ async def run_with_listener() -> None:
async def run_batch_only() -> None:
logger.info("Telegram listener disabled")
await worker_loop(tg_client=None)
await worker_loop(ctx=WorkerContext(tg_client=None))
def main() -> None:
try:
if LISTENER_ENABLED:
if LISTENER_ENABLED and "telegram" in REGISTRY:
asyncio.run(run_with_listener())
else:
asyncio.run(run_batch_only())
@@ -0,0 +1,5 @@
"""CP source adapters (telegram, crawl4ai, viina, …)."""
from workers.adapters.registry import AdapterRegistry, build_registry
__all__ = ["AdapterRegistry", "build_registry"]
@@ -0,0 +1,28 @@
"""Source adapter protocol and shared worker context."""
from __future__ import annotations
from dataclasses import dataclass, field
from typing import Any, Protocol
@dataclass
class WorkerContext:
"""Runtime deps shared by adapters (optional Telegram client, etc.)."""
tg_client: Any | None = None
extras: dict[str, Any] = field(default_factory=dict)
class SourceAdapter(Protocol):
source_type: str
async def run(
self,
job_id: int,
source_config: dict,
*,
ctx: WorkerContext,
) -> tuple[list[dict], str | None]:
"""Return ingest-ready dicts (IngestEventItem-shaped) or an error string."""
...
@@ -0,0 +1,286 @@
"""Crawl4AI web adapter: crawl URLs → map fields → ingest events."""
from __future__ import annotations
import json
import logging
import re
from typing import Any
from urllib.parse import urlparse
from contracts.sources import Crawl4AISourceConfig
from workers.adapters.base import WorkerContext
from workers.llm_extract import (
DEFAULT_INSTRUCTION,
crawl4ai_llm_config,
extract_event_fields,
fields_to_ingest,
llm_enabled,
parse_coords,
parse_date,
)
logger = logging.getLogger("cp-worker.crawl4ai")
COORDS_RE = re.compile(r"(-?\d{1,3}\.\d+)\s*,\s*(-?\d{1,3}\.\d+)")
DATE_RE = re.compile(
r"\b(\d{1,2}[./]\d{1,2}[./]\d{2,4}|\d{4}-\d{2}-\d{2})\b",
)
class Crawl4AIAdapter:
source_type = "crawl4ai"
async def run(
self,
job_id: int,
source_config: dict,
*,
ctx: WorkerContext,
) -> tuple[list[dict], str | None]:
try:
cfg = Crawl4AISourceConfig.model_validate(source_config or {})
except Exception as exc:
return [], f"Invalid crawl4ai source_config: {exc}"
try:
from crawl4ai import AsyncWebCrawler # type: ignore
except ImportError:
return [], (
"crawl4ai is not installed in this worker image. "
"Use cp-workers-web (ENABLED_ADAPTERS=crawl4ai)."
)
if cfg.extract_mode == "llm" and not llm_enabled():
return [], "extract_mode=llm requires DEEPSEEK_API_KEY in worker env"
events: list[dict] = []
errors: list[str] = []
async with AsyncWebCrawler(verbose=False) as crawler:
for url in cfg.urls:
try:
if cfg.extract_mode == "llm":
item, err = await _crawl_with_llm(crawler, url, cfg)
else:
item, err = await _crawl_heuristic(crawler, url, cfg)
if err:
errors.append(err)
if item:
events.append(item)
except Exception as exc:
logger.exception("Crawl4AI failed for %s", url)
errors.append(f"{url}: {exc}")
if not events and errors:
return [], "; ".join(errors)
return events, None
async def _crawl_heuristic(crawler: Any, url: str, cfg: Crawl4AISourceConfig):
result = await crawler.arun(url=url)
markdown = _result_markdown(result)
if not markdown:
return None, f"{url}: empty crawl result"
return (
_markdown_to_ingest(
url=url,
markdown=markdown,
extract_schema=cfg.extract_schema,
domain_profile=cfg.domain_profile,
),
None,
)
async def _crawl_with_llm(crawler: Any, url: str, cfg: Crawl4AISourceConfig):
"""Prefer Crawl4AI LLMExtractionStrategy; fall back to DeepSeek on markdown."""
instruction = cfg.instruction or DEFAULT_INSTRUCTION
try:
from crawl4ai import CacheMode, CrawlerRunConfig # type: ignore
from crawl4ai import LLMExtractionStrategy # type: ignore
strategy = LLMExtractionStrategy(
llm_config=crawl4ai_llm_config(),
schema=_pydantic_like_schema(cfg.extract_schema),
extraction_type="schema",
instruction=instruction,
input_format="markdown",
apply_chunking=False,
extra_args={"temperature": 0.1, "max_tokens": 1200},
)
run_config = CrawlerRunConfig(
extraction_strategy=strategy,
cache_mode=CacheMode.BYPASS,
)
result = await crawler.arun(url=url, config=run_config)
markdown = _result_markdown(result)
fields = _parse_extracted_content(getattr(result, "extracted_content", None))
if fields:
return (
fields_to_ingest(
source_type="crawl4ai",
source_url=url,
raw_text=markdown or json.dumps(fields, ensure_ascii=False),
fields=fields,
domain_profile=cfg.domain_profile,
extra_metadata={"extractor": "crawl4ai_llm"},
),
None,
)
if markdown:
# Fallback: same DeepSeek path as Telegram
fields = await extract_event_fields(
markdown,
extract_schema=cfg.extract_schema,
instruction=instruction,
)
if fields.get("is_event", True):
return (
fields_to_ingest(
source_type="crawl4ai",
source_url=url,
raw_text=markdown,
fields=fields,
domain_profile=cfg.domain_profile,
extra_metadata={"extractor": "deepseek_fallback"},
),
None,
)
return None, f"{url}: LLM marked as non-event"
return None, f"{url}: empty LLM extraction"
except Exception as exc:
logger.warning("Crawl4AI LLMStrategy failed for %s: %s; trying DeepSeek on markdown", url, exc)
result = await crawler.arun(url=url)
markdown = _result_markdown(result)
if not markdown:
return None, f"{url}: empty crawl result ({exc})"
fields = await extract_event_fields(
markdown,
extract_schema=cfg.extract_schema,
instruction=instruction,
)
if not fields.get("is_event", True):
return None, None
return (
fields_to_ingest(
source_type="crawl4ai",
source_url=url,
raw_text=markdown,
fields=fields,
domain_profile=cfg.domain_profile,
extra_metadata={"extractor": "deepseek_fallback", "llm_strategy_error": str(exc)},
),
None,
)
def _pydantic_like_schema(extract_schema: dict[str, str]) -> dict:
properties = {
key: {"type": "string", "description": desc}
for key, desc in extract_schema.items()
}
return {
"title": "MapEvent",
"type": "object",
"properties": properties,
"required": list(extract_schema.keys()),
}
def _parse_extracted_content(raw: Any) -> dict[str, Any]:
if not raw:
return {}
try:
data = json.loads(raw) if isinstance(raw, str) else raw
except json.JSONDecodeError:
return {}
if isinstance(data, list) and data:
data = data[0]
if not isinstance(data, dict):
return {}
# Unwrap common nesting
if isinstance(data.get("fields"), dict):
data = data["fields"]
return {k: str(v).strip() if v is not None else "" for k, v in data.items()}
def _result_markdown(result: Any) -> str:
markdown = getattr(result, "markdown", None) or ""
if hasattr(markdown, "raw_markdown"):
return str(markdown.raw_markdown or "")
if isinstance(markdown, str):
return markdown
fit = getattr(result, "fit_markdown", None)
return str(fit or "")
def _markdown_to_ingest(
*,
url: str,
markdown: str,
extract_schema: dict[str, str],
domain_profile: str,
) -> dict:
fields = _extract_fields(markdown, extract_schema)
lat, lng = parse_coords(fields.get("coords", ""))
description = fields.get("description") or markdown[:4000]
locality = fields.get("locality") or ""
title = fields.get("title") or locality or _title_from_url(url) or description.splitlines()[0][:120]
event_date = parse_date(fields.get("event_date", ""))
return {
"source_type": "crawl4ai",
"source_url": url,
"raw_text": markdown[:20000],
"title": title[:255],
"description": description[:8000],
"locality": locality,
"latitude": lat,
"longitude": lng,
"event_date": event_date.isoformat() if event_date else None,
"region": locality or None,
"topic": fields.get("topic") or domain_profile,
"tags": ["crawl4ai", domain_profile],
"metadata": {
"domain_profile": domain_profile,
"extract_schema": extract_schema,
"extracted": fields,
"extract_mode": "heuristic",
},
}
def _extract_fields(markdown: str, schema: dict[str, str]) -> dict[str, str]:
fields: dict[str, str] = {}
for key in schema:
if key == "coords":
match = COORDS_RE.search(markdown)
fields[key] = match.group(0) if match else ""
elif key == "event_date":
match = DATE_RE.search(markdown)
fields[key] = match.group(1) if match else ""
elif key == "description":
fields[key] = markdown.strip()[:4000]
elif key == "locality":
fields[key] = _guess_locality(markdown)
elif key == "title":
fields[key] = _guess_locality(markdown)
else:
fields[key] = ""
return fields
def _guess_locality(markdown: str) -> str:
for line in markdown.splitlines():
stripped = line.strip().lstrip("#").strip()
if 2 <= len(stripped) <= 80 and not COORDS_RE.search(stripped):
return stripped
return ""
def _title_from_url(url: str) -> str:
path = urlparse(url).path.rstrip("/")
if not path:
return urlparse(url).netloc
return path.rsplit("/", 1)[-1].replace("-", " ").replace("_", " ")
@@ -0,0 +1,89 @@
"""Registry of source adapters enabled for this worker process."""
from __future__ import annotations
import logging
import os
from typing import Callable
from workers.adapters.base import SourceAdapter
logger = logging.getLogger("cp-worker.adapters")
AdapterFactory = Callable[[], SourceAdapter]
class AdapterRegistry:
def __init__(self) -> None:
self._adapters: dict[str, SourceAdapter] = {}
def register(self, adapter: SourceAdapter) -> None:
self._adapters[adapter.source_type] = adapter
logger.info("Registered adapter: %s", adapter.source_type)
def get(self, source_type: str) -> SourceAdapter | None:
return self._adapters.get(source_type)
def enabled_types(self) -> list[str]:
return sorted(self._adapters.keys())
def __contains__(self, source_type: str) -> bool:
return source_type in self._adapters
def _parse_enabled_adapters() -> set[str] | None:
raw = os.getenv("ENABLED_ADAPTERS", "").strip()
if not raw:
return None
return {part.strip() for part in raw.split(",") if part.strip()}
def _factories() -> dict[str, AdapterFactory]:
# Lazy imports so telegram-only / web-only / nlp-only images do not need all deps
def telegram() -> SourceAdapter:
from workers.adapters.telegram import TelegramAdapter
return TelegramAdapter()
def crawl4ai() -> SourceAdapter:
from workers.adapters.crawl4ai_adapter import Crawl4AIAdapter
return Crawl4AIAdapter()
def viina() -> SourceAdapter:
from workers.adapters.viina import ViinaAdapter
return ViinaAdapter()
return {
"telegram": telegram,
"crawl4ai": crawl4ai,
"viina": viina,
}
def build_registry() -> AdapterRegistry:
"""Register adapters filtered by ENABLED_ADAPTERS (comma-separated).
Empty ENABLED_ADAPTERS → attempt to load every known adapter; skip import failures.
"""
enabled = _parse_enabled_adapters()
factories = _factories()
names = sorted(enabled) if enabled is not None else sorted(factories)
registry = AdapterRegistry()
for name in names:
factory = factories.get(name)
if factory is None:
logger.warning("Unknown adapter in ENABLED_ADAPTERS: %s", name)
continue
try:
registry.register(factory())
except Exception:
logger.exception("Failed to load adapter %s", name)
if not registry.enabled_types():
logger.warning("No adapters enabled (ENABLED_ADAPTERS=%r)", os.getenv("ENABLED_ADAPTERS"))
else:
logger.info("Enabled adapters: %s", ", ".join(registry.enabled_types()))
return registry
@@ -0,0 +1,103 @@
"""Telegram batch adapter (Telethon history → ingest events)."""
from __future__ import annotations
import logging
from contracts.sources import TelegramSourceConfig
from workers.adapters.base import WorkerContext
from workers.converter import event_record_to_ingest
from workers.llm_extract import (
DEFAULT_EXTRACT_SCHEMA,
DEFAULT_INSTRUCTION,
extract_event_fields,
fields_to_ingest,
llm_enabled,
)
from workers.parsers.telegram_events import parse_event_posts
from workers.sources.telegram_client import (
TelegramAuthError,
TelegramConfigError,
fetch_channel_posts,
normalize_channel,
)
logger = logging.getLogger("cp-worker.telegram")
class TelegramAdapter:
source_type = "telegram"
async def run(
self,
job_id: int,
source_config: dict,
*,
ctx: WorkerContext,
) -> tuple[list[dict], str | None]:
try:
cfg = TelegramSourceConfig.model_validate(source_config or {})
except Exception as exc:
return [], f"Invalid telegram source_config: {exc}"
try:
username = normalize_channel(cfg.channel)
posts = await fetch_channel_posts(
username,
limit=cfg.limit,
client=ctx.tg_client,
)
except (TelegramConfigError, TelegramAuthError, ValueError) as exc:
return [], str(exc)
except Exception as exc:
return [], f"Telegram: {exc}"
if cfg.extract_mode == "llm":
if not llm_enabled():
return [], "extract_mode=llm requires DEEPSEEK_API_KEY in worker env"
return await _extract_posts_with_llm(posts, cfg)
records = parse_event_posts(posts)
events = [event_record_to_ingest(r) for r in records]
return events, None
async def _extract_posts_with_llm(posts, cfg: TelegramSourceConfig) -> tuple[list[dict], str | None]:
schema = cfg.extract_schema or DEFAULT_EXTRACT_SCHEMA
instruction = cfg.instruction or DEFAULT_INSTRUCTION
events: list[dict] = []
errors: list[str] = []
for post in posts:
text = (post.text or "").strip()
if not text:
continue
try:
fields = await extract_event_fields(
text,
extract_schema=schema,
instruction=instruction,
)
if not fields.get("is_event", True):
continue
events.append(
fields_to_ingest(
source_type="telegram",
source_url=post.url,
raw_text=text,
fields=fields,
domain_profile="telegram_llm",
extra_metadata={
"channel": post.channel,
"message_id": post.id,
"post_date": post.date.isoformat() if post.date else None,
},
)
)
except Exception as exc:
logger.exception("LLM extract failed for %s", post.url)
errors.append(f"{post.url}: {exc}")
if not events and errors:
return [], "; ".join(errors[:5])
return events, None
@@ -0,0 +1,155 @@
"""VIINA-style news incident adapter: fetch/text → extract → ingest."""
from __future__ import annotations
import logging
import re
from datetime import datetime, timezone
from email.utils import parsedate_to_datetime
from typing import Any
import httpx
from bs4 import BeautifulSoup
from contracts.sources import ViinaSourceConfig
from workers.adapters.base import WorkerContext
logger = logging.getLogger("cp-worker.viina")
COORDS_RE = re.compile(r"(-?\d{1,3}\.\d+)\s*,\s*(-?\d{1,3}\.\d+)")
DATE_RE = re.compile(
r"\b(\d{1,2}[./]\d{1,2}[./]\d{2,4}|\d{4}-\d{2}-\d{2})\b",
)
# Lightweight incident cues inspired by VIINA-style violent-event coding
INCIDENT_CUES = re.compile(
r"\b(attack|shelling|strike|explosion|casualty|killed|wounded|"
r"обстрел|удар|взрыв|погибли|ранены|атака)\b",
re.IGNORECASE,
)
class ViinaAdapter:
source_type = "viina"
async def run(
self,
job_id: int,
source_config: dict,
*,
ctx: WorkerContext,
) -> tuple[list[dict], str | None]:
try:
cfg = ViinaSourceConfig.model_validate(source_config or {})
except Exception as exc:
return [], f"Invalid viina source_config: {exc}"
articles: list[tuple[str, str]] = []
errors: list[str] = []
if cfg.input_mode in ("urls", "mixed"):
for url in cfg.urls:
try:
text = await _fetch_article_text(url)
if text:
articles.append((url, text))
else:
errors.append(f"{url}: empty article")
except Exception as exc:
logger.exception("VIINA fetch failed for %s", url)
errors.append(f"{url}: {exc}")
if cfg.input_mode in ("texts", "mixed"):
for idx, text in enumerate(cfg.texts):
articles.append((f"viina:text:{job_id}:{idx}", text))
events = [_article_to_ingest(source_url, text) for source_url, text in articles]
# Keep articles without strong cues — still useful raw intelligence
if not events and errors:
return [], "; ".join(errors)
return events, None
async def _fetch_article_text(url: str) -> str:
async with httpx.AsyncClient(timeout=60.0, follow_redirects=True) as client:
response = await client.get(
url,
headers={"User-Agent": "MapMil-CP-Viina/1.0"},
)
response.raise_for_status()
content_type = response.headers.get("content-type", "")
if "html" in content_type or url.endswith(".html"):
return _html_to_text(response.text)
return response.text.strip()
def _html_to_text(html: str) -> str:
soup = BeautifulSoup(html, "lxml")
for tag in soup(["script", "style", "noscript", "nav", "footer", "header"]):
tag.decompose()
article = soup.find("article") or soup.find("main") or soup.body
if article is None:
return soup.get_text("\n", strip=True)
return article.get_text("\n", strip=True)
def _article_to_ingest(source_url: str, text: str) -> dict[str, Any]:
lat, lng = _parse_coords(text)
date_match = DATE_RE.search(text)
event_date = _parse_date(date_match.group(1) if date_match else "")
cue = INCIDENT_CUES.search(text)
title = next((ln.strip() for ln in text.splitlines() if ln.strip()), source_url)[:120]
locality = _guess_locality(text)
return {
"source_type": "viina",
"source_url": source_url,
"raw_text": text[:20000],
"title": title,
"description": text[:8000],
"locality": locality,
"latitude": lat,
"longitude": lng,
"event_date": event_date.isoformat() if event_date else None,
"region": locality or None,
"topic": "violent_incident" if cue else "news",
"tags": ["viina", "news"] + ([cue.group(0).lower()] if cue else []),
"metadata": {
"extractor": "viina_heuristic",
"incident_cue": cue.group(0) if cue else None,
},
}
def _parse_coords(raw: str) -> tuple[float | None, float | None]:
match = COORDS_RE.search(raw or "")
if not match:
return None, None
return float(match.group(1)), float(match.group(2))
def _parse_date(raw: str) -> datetime | None:
if not raw:
return None
raw = raw.strip()
for fmt in ("%d.%m.%Y", "%d.%m.%y", "%d/%m/%Y", "%d/%m/%y", "%Y-%m-%d"):
try:
return datetime.strptime(raw, fmt).replace(tzinfo=timezone.utc)
except ValueError:
continue
try:
dt = parsedate_to_datetime(raw)
if dt.tzinfo is None:
dt = dt.replace(tzinfo=timezone.utc)
return dt
except (TypeError, ValueError, IndexError):
return None
def _guess_locality(text: str) -> str:
for line in text.splitlines()[:15]:
stripped = line.strip()
if 2 <= len(stripped) <= 60 and not DATE_RE.search(stripped):
if INCIDENT_CUES.search(stripped):
continue
return stripped
return ""
@@ -0,0 +1,179 @@
"""DeepSeek / OpenAI-compatible LLM extraction for unstructured text."""
from __future__ import annotations
import json
import logging
import os
import re
from datetime import datetime, timezone
from typing import Any
import httpx
logger = logging.getLogger("cp-worker.llm")
COORDS_RE = re.compile(r"(-?\d{1,3}\.\d+)\s*,\s*(-?\d{1,3}\.\d+)")
DEFAULT_EXTRACT_SCHEMA: dict[str, str] = {
"title": "string — short event title",
"locality": "string — place / settlement name",
"event_date": "string — date as DD.MM.YYYY or YYYY-MM-DD if known",
"description": "string — concise event summary",
"coords": "string — latitude, longitude if present else empty",
"topic": "string — short topic tag",
}
DEFAULT_INSTRUCTION = (
"Extract structured military/news event fields from the text. "
"If the text is not an event, return is_event=false. "
"Respond with a single JSON object only."
)
def llm_enabled() -> bool:
return bool(os.getenv("DEEPSEEK_API_KEY", "").strip())
def llm_settings() -> dict[str, str]:
return {
"api_key": os.getenv("DEEPSEEK_API_KEY", "").strip(),
"base_url": os.getenv("DEEPSEEK_BASE_URL", "https://api.deepseek.com").rstrip("/"),
"model": os.getenv("DEEPSEEK_MODEL", "deepseek-chat"),
}
async def extract_event_fields(
text: str,
*,
extract_schema: dict[str, str] | None = None,
instruction: str | None = None,
) -> dict[str, Any]:
"""Ask DeepSeek to fill schema fields from free text. Returns dict (+ is_event)."""
settings = llm_settings()
if not settings["api_key"]:
raise RuntimeError(
"DEEPSEEK_API_KEY is not set. Add it to .env for LLM extract_mode."
)
schema = extract_schema or DEFAULT_EXTRACT_SCHEMA
instr = instruction or DEFAULT_INSTRUCTION
schema_lines = "\n".join(f"- {k}: {v}" for k, v in schema.items())
user_prompt = (
f"{instr}\n\n"
f"Fields to extract:\n{schema_lines}\n\n"
'Return JSON: {"is_event": true|false, "fields": {<field>: <string>}}\n\n'
f"Text:\n{text[:12000]}"
)
payload = {
"model": settings["model"],
"messages": [
{
"role": "system",
"content": (
"You extract structured event data for a geoint map. "
"Output valid JSON only, no markdown."
),
},
{"role": "user", "content": user_prompt},
],
"temperature": 0.1,
"response_format": {"type": "json_object"},
}
url = f"{settings['base_url']}/chat/completions"
async with httpx.AsyncClient(timeout=90.0) as client:
response = await client.post(
url,
headers={
"Authorization": f"Bearer {settings['api_key']}",
"Content-Type": "application/json",
},
json=payload,
)
response.raise_for_status()
data = response.json()
content = data["choices"][0]["message"]["content"]
parsed = json.loads(content)
fields = parsed.get("fields") if isinstance(parsed.get("fields"), dict) else parsed
if not isinstance(fields, dict):
fields = {}
# Normalize to strings for known keys
result = {key: str(fields.get(key) or "").strip() for key in schema}
result["is_event"] = bool(parsed.get("is_event", True))
return result
def fields_to_ingest(
*,
source_type: str,
source_url: str,
raw_text: str,
fields: dict[str, Any],
domain_profile: str = "llm",
extra_metadata: dict | None = None,
) -> dict:
lat, lng = parse_coords(str(fields.get("coords") or ""))
description = str(fields.get("description") or raw_text)[:8000]
locality = str(fields.get("locality") or "")
title = str(fields.get("title") or locality or description.splitlines()[0][:120])
topic = str(fields.get("topic") or domain_profile)
event_date = parse_date(str(fields.get("event_date") or ""))
meta = {
"extract_mode": "llm",
"extracted": {k: fields.get(k) for k in fields if k != "is_event"},
}
if extra_metadata:
meta.update(extra_metadata)
return {
"source_type": source_type,
"source_url": source_url,
"raw_text": raw_text[:20000],
"title": title[:255],
"description": description,
"locality": locality,
"latitude": lat,
"longitude": lng,
"event_date": event_date.isoformat() if event_date else None,
"region": locality or None,
"topic": topic,
"tags": [source_type, "llm", domain_profile],
"metadata": meta,
}
def parse_coords(raw: str) -> tuple[float | None, float | None]:
match = COORDS_RE.search(raw or "")
if not match:
return None, None
return float(match.group(1)), float(match.group(2))
def parse_date(raw: str) -> datetime | None:
if not raw:
return None
raw = raw.strip()
for fmt in ("%d.%m.%Y", "%d.%m.%y", "%d/%m/%Y", "%d/%m/%y", "%Y-%m-%d"):
try:
return datetime.strptime(raw, fmt).replace(tzinfo=timezone.utc)
except ValueError:
continue
return None
def crawl4ai_llm_config():
"""Build Crawl4AI LLMConfig for DeepSeek (OpenAI-compatible)."""
from crawl4ai import LLMConfig # type: ignore
settings = llm_settings()
if not settings["api_key"]:
raise RuntimeError("DEEPSEEK_API_KEY is not set")
return LLMConfig(
provider=f"openai/{settings['model']}",
api_token=settings["api_key"],
base_url=settings["base_url"],
)