Add CP source adapter registry with multi-worker queues and LLM extract.

Replace legacy root backend/frontend with Telegram, Crawl4AI, and VIINA adapters routed by Redis job families.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-14 11:34:28 +03:00
co-authored by Cursor
parent 1492576fd9
commit 8fbabd3c11
54 changed files with 1625 additions and 2521 deletions
@@ -1,5 +1,5 @@
<script setup lang="ts">
import { onMounted, onUnmounted, ref } from "vue";
import { computed, onMounted, onUnmounted, ref } from "vue";
import { createJob, deleteJob, fetchJobs, retryJob, updateJob } from "../api/admin";
import type { ParseJob } from "../types/admin";
@@ -12,22 +12,39 @@ const INTERVAL_OPTIONS = [
{ value: 86400, label: "24 ч" },
];
const SOURCE_TYPES = [
{ value: "telegram", label: "Telegram" },
{ value: "crawl4ai", label: "Crawl4AI (web)" },
{ value: "viina", label: "VIINA (news NLP)" },
] as const;
const jobs = ref<ParseJob[]>([]);
const loading = ref(true);
const error = ref("");
const submitting = ref(false);
const editingJob = ref<ParseJob | null>(null);
const editForm = ref({
channel: "",
limit: 50,
urlsText: "",
textsText: "",
domain_profile: "generic_news",
input_mode: "urls" as "urls" | "texts" | "mixed",
extract_mode: "heuristic" as "heuristic" | "llm",
interval_seconds: 3600,
is_active: true,
});
const form = ref({
source_type: "telegram",
source_type: "telegram" as "telegram" | "crawl4ai" | "viina",
channel: "",
limit: 50,
urlsText: "",
textsText: "",
domain_profile: "generic_news",
input_mode: "urls" as "urls" | "texts" | "mixed",
extract_mode: "heuristic" as "heuristic" | "llm",
interval_seconds: 3600,
});
@@ -42,6 +59,38 @@ function limitFromConfig(config: Record<string, unknown>): number {
return typeof limit === "number" ? limit : 50;
}
function urlsFromConfig(config: Record<string, unknown>): string[] {
return Array.isArray(config.urls)
? config.urls.filter((u): u is string => typeof u === "string")
: [];
}
function textsFromConfig(config: Record<string, unknown>): string[] {
return Array.isArray(config.texts)
? config.texts.filter((t): t is string => typeof t === "string")
: [];
}
function parseLines(text: string): string[] {
return text
.split(/\r?\n/)
.map((line) => line.trim())
.filter(Boolean);
}
function sourceSummary(job: ParseJob): string {
const cfg = job.source_config;
const mode = cfg.extract_mode === "llm" ? " [LLM]" : "";
if (job.source_type === "telegram") {
return (channelFromConfig(cfg) || "—") + mode;
}
const urls = urlsFromConfig(cfg);
if (urls.length) return (urls.length === 1 ? urls[0] : `${urls.length} URL`) + mode;
const texts = textsFromConfig(cfg);
if (texts.length) return `${texts.length} текст(ов)`;
return "—";
}
function formatInterval(seconds: number): string {
const opt = INTERVAL_OPTIONS.find((item) => item.value === seconds);
if (opt) return opt.label;
@@ -50,6 +99,53 @@ function formatInterval(seconds: number): string {
return `${Math.round(seconds / 86400)} д`;
}
function buildSourceConfig(
sourceType: string,
data: {
channel: string;
limit: number;
urlsText: string;
textsText: string;
domain_profile: string;
input_mode: "urls" | "texts" | "mixed";
extract_mode: "heuristic" | "llm";
},
): Record<string, unknown> {
if (sourceType === "telegram") {
return {
channel: data.channel.trim(),
limit: data.limit,
extract_mode: data.extract_mode,
};
}
if (sourceType === "crawl4ai") {
return {
urls: parseLines(data.urlsText),
domain_profile: data.domain_profile.trim() || "generic_news",
extract_mode: data.extract_mode,
};
}
return {
urls: parseLines(data.urlsText),
texts: parseLines(data.textsText),
input_mode: data.input_mode,
};
}
const formHint = computed(() => {
if (form.value.source_type === "telegram") {
return form.value.extract_mode === "llm"
? "LLM (DeepSeek): неструктурированные посты канала разбираются в locality/date/coords/description."
: "Активные Telegram-парсеры подхватываются real-time listener; batch забирает последние N постов (эвристики).";
}
if (form.value.source_type === "crawl4ai") {
return form.value.extract_mode === "llm"
? "Crawl4AI + LLM (DeepSeek): страница → structured event. Нужен DEEPSEEK_API_KEY и cp-workers-web."
: "Crawl4AI обходит URL и нормализует страницы эвристиками. Обрабатывает cp-workers-web.";
}
return "VIINA-адаптер извлекает инциденты из новостных URL/текстов. Обрабатывает cp-workers-nlp.";
});
async function loadJobs() {
try {
jobs.value = await fetchJobs();
@@ -62,24 +158,43 @@ async function loadJobs() {
}
async function handleSubmit() {
if (!form.value.channel.trim()) {
if (form.value.source_type === "telegram" && !form.value.channel.trim()) {
error.value = "Укажите канал Telegram";
return;
}
if (form.value.source_type === "crawl4ai" && !parseLines(form.value.urlsText).length) {
error.value = "Укажите хотя бы один URL";
return;
}
if (form.value.source_type === "viina") {
const urls = parseLines(form.value.urlsText);
const texts = parseLines(form.value.textsText);
if (form.value.input_mode === "urls" && !urls.length) {
error.value = "Укажите URL для VIINA";
return;
}
if (form.value.input_mode === "texts" && !texts.length) {
error.value = "Укажите тексты для VIINA";
return;
}
if (form.value.input_mode === "mixed" && !urls.length && !texts.length) {
error.value = "Укажите URL или тексты";
return;
}
}
submitting.value = true;
error.value = "";
try {
await createJob({
source_type: form.value.source_type,
source_config: {
channel: form.value.channel.trim(),
limit: form.value.limit,
},
source_config: buildSourceConfig(form.value.source_type, form.value),
interval_seconds: form.value.interval_seconds,
is_active: true,
});
form.value.channel = "";
form.value.urlsText = "";
form.value.textsText = "";
await loadJobs();
} catch (err) {
error.value = err instanceof Error ? err.message : "Не удалось создать задание";
@@ -93,6 +208,17 @@ function openEdit(job: ParseJob) {
editForm.value = {
channel: channelFromConfig(job.source_config),
limit: limitFromConfig(job.source_config),
urlsText: urlsFromConfig(job.source_config).join("\n"),
textsText: textsFromConfig(job.source_config).join("\n"),
domain_profile:
typeof job.source_config.domain_profile === "string"
? job.source_config.domain_profile
: "generic_news",
input_mode:
job.source_config.input_mode === "texts" || job.source_config.input_mode === "mixed"
? job.source_config.input_mode
: "urls",
extract_mode: job.source_config.extract_mode === "llm" ? "llm" : "heuristic",
interval_seconds: job.interval_seconds,
is_active: job.is_active,
};
@@ -104,7 +230,9 @@ function closeEdit() {
async function handleSaveEdit() {
if (!editingJob.value) return;
if (!editForm.value.channel.trim()) {
const sourceType = editingJob.value.source_type;
if (sourceType === "telegram" && !editForm.value.channel.trim()) {
error.value = "Укажите канал Telegram";
return;
}
@@ -113,10 +241,7 @@ async function handleSaveEdit() {
error.value = "";
try {
await updateJob(editingJob.value.id, {
source_config: {
channel: editForm.value.channel.trim(),
limit: editForm.value.limit,
},
source_config: buildSourceConfig(sourceType, editForm.value),
interval_seconds: editForm.value.interval_seconds,
is_active: editForm.value.is_active,
});
@@ -130,7 +255,7 @@ async function handleSaveEdit() {
}
async function handleDelete(job: ParseJob) {
if (!window.confirm(`Удалить парсер #${job.id} (${channelFromConfig(job.source_config)})?`)) {
if (!window.confirm(`Удалить парсер #${job.id} (${sourceSummary(job)})?`)) {
return;
}
error.value = "";
@@ -181,26 +306,75 @@ onUnmounted(() => {
<section class="card">
<h3>Новый парсер</h3>
<p class="hint">
Активные парсеры подхватываются real-time listener (новые посты сразу в БД).
Дополнительно batch-прогон по интервалу забирает последние N постов.
{{ formHint }}
Дубликаты по <code>source_url</code> не записываются.
</p>
<form class="admin-form" @submit.prevent="handleSubmit">
<div class="form-row">
<label>
Тип источника
<select v-model="form.source_type" disabled>
<option value="telegram">Telegram</option>
<select v-model="form.source_type">
<option v-for="opt in SOURCE_TYPES" :key="opt.value" :value="opt.value">
{{ opt.label }}
</option>
</select>
</label>
<label>
Канал
<input v-model="form.channel" type="text" placeholder="creamy_caprice" required />
</label>
<label>
Лимит
<input v-model.number="form.limit" type="number" min="1" max="1000" />
</label>
<template v-if="form.source_type === 'telegram'">
<label>
Канал
<input v-model="form.channel" type="text" placeholder="creamy_caprice" required />
</label>
<label>
Лимит
<input v-model.number="form.limit" type="number" min="1" max="1000" />
</label>
<label>
Извлечение
<select v-model="form.extract_mode">
<option value="heuristic">Эвристики (структурированные посты)</option>
<option value="llm">LLM DeepSeek (неструктурированные)</option>
</select>
</label>
</template>
<template v-else-if="form.source_type === 'crawl4ai'">
<label class="span-2">
URL (по одному в строке)
<textarea v-model="form.urlsText" rows="3" placeholder="https://example.com/news/…" required />
</label>
<label>
Domain profile
<input v-model="form.domain_profile" type="text" placeholder="generic_news" />
</label>
<label>
Извлечение
<select v-model="form.extract_mode">
<option value="heuristic">Эвристики (regex)</option>
<option value="llm">LLM DeepSeek (Crawl4AI)</option>
</select>
</label>
</template>
<template v-else>
<label>
Режим ввода
<select v-model="form.input_mode">
<option value="urls">URLs</option>
<option value="texts">Texts</option>
<option value="mixed">Mixed</option>
</select>
</label>
<label v-if="form.input_mode !== 'texts'" class="span-2">
URL (по одному в строке)
<textarea v-model="form.urlsText" rows="3" placeholder="https://example.com/article…" />
</label>
<label v-if="form.input_mode !== 'urls'" class="span-2">
Тексты (по одному блоку в строке)
<textarea v-model="form.textsText" rows="3" placeholder="Текст статьи…" />
</label>
</template>
<label>
Интервал
<select v-model.number="form.interval_seconds">
@@ -226,8 +400,8 @@ onUnmounted(() => {
<thead>
<tr>
<th>ID</th>
<th>Канал</th>
<th>Лимит</th>
<th>Тип</th>
<th>Источник</th>
<th>Интервал</th>
<th>Активен</th>
<th>Статус</th>
@@ -242,8 +416,8 @@ onUnmounted(() => {
</tr>
<tr v-for="job in jobs" :key="job.id">
<td>{{ job.id }}</td>
<td>{{ channelFromConfig(job.source_config) }}</td>
<td>{{ limitFromConfig(job.source_config) }}</td>
<td>{{ job.source_type }}</td>
<td class="source-cell">{{ sourceSummary(job) }}</td>
<td>{{ formatInterval(job.interval_seconds) }}</td>
<td>{{ job.is_active ? "да" : "нет" }}</td>
<td>
@@ -278,17 +452,61 @@ onUnmounted(() => {
<div v-if="editingJob" class="modal-backdrop" @click.self="closeEdit">
<div class="modal card">
<h3>Редактировать парсер #{{ editingJob.id }}</h3>
<h3>Редактировать парсер #{{ editingJob.id }} ({{ editingJob.source_type }})</h3>
<form class="admin-form" @submit.prevent="handleSaveEdit">
<div class="form-row">
<label>
Канал
<input v-model="editForm.channel" type="text" required />
</label>
<label>
Лимит
<input v-model.number="editForm.limit" type="number" min="1" max="1000" />
</label>
<template v-if="editingJob.source_type === 'telegram'">
<label>
Канал
<input v-model="editForm.channel" type="text" required />
</label>
<label>
Лимит
<input v-model.number="editForm.limit" type="number" min="1" max="1000" />
</label>
<label>
Извлечение
<select v-model="editForm.extract_mode">
<option value="heuristic">Эвристики</option>
<option value="llm">LLM DeepSeek</option>
</select>
</label>
</template>
<template v-else-if="editingJob.source_type === 'crawl4ai'">
<label class="span-2">
URL (по одному в строке)
<textarea v-model="editForm.urlsText" rows="3" required />
</label>
<label>
Domain profile
<input v-model="editForm.domain_profile" type="text" />
</label>
<label>
Извлечение
<select v-model="editForm.extract_mode">
<option value="heuristic">Эвристики</option>
<option value="llm">LLM DeepSeek</option>
</select>
</label>
</template>
<template v-else>
<label>
Режим ввода
<select v-model="editForm.input_mode">
<option value="urls">URLs</option>
<option value="texts">Texts</option>
<option value="mixed">Mixed</option>
</select>
</label>
<label v-if="editForm.input_mode !== 'texts'" class="span-2">
URL
<textarea v-model="editForm.urlsText" rows="3" />
</label>
<label v-if="editForm.input_mode !== 'urls'" class="span-2">
Тексты
<textarea v-model="editForm.textsText" rows="3" />
</label>
</template>
<label>
Интервал
<select v-model.number="editForm.interval_seconds">
@@ -322,6 +540,26 @@ onUnmounted(() => {
line-height: 1.5;
}
.form-row .span-2 {
grid-column: span 2;
}
.form-row textarea {
width: 100%;
font: inherit;
padding: 0.4rem 0.5rem;
border: 1px solid #d1d5db;
border-radius: 4px;
resize: vertical;
}
.source-cell {
max-width: 280px;
overflow: hidden;
text-overflow: ellipsis;
white-space: nowrap;
}
.actions-cell {
white-space: nowrap;
}
@@ -346,8 +584,10 @@ onUnmounted(() => {
}
.modal {
width: min(520px, 92vw);
width: min(560px, 92vw);
margin: 0;
max-height: 90vh;
overflow: auto;
}
.checkbox-row {