Add CP source adapter registry with multi-worker queues and LLM extract.
Replace legacy root backend/frontend with Telegram, Crawl4AI, and VIINA adapters routed by Redis job families. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -0,0 +1 @@
|
||||
"""Shared CA ↔ CP contracts (jobs, ingest, source configs, queues)."""
|
||||
|
||||
@@ -2,8 +2,13 @@
|
||||
|
||||
from pydantic import BaseModel, Field
|
||||
|
||||
from .queues import queue_key_for_source
|
||||
|
||||
|
||||
class JobPayload(BaseModel):
|
||||
job_id: int
|
||||
source_type: str
|
||||
source_config: dict = Field(default_factory=dict)
|
||||
|
||||
def queue_key(self) -> str:
|
||||
return queue_key_for_source(self.source_type)
|
||||
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Redis queue routing for CP adapter families."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
# source_type → worker family (separate Redis list + Docker image)
|
||||
SOURCE_FAMILY: dict[str, str] = {
|
||||
"telegram": "telegram",
|
||||
"crawl4ai": "web",
|
||||
"viina": "nlp",
|
||||
}
|
||||
|
||||
LEGACY_JOB_QUEUE_KEY = "cp:jobs"
|
||||
QUEUE_PREFIX = "cp:jobs"
|
||||
|
||||
|
||||
def family_for_source(source_type: str) -> str:
|
||||
try:
|
||||
return SOURCE_FAMILY[source_type]
|
||||
except KeyError as exc:
|
||||
raise ValueError(f"Unknown source_type: {source_type}") from exc
|
||||
|
||||
|
||||
def queue_key_for_source(source_type: str) -> str:
|
||||
return f"{QUEUE_PREFIX}:{family_for_source(source_type)}"
|
||||
|
||||
|
||||
def queue_key_for_family(family: str) -> str:
|
||||
return f"{QUEUE_PREFIX}:{family}"
|
||||
|
||||
|
||||
def known_source_types() -> list[str]:
|
||||
return sorted(SOURCE_FAMILY.keys())
|
||||
@@ -0,0 +1,86 @@
|
||||
"""Per-source_type source_config schemas (CA admin + CP adapters)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field, field_validator, model_validator
|
||||
|
||||
|
||||
class TelegramSourceConfig(BaseModel):
|
||||
channel: str = Field(min_length=1)
|
||||
limit: int = Field(default=100, ge=1, le=1000)
|
||||
# heuristic = legacy telegram_events parser; llm = DeepSeek structured extract
|
||||
extract_mode: Literal["heuristic", "llm"] = "heuristic"
|
||||
extract_schema: dict[str, str] | None = None
|
||||
instruction: str | None = None
|
||||
|
||||
@field_validator("channel")
|
||||
@classmethod
|
||||
def strip_channel(cls, value: str) -> str:
|
||||
return value.strip()
|
||||
|
||||
|
||||
class Crawl4AISourceConfig(BaseModel):
|
||||
urls: list[str] = Field(min_length=1)
|
||||
extract_schema: dict[str, str] = Field(
|
||||
default_factory=lambda: {
|
||||
"title": "string",
|
||||
"locality": "string",
|
||||
"event_date": "string",
|
||||
"description": "string",
|
||||
"coords": "string",
|
||||
"topic": "string",
|
||||
}
|
||||
)
|
||||
domain_profile: str = "generic_news"
|
||||
extract_mode: Literal["heuristic", "llm"] = "heuristic"
|
||||
instruction: str | None = None
|
||||
|
||||
@field_validator("urls")
|
||||
@classmethod
|
||||
def non_empty_urls(cls, value: list[str]) -> list[str]:
|
||||
cleaned = [u.strip() for u in value if u and u.strip()]
|
||||
if not cleaned:
|
||||
raise ValueError("urls must contain at least one URL")
|
||||
return cleaned
|
||||
|
||||
|
||||
class ViinaSourceConfig(BaseModel):
|
||||
urls: list[str] = Field(default_factory=list)
|
||||
texts: list[str] = Field(default_factory=list)
|
||||
input_mode: Literal["urls", "texts", "mixed"] = "urls"
|
||||
|
||||
@field_validator("urls")
|
||||
@classmethod
|
||||
def strip_urls(cls, value: list[str]) -> list[str]:
|
||||
return [u.strip() for u in value if u and u.strip()]
|
||||
|
||||
@field_validator("texts")
|
||||
@classmethod
|
||||
def strip_texts(cls, value: list[str]) -> list[str]:
|
||||
return [t.strip() for t in value if t and t.strip()]
|
||||
|
||||
@model_validator(mode="after")
|
||||
def require_inputs(self) -> "ViinaSourceConfig":
|
||||
if self.input_mode == "urls" and not self.urls:
|
||||
raise ValueError("urls required when input_mode=urls")
|
||||
if self.input_mode == "texts" and not self.texts:
|
||||
raise ValueError("texts required when input_mode=texts")
|
||||
if self.input_mode == "mixed" and not self.urls and not self.texts:
|
||||
raise ValueError("urls or texts required when input_mode=mixed")
|
||||
return self
|
||||
|
||||
|
||||
CONFIG_MODELS: dict[str, type[BaseModel]] = {
|
||||
"telegram": TelegramSourceConfig,
|
||||
"crawl4ai": Crawl4AISourceConfig,
|
||||
"viina": ViinaSourceConfig,
|
||||
}
|
||||
|
||||
|
||||
def parse_source_config(source_type: str, raw: dict) -> BaseModel:
|
||||
model = CONFIG_MODELS.get(source_type)
|
||||
if model is None:
|
||||
raise ValueError(f"Unknown source_type: {source_type}")
|
||||
return model.model_validate(raw or {})
|
||||
Reference in New Issue
Block a user