Add CP source adapter registry with multi-worker queues and LLM extract.
Replace legacy root backend/frontend with Telegram, Crawl4AI, and VIINA adapters routed by Redis job families. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -0,0 +1,86 @@
|
||||
"""Per-source_type source_config schemas (CA admin + CP adapters)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Literal
|
||||
|
||||
from pydantic import BaseModel, Field, field_validator, model_validator
|
||||
|
||||
|
||||
class TelegramSourceConfig(BaseModel):
|
||||
channel: str = Field(min_length=1)
|
||||
limit: int = Field(default=100, ge=1, le=1000)
|
||||
# heuristic = legacy telegram_events parser; llm = DeepSeek structured extract
|
||||
extract_mode: Literal["heuristic", "llm"] = "heuristic"
|
||||
extract_schema: dict[str, str] | None = None
|
||||
instruction: str | None = None
|
||||
|
||||
@field_validator("channel")
|
||||
@classmethod
|
||||
def strip_channel(cls, value: str) -> str:
|
||||
return value.strip()
|
||||
|
||||
|
||||
class Crawl4AISourceConfig(BaseModel):
|
||||
urls: list[str] = Field(min_length=1)
|
||||
extract_schema: dict[str, str] = Field(
|
||||
default_factory=lambda: {
|
||||
"title": "string",
|
||||
"locality": "string",
|
||||
"event_date": "string",
|
||||
"description": "string",
|
||||
"coords": "string",
|
||||
"topic": "string",
|
||||
}
|
||||
)
|
||||
domain_profile: str = "generic_news"
|
||||
extract_mode: Literal["heuristic", "llm"] = "heuristic"
|
||||
instruction: str | None = None
|
||||
|
||||
@field_validator("urls")
|
||||
@classmethod
|
||||
def non_empty_urls(cls, value: list[str]) -> list[str]:
|
||||
cleaned = [u.strip() for u in value if u and u.strip()]
|
||||
if not cleaned:
|
||||
raise ValueError("urls must contain at least one URL")
|
||||
return cleaned
|
||||
|
||||
|
||||
class ViinaSourceConfig(BaseModel):
|
||||
urls: list[str] = Field(default_factory=list)
|
||||
texts: list[str] = Field(default_factory=list)
|
||||
input_mode: Literal["urls", "texts", "mixed"] = "urls"
|
||||
|
||||
@field_validator("urls")
|
||||
@classmethod
|
||||
def strip_urls(cls, value: list[str]) -> list[str]:
|
||||
return [u.strip() for u in value if u and u.strip()]
|
||||
|
||||
@field_validator("texts")
|
||||
@classmethod
|
||||
def strip_texts(cls, value: list[str]) -> list[str]:
|
||||
return [t.strip() for t in value if t and t.strip()]
|
||||
|
||||
@model_validator(mode="after")
|
||||
def require_inputs(self) -> "ViinaSourceConfig":
|
||||
if self.input_mode == "urls" and not self.urls:
|
||||
raise ValueError("urls required when input_mode=urls")
|
||||
if self.input_mode == "texts" and not self.texts:
|
||||
raise ValueError("texts required when input_mode=texts")
|
||||
if self.input_mode == "mixed" and not self.urls and not self.texts:
|
||||
raise ValueError("urls or texts required when input_mode=mixed")
|
||||
return self
|
||||
|
||||
|
||||
CONFIG_MODELS: dict[str, type[BaseModel]] = {
|
||||
"telegram": TelegramSourceConfig,
|
||||
"crawl4ai": Crawl4AISourceConfig,
|
||||
"viina": ViinaSourceConfig,
|
||||
}
|
||||
|
||||
|
||||
def parse_source_config(source_type: str, raw: dict) -> BaseModel:
|
||||
model = CONFIG_MODELS.get(source_type)
|
||||
if model is None:
|
||||
raise ValueError(f"Unknown source_type: {source_type}")
|
||||
return model.model_validate(raw or {})
|
||||
Reference in New Issue
Block a user