Add Profile/Channel entities with CRUD and remove legacy parser builder.

Introduce reusable ParserProfile and ParseChannel, pair jobs with enqueue flatten, Events admin CRUD, and drop inline/legacy parser-builder UI and aliases.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-16 14:51:03 +03:00
co-authored by Cursor
parent e196de7320
commit 71f8cf169e
30 changed files with 2157 additions and 954 deletions
+52 -1
View File
@@ -4,6 +4,7 @@ import sys
from pathlib import Path
import redis
from sqlalchemy.orm import Session
# Allow importing shared contracts from monorepo root in local runs
_HERE = Path(__file__).resolve()
@@ -14,7 +15,9 @@ for _candidate in (_HERE.parent, *_HERE.parents):
break
from contracts.queues import queue_key_for_source # noqa: E402
from contracts.sources import parse_source_config # noqa: E402
from contracts.sources import TelegramSourceConfig, parse_source_config # noqa: E402
from ..models import ParseChannel, ParseJob, ParserProfile # noqa: E402
REDIS_URL = os.getenv("REDIS_URL", "redis://redis:6379/0")
@@ -23,6 +26,47 @@ def get_redis() -> redis.Redis:
return redis.from_url(REDIS_URL, decode_responses=True)
def flatten_pair_config(
channel: ParseChannel,
profile: ParserProfile,
*,
limit: int = 100,
) -> dict:
"""Expand Profile + Channel into Redis/CP source_config."""
if not profile.heuristic_profile:
raise ValueError("ParserProfile.heuristic_profile is empty")
cfg = TelegramSourceConfig(
channel=channel.channel.strip(),
limit=limit,
extract_mode="profile",
heuristic_profile=profile.heuristic_profile,
sample_post=profile.sample_post or None,
)
return cfg.model_dump()
def resolve_job_source_config(db: Session, job: ParseJob) -> dict:
"""
For pair jobs, rebuild flat source_config from current Profile/Channel.
Legacy jobs keep stored source_config.
"""
if job.profile_id and job.channel_id:
profile = db.query(ParserProfile).filter(ParserProfile.id == job.profile_id).first()
channel = db.query(ParseChannel).filter(ParseChannel.id == job.channel_id).first()
if not profile:
raise ValueError(f"ParserProfile {job.profile_id} not found")
if not channel:
raise ValueError(f"ParseChannel {job.channel_id} not found")
existing = job.source_config or {}
limit = existing.get("limit", 100)
if not isinstance(limit, int) or limit < 1:
limit = 100
flattened = flatten_pair_config(channel, profile, limit=limit)
job.source_config = flattened
return flattened
return dict(job.source_config or {})
def enqueue_job(job_id: int, source_type: str, source_config: dict) -> None:
# Validate known configs early; unknown types still raise from queue_key_for_source
try:
@@ -38,3 +82,10 @@ def enqueue_job(job_id: int, source_type: str, source_config: dict) -> None:
}
key = queue_key_for_source(source_type)
get_redis().rpush(key, json.dumps(payload))
def enqueue_parse_job(db: Session, job: ParseJob) -> None:
"""Resolve (flatten if pair) then push to Redis."""
source_config = resolve_job_source_config(db, job)
db.commit()
enqueue_job(job.id, job.source_type, source_config)
@@ -5,20 +5,29 @@ from sqlalchemy.engine import Engine
def migrate_schema(engine: Engine) -> None:
"""Apply lightweight schema updates for existing deployments."""
inspector = inspect(engine)
if "parse_jobs" not in inspector.get_table_names():
return
columns = {col["name"] for col in inspector.get_columns("parse_jobs")}
tables = set(inspector.get_table_names())
statements: list[str] = []
if "interval_seconds" not in columns:
statements.append(
"ALTER TABLE parse_jobs ADD COLUMN interval_seconds INTEGER NOT NULL DEFAULT 3600"
)
if "is_active" not in columns:
statements.append(
"ALTER TABLE parse_jobs ADD COLUMN is_active BOOLEAN NOT NULL DEFAULT TRUE"
)
if "parse_jobs" in tables:
columns = {col["name"] for col in inspector.get_columns("parse_jobs")}
if "interval_seconds" not in columns:
statements.append(
"ALTER TABLE parse_jobs ADD COLUMN interval_seconds INTEGER NOT NULL DEFAULT 3600"
)
if "is_active" not in columns:
statements.append(
"ALTER TABLE parse_jobs ADD COLUMN is_active BOOLEAN NOT NULL DEFAULT TRUE"
)
if "profile_id" not in columns:
statements.append("ALTER TABLE parse_jobs ADD COLUMN profile_id INTEGER")
if "channel_id" not in columns:
statements.append("ALTER TABLE parse_jobs ADD COLUMN channel_id INTEGER")
# create_all handles new tables; FKs on existing DBs may need indexes
if "parse_jobs" in tables:
# Re-inspect after potential adds is not needed for FK constraints here —
# create_all + nullable FKs are enough for MVP; optional constraints below.
pass
if not statements:
return
@@ -5,7 +5,7 @@ from datetime import datetime, timezone
from ..database import SessionLocal
from ..models import ParseJob
from .jobs import enqueue_job
from .jobs import enqueue_parse_job
logger = logging.getLogger(__name__)
@@ -36,7 +36,14 @@ def run_scheduler_tick() -> None:
job.status = "queued"
job.last_error = None
db.commit()
enqueue_job(job.id, job.source_type, job.source_config)
try:
enqueue_parse_job(db, job)
except ValueError as exc:
job.status = "failed"
job.last_error = str(exc)
db.commit()
logger.warning("Skip re-queue job %s: %s", job.id, exc)
continue
logger.info("Re-queued recurring job %s (interval %ss)", job.id, job.interval_seconds)
except Exception:
logger.exception("Scheduler tick failed")