Add parser builder: one-shot DeepSeek profile for Telegram extract_mode=profile.

Generate static HeuristicProfile in CA admin, preview and run without LLM on each post via shared interpreter in CP batch and listener.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-16 14:14:18 +03:00
co-authored by Cursor
parent 6dff3c1c3d
commit 699e9be503
23 changed files with 1042 additions and 20 deletions
+165
View File
@@ -0,0 +1,165 @@
"""Declarative heuristic profile: schema + static rule interpreter (CA preview + CP runtime)."""
from __future__ import annotations
import re
from typing import Any, Literal
from pydantic import BaseModel, Field, field_validator, model_validator
TARGET_FIELDS: tuple[str, ...] = (
"title",
"description",
"locality",
"event_date",
"coords",
"topic",
"region",
)
Strategy = Literal["regex", "line", "after_marker", "between", "full_text", "literal"]
class FieldRule(BaseModel):
strategy: Strategy
# regex: named or group(1); line: 0-based index; after_marker/between: markers
pattern: str | None = None
group: int = 1
line_index: int | None = None
marker: str | None = None
end_marker: str | None = None
value: str | None = None # literal
flags: str = "" # e.g. "im" → re.I|re.M
strip: bool = True
@field_validator("pattern", "marker", "end_marker", "value", mode="before")
@classmethod
def empty_to_none(cls, value: Any) -> Any:
if value is None:
return None
if isinstance(value, str) and not value.strip():
return None
return value
class HeuristicProfile(BaseModel):
version: Literal[1] = 1
fields: dict[str, FieldRule] = Field(default_factory=dict)
notes: str = ""
@model_validator(mode="after")
def known_fields_only(self) -> "HeuristicProfile":
unknown = set(self.fields) - set(TARGET_FIELDS)
if unknown:
raise ValueError(f"Unknown profile fields: {sorted(unknown)}")
return self
def target_field_specs() -> list[dict[str, str]]:
"""Fixed target table for parser-builder UI (roadmap: custom tables later)."""
return [
{"name": "title", "type": "string", "description": "Short event title"},
{"name": "description", "type": "string", "description": "Event summary / body"},
{"name": "locality", "type": "string", "description": "Place / settlement name"},
{
"name": "event_date",
"type": "string",
"description": "Date as DD.MM.YYYY or YYYY-MM-DD",
},
{
"name": "coords",
"type": "string",
"description": "Latitude, longitude if present",
},
{"name": "topic", "type": "string", "description": "Short topic tag"},
{"name": "region", "type": "string", "description": "Region (optional)"},
]
def _compile_flags(flags: str) -> int:
mapping = {
"i": re.IGNORECASE,
"m": re.MULTILINE,
"s": re.DOTALL,
}
result = 0
for ch in (flags or "").lower():
result |= mapping.get(ch, 0)
return result
def _apply_rule(text: str, rule: FieldRule) -> str:
raw = text or ""
value = ""
if rule.strategy == "literal":
value = rule.value or ""
elif rule.strategy == "full_text":
value = raw
elif rule.strategy == "line":
lines = raw.splitlines()
idx = 0 if rule.line_index is None else rule.line_index
if 0 <= idx < len(lines):
value = lines[idx]
elif rule.strategy == "regex":
if not rule.pattern:
return ""
match = re.search(rule.pattern, raw, _compile_flags(rule.flags))
if match:
try:
value = match.group(rule.group)
except IndexError:
value = match.group(0)
elif rule.strategy == "after_marker":
marker = rule.marker or ""
if not marker:
return ""
pos = raw.find(marker)
if pos < 0:
return ""
start = pos + len(marker)
rest = raw[start:]
if rule.end_marker:
end = rest.find(rule.end_marker)
value = rest[:end] if end >= 0 else rest
elif rule.pattern:
match = re.search(rule.pattern, rest, _compile_flags(rule.flags))
if match:
try:
value = match.group(rule.group)
except IndexError:
value = match.group(0)
else:
# first non-empty line after marker
for line in rest.splitlines():
if line.strip():
value = line
break
elif rule.strategy == "between":
start_m = rule.marker or ""
end_m = rule.end_marker or ""
if not start_m or not end_m:
return ""
start = raw.find(start_m)
if start < 0:
return ""
start += len(start_m)
end = raw.find(end_m, start)
if end < 0:
return ""
value = raw[start:end]
if rule.strip:
value = value.strip()
return value
def apply_profile(text: str, profile: HeuristicProfile | dict[str, Any]) -> dict[str, str]:
"""Apply static rules to post text → string field map (no LLM)."""
if isinstance(profile, dict):
profile = HeuristicProfile.model_validate(profile)
result: dict[str, str] = {name: "" for name in TARGET_FIELDS}
for name, rule in profile.fields.items():
result[name] = _apply_rule(text, rule)
return result
+15 -2
View File
@@ -10,16 +10,29 @@ from pydantic import BaseModel, Field, field_validator, model_validator
class TelegramSourceConfig(BaseModel):
channel: str = Field(min_length=1)
limit: int = Field(default=100, ge=1, le=1000)
# heuristic = legacy telegram_events parser; llm = DeepSeek structured extract
extract_mode: Literal["heuristic", "llm"] = "heuristic"
# heuristic = legacy telegram_events; llm = DeepSeek per post; profile = static rules
extract_mode: Literal["heuristic", "llm", "profile"] = "heuristic"
extract_schema: dict[str, str] | None = None
instruction: str | None = None
heuristic_profile: dict | None = None
sample_post: str | None = None # audit / re-generate; not required at runtime
@field_validator("channel")
@classmethod
def strip_channel(cls, value: str) -> str:
return value.strip()
@model_validator(mode="after")
def profile_requires_rules(self) -> "TelegramSourceConfig":
if self.extract_mode != "profile":
return self
if not self.heuristic_profile:
raise ValueError("heuristic_profile required when extract_mode=profile")
from contracts.heuristic_profile import HeuristicProfile
HeuristicProfile.model_validate(self.heuristic_profile)
return self
class Crawl4AISourceConfig(BaseModel):
urls: list[str] = Field(min_length=1)