Files
gitrusprusandCursor 3f9dc6643b Add reusable LLM parser profiles with multi-event extract.
Support kind=llm profiles (instruction/schema), optional multi-event posts via #eN URLs, and recover stale running/queued parse jobs after worker crashes.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-09-13 20:22:38 +03:00

87 lines
2.9 KiB
Python

"""Reusable LLM extract profile (CA storage + flattened into TelegramSourceConfig)."""
from __future__ import annotations
from typing import Any
from pydantic import BaseModel, Field, field_validator, model_validator
from contracts.heuristic_profile import TARGET_FIELDS
# LLM defaults omit region (same as workers/llm_extract.DEFAULT_EXTRACT_SCHEMA).
LLM_SCHEMA_FIELDS: tuple[str, ...] = (
"title",
"locality",
"event_date",
"description",
"coords",
"topic",
)
DEFAULT_EXTRACT_SCHEMA: dict[str, str] = {
"title": "string — short event title",
"locality": "string — place / settlement name",
"event_date": "string — date as DD.MM.YYYY or YYYY-MM-DD if known",
"description": "string — concise event summary",
"coords": "string — latitude, longitude if present else empty",
"topic": "string — short topic tag",
}
DEFAULT_INSTRUCTION = (
"Extract structured military/news event fields from the text. "
"If the text is not an event, return is_event=false. "
"Respond with a single JSON object only."
)
class LlmProfile(BaseModel):
instruction: str | None = None
extract_schema: dict[str, str] = Field(default_factory=lambda: dict(DEFAULT_EXTRACT_SCHEMA))
required_fields: list[str] = Field(default_factory=list)
# When true, LLM may return multiple events per post (array "events")
multi_event: bool = False
@field_validator("instruction", mode="before")
@classmethod
def empty_instruction_to_none(cls, value: Any) -> Any:
if value is None:
return None
if isinstance(value, str) and not value.strip():
return None
return value
@field_validator("required_fields")
@classmethod
def known_required(cls, value: list[str]) -> list[str]:
seen: list[str] = []
unknown: list[str] = []
for name in value:
if name not in TARGET_FIELDS:
unknown.append(name)
continue
if name not in seen:
seen.append(name)
if unknown:
raise ValueError(f"Unknown required_fields: {sorted(set(unknown))}")
return seen
@model_validator(mode="after")
def known_schema_keys(self) -> "LlmProfile":
if not self.extract_schema:
raise ValueError("extract_schema must not be empty")
unknown = set(self.extract_schema) - set(TARGET_FIELDS)
if unknown:
raise ValueError(f"Unknown extract_schema fields: {sorted(unknown)}")
return self
def match_llm_required(fields: dict[str, Any], required_fields: list[str]) -> bool:
"""True if all required_fields are non-empty strings (empty required = no gate)."""
if not required_fields:
return True
for name in required_fields:
val = fields.get(name)
if val is None or not str(val).strip():
return False
return True