Support kind=llm profiles (instruction/schema), optional multi-event posts via #eN URLs, and recover stale running/queued parse jobs after worker crashes. Co-authored-by: Cursor <cursoragent@cursor.com>
87 lines
2.9 KiB
Python
87 lines
2.9 KiB
Python
"""Reusable LLM extract profile (CA storage + flattened into TelegramSourceConfig)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Any
|
|
|
|
from pydantic import BaseModel, Field, field_validator, model_validator
|
|
|
|
from contracts.heuristic_profile import TARGET_FIELDS
|
|
|
|
# LLM defaults omit region (same as workers/llm_extract.DEFAULT_EXTRACT_SCHEMA).
|
|
LLM_SCHEMA_FIELDS: tuple[str, ...] = (
|
|
"title",
|
|
"locality",
|
|
"event_date",
|
|
"description",
|
|
"coords",
|
|
"topic",
|
|
)
|
|
|
|
DEFAULT_EXTRACT_SCHEMA: dict[str, str] = {
|
|
"title": "string — short event title",
|
|
"locality": "string — place / settlement name",
|
|
"event_date": "string — date as DD.MM.YYYY or YYYY-MM-DD if known",
|
|
"description": "string — concise event summary",
|
|
"coords": "string — latitude, longitude if present else empty",
|
|
"topic": "string — short topic tag",
|
|
}
|
|
|
|
DEFAULT_INSTRUCTION = (
|
|
"Extract structured military/news event fields from the text. "
|
|
"If the text is not an event, return is_event=false. "
|
|
"Respond with a single JSON object only."
|
|
)
|
|
|
|
|
|
class LlmProfile(BaseModel):
|
|
instruction: str | None = None
|
|
extract_schema: dict[str, str] = Field(default_factory=lambda: dict(DEFAULT_EXTRACT_SCHEMA))
|
|
required_fields: list[str] = Field(default_factory=list)
|
|
# When true, LLM may return multiple events per post (array "events")
|
|
multi_event: bool = False
|
|
|
|
@field_validator("instruction", mode="before")
|
|
@classmethod
|
|
def empty_instruction_to_none(cls, value: Any) -> Any:
|
|
if value is None:
|
|
return None
|
|
if isinstance(value, str) and not value.strip():
|
|
return None
|
|
return value
|
|
|
|
@field_validator("required_fields")
|
|
@classmethod
|
|
def known_required(cls, value: list[str]) -> list[str]:
|
|
seen: list[str] = []
|
|
unknown: list[str] = []
|
|
for name in value:
|
|
if name not in TARGET_FIELDS:
|
|
unknown.append(name)
|
|
continue
|
|
if name not in seen:
|
|
seen.append(name)
|
|
if unknown:
|
|
raise ValueError(f"Unknown required_fields: {sorted(set(unknown))}")
|
|
return seen
|
|
|
|
@model_validator(mode="after")
|
|
def known_schema_keys(self) -> "LlmProfile":
|
|
if not self.extract_schema:
|
|
raise ValueError("extract_schema must not be empty")
|
|
unknown = set(self.extract_schema) - set(TARGET_FIELDS)
|
|
if unknown:
|
|
raise ValueError(f"Unknown extract_schema fields: {sorted(unknown)}")
|
|
return self
|
|
|
|
|
|
def match_llm_required(fields: dict[str, Any], required_fields: list[str]) -> bool:
|
|
"""True if all required_fields are non-empty strings (empty required = no gate)."""
|
|
if not required_fields:
|
|
return True
|
|
for name in required_fields:
|
|
val = fields.get(name)
|
|
if val is None or not str(val).strip():
|
|
return False
|
|
return True
|