Add reusable LLM parser profiles with multi-event extract.
Support kind=llm profiles (instruction/schema), optional multi-event posts via #eN URLs, and recover stale running/queued parse jobs after worker crashes. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -0,0 +1,86 @@
|
||||
"""Reusable LLM extract profile (CA storage + flattened into TelegramSourceConfig)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from pydantic import BaseModel, Field, field_validator, model_validator
|
||||
|
||||
from contracts.heuristic_profile import TARGET_FIELDS
|
||||
|
||||
# LLM defaults omit region (same as workers/llm_extract.DEFAULT_EXTRACT_SCHEMA).
|
||||
LLM_SCHEMA_FIELDS: tuple[str, ...] = (
|
||||
"title",
|
||||
"locality",
|
||||
"event_date",
|
||||
"description",
|
||||
"coords",
|
||||
"topic",
|
||||
)
|
||||
|
||||
DEFAULT_EXTRACT_SCHEMA: dict[str, str] = {
|
||||
"title": "string — short event title",
|
||||
"locality": "string — place / settlement name",
|
||||
"event_date": "string — date as DD.MM.YYYY or YYYY-MM-DD if known",
|
||||
"description": "string — concise event summary",
|
||||
"coords": "string — latitude, longitude if present else empty",
|
||||
"topic": "string — short topic tag",
|
||||
}
|
||||
|
||||
DEFAULT_INSTRUCTION = (
|
||||
"Extract structured military/news event fields from the text. "
|
||||
"If the text is not an event, return is_event=false. "
|
||||
"Respond with a single JSON object only."
|
||||
)
|
||||
|
||||
|
||||
class LlmProfile(BaseModel):
|
||||
instruction: str | None = None
|
||||
extract_schema: dict[str, str] = Field(default_factory=lambda: dict(DEFAULT_EXTRACT_SCHEMA))
|
||||
required_fields: list[str] = Field(default_factory=list)
|
||||
# When true, LLM may return multiple events per post (array "events")
|
||||
multi_event: bool = False
|
||||
|
||||
@field_validator("instruction", mode="before")
|
||||
@classmethod
|
||||
def empty_instruction_to_none(cls, value: Any) -> Any:
|
||||
if value is None:
|
||||
return None
|
||||
if isinstance(value, str) and not value.strip():
|
||||
return None
|
||||
return value
|
||||
|
||||
@field_validator("required_fields")
|
||||
@classmethod
|
||||
def known_required(cls, value: list[str]) -> list[str]:
|
||||
seen: list[str] = []
|
||||
unknown: list[str] = []
|
||||
for name in value:
|
||||
if name not in TARGET_FIELDS:
|
||||
unknown.append(name)
|
||||
continue
|
||||
if name not in seen:
|
||||
seen.append(name)
|
||||
if unknown:
|
||||
raise ValueError(f"Unknown required_fields: {sorted(set(unknown))}")
|
||||
return seen
|
||||
|
||||
@model_validator(mode="after")
|
||||
def known_schema_keys(self) -> "LlmProfile":
|
||||
if not self.extract_schema:
|
||||
raise ValueError("extract_schema must not be empty")
|
||||
unknown = set(self.extract_schema) - set(TARGET_FIELDS)
|
||||
if unknown:
|
||||
raise ValueError(f"Unknown extract_schema fields: {sorted(unknown)}")
|
||||
return self
|
||||
|
||||
|
||||
def match_llm_required(fields: dict[str, Any], required_fields: list[str]) -> bool:
|
||||
"""True if all required_fields are non-empty strings (empty required = no gate)."""
|
||||
if not required_fields:
|
||||
return True
|
||||
for name in required_fields:
|
||||
val = fields.get(name)
|
||||
if val is None or not str(val).strip():
|
||||
return False
|
||||
return True
|
||||
Reference in New Issue
Block a user