"""Reusable LLM extract profile (CA storage + flattened into TelegramSourceConfig).""" from __future__ import annotations from typing import Any from pydantic import BaseModel, Field, field_validator, model_validator from contracts.heuristic_profile import TARGET_FIELDS # LLM defaults omit region (same as workers/llm_extract.DEFAULT_EXTRACT_SCHEMA). LLM_SCHEMA_FIELDS: tuple[str, ...] = ( "title", "locality", "event_date", "description", "coords", "topic", ) DEFAULT_EXTRACT_SCHEMA: dict[str, str] = { "title": "string — short event title", "locality": "string — place / settlement name", "event_date": "string — date as DD.MM.YYYY or YYYY-MM-DD if known", "description": "string — concise event summary", "coords": "string — latitude, longitude if present else empty", "topic": "string — short topic tag", } DEFAULT_INSTRUCTION = ( "Extract structured military/news event fields from the text. " "If the text is not an event, return is_event=false. " "Respond with a single JSON object only." ) class LlmProfile(BaseModel): instruction: str | None = None extract_schema: dict[str, str] = Field(default_factory=lambda: dict(DEFAULT_EXTRACT_SCHEMA)) required_fields: list[str] = Field(default_factory=list) # When true, LLM may return multiple events per post (array "events") multi_event: bool = False @field_validator("instruction", mode="before") @classmethod def empty_instruction_to_none(cls, value: Any) -> Any: if value is None: return None if isinstance(value, str) and not value.strip(): return None return value @field_validator("required_fields") @classmethod def known_required(cls, value: list[str]) -> list[str]: seen: list[str] = [] unknown: list[str] = [] for name in value: if name not in TARGET_FIELDS: unknown.append(name) continue if name not in seen: seen.append(name) if unknown: raise ValueError(f"Unknown required_fields: {sorted(set(unknown))}") return seen @model_validator(mode="after") def known_schema_keys(self) -> "LlmProfile": if not self.extract_schema: raise ValueError("extract_schema must not be empty") unknown = set(self.extract_schema) - set(TARGET_FIELDS) if unknown: raise ValueError(f"Unknown extract_schema fields: {sorted(unknown)}") return self def match_llm_required(fields: dict[str, Any], required_fields: list[str]) -> bool: """True if all required_fields are non-empty strings (empty required = no gate).""" if not required_fields: return True for name in required_fields: val = fields.get(name) if val is None or not str(val).strip(): return False return True