Add reusable LLM parser profiles with multi-event extract.
Support kind=llm profiles (instruction/schema), optional multi-event posts via #eN URLs, and recover stale running/queued parse jobs after worker crashes. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -11,9 +11,11 @@ from workers.heuristic_profile import extract_with_profile
|
||||
from workers.llm_extract import (
|
||||
DEFAULT_EXTRACT_SCHEMA,
|
||||
DEFAULT_INSTRUCTION,
|
||||
extract_event_fields,
|
||||
event_source_url,
|
||||
extract_event_list,
|
||||
fields_to_ingest,
|
||||
llm_enabled,
|
||||
match_llm_required,
|
||||
)
|
||||
from workers.parsers.telegram_events import parse_event_posts
|
||||
from workers.sources.telegram_client import (
|
||||
@@ -73,54 +75,72 @@ def _extract_posts_with_profile(posts, cfg: TelegramSourceConfig) -> tuple[list[
|
||||
text = (post.text or "").strip()
|
||||
if not text:
|
||||
continue
|
||||
events.append(
|
||||
extract_with_profile(
|
||||
text,
|
||||
profile,
|
||||
source_url=post.url,
|
||||
source_type="telegram",
|
||||
extra_metadata={
|
||||
"channel": post.channel,
|
||||
"message_id": post.id,
|
||||
"post_date": post.date.isoformat() if post.date else None,
|
||||
},
|
||||
)
|
||||
event = extract_with_profile(
|
||||
text,
|
||||
profile,
|
||||
source_url=post.url,
|
||||
source_type="telegram",
|
||||
extra_metadata={
|
||||
"channel": post.channel,
|
||||
"message_id": post.id,
|
||||
"post_date": post.date.isoformat() if post.date else None,
|
||||
},
|
||||
)
|
||||
if event is None:
|
||||
logger.debug("Profile skip %s (required fields missing)", post.url)
|
||||
continue
|
||||
events.append(event)
|
||||
return events, None
|
||||
|
||||
|
||||
async def _extract_posts_with_llm(posts, cfg: TelegramSourceConfig) -> tuple[list[dict], str | None]:
|
||||
schema = cfg.extract_schema or DEFAULT_EXTRACT_SCHEMA
|
||||
instruction = cfg.instruction or DEFAULT_INSTRUCTION
|
||||
multi_event = bool(cfg.multi_event)
|
||||
events: list[dict] = []
|
||||
errors: list[str] = []
|
||||
required = list(cfg.required_fields or [])
|
||||
|
||||
for post in posts:
|
||||
text = (post.text or "").strip()
|
||||
if not text:
|
||||
continue
|
||||
try:
|
||||
fields = await extract_event_fields(
|
||||
field_list = await extract_event_list(
|
||||
text,
|
||||
extract_schema=schema,
|
||||
instruction=instruction,
|
||||
multi_event=multi_event,
|
||||
)
|
||||
if not fields.get("is_event", True):
|
||||
if not field_list:
|
||||
continue
|
||||
events.append(
|
||||
fields_to_ingest(
|
||||
source_type="telegram",
|
||||
source_url=post.url,
|
||||
raw_text=text,
|
||||
fields=fields,
|
||||
domain_profile="telegram_llm",
|
||||
extra_metadata={
|
||||
"channel": post.channel,
|
||||
"message_id": post.id,
|
||||
"post_date": post.date.isoformat() if post.date else None,
|
||||
},
|
||||
total = len(field_list)
|
||||
for idx, fields in enumerate(field_list):
|
||||
if not match_llm_required(fields, required):
|
||||
logger.debug(
|
||||
"LLM skip %s event %s/%s (required fields missing)",
|
||||
post.url,
|
||||
idx + 1,
|
||||
total,
|
||||
)
|
||||
continue
|
||||
events.append(
|
||||
fields_to_ingest(
|
||||
source_type="telegram",
|
||||
source_url=event_source_url(post.url, idx, total),
|
||||
raw_text=text,
|
||||
fields=fields,
|
||||
domain_profile="telegram_llm",
|
||||
extra_metadata={
|
||||
"channel": post.channel,
|
||||
"message_id": post.id,
|
||||
"post_date": post.date.isoformat() if post.date else None,
|
||||
"event_index": idx + 1,
|
||||
"event_count": total,
|
||||
"multi_event": multi_event and total > 1,
|
||||
},
|
||||
)
|
||||
)
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.exception("LLM extract failed for %s", post.url)
|
||||
errors.append(f"{post.url}: {exc}")
|
||||
|
||||
@@ -4,7 +4,7 @@ from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from contracts.heuristic_profile import HeuristicProfile, apply_profile
|
||||
from contracts.heuristic_profile import HeuristicProfile, match_profile
|
||||
from workers.llm_extract import parse_coords, parse_date
|
||||
|
||||
|
||||
@@ -62,8 +62,10 @@ def extract_with_profile(
|
||||
source_url: str,
|
||||
source_type: str = "telegram",
|
||||
extra_metadata: dict | None = None,
|
||||
) -> dict:
|
||||
fields = apply_profile(text, profile)
|
||||
) -> dict | None:
|
||||
matched, fields, _missing = match_profile(text, profile)
|
||||
if not matched:
|
||||
return None
|
||||
return profile_fields_to_ingest(
|
||||
source_type=source_type,
|
||||
source_url=source_url,
|
||||
|
||||
@@ -5,30 +5,31 @@ from __future__ import annotations
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from contracts.heuristic_profile import parse_coords
|
||||
from contracts.llm_profile import (
|
||||
DEFAULT_EXTRACT_SCHEMA,
|
||||
DEFAULT_INSTRUCTION,
|
||||
match_llm_required,
|
||||
)
|
||||
|
||||
logger = logging.getLogger("cp-worker.llm")
|
||||
|
||||
COORDS_RE = re.compile(r"(-?\d{1,3}\.\d+)\s*,\s*(-?\d{1,3}\.\d+)")
|
||||
|
||||
DEFAULT_EXTRACT_SCHEMA: dict[str, str] = {
|
||||
"title": "string — short event title",
|
||||
"locality": "string — place / settlement name",
|
||||
"event_date": "string — date as DD.MM.YYYY or YYYY-MM-DD if known",
|
||||
"description": "string — concise event summary",
|
||||
"coords": "string — latitude, longitude if present else empty",
|
||||
"topic": "string — short topic tag",
|
||||
}
|
||||
|
||||
DEFAULT_INSTRUCTION = (
|
||||
"Extract structured military/news event fields from the text. "
|
||||
"If the text is not an event, return is_event=false. "
|
||||
"Respond with a single JSON object only."
|
||||
)
|
||||
# Re-export for adapters that import from this module
|
||||
__all__ = [
|
||||
"DEFAULT_EXTRACT_SCHEMA",
|
||||
"DEFAULT_INSTRUCTION",
|
||||
"event_source_url",
|
||||
"extract_event_fields",
|
||||
"extract_event_list",
|
||||
"fields_to_ingest",
|
||||
"llm_enabled",
|
||||
"match_llm_required",
|
||||
]
|
||||
|
||||
|
||||
def llm_enabled() -> bool:
|
||||
@@ -43,29 +44,57 @@ def llm_settings() -> dict[str, str]:
|
||||
}
|
||||
|
||||
|
||||
async def extract_event_fields(
|
||||
text: str,
|
||||
def event_source_url(post_url: str, index: int, total: int) -> str:
|
||||
"""Stable per-event URL for CA dedup. Single event keeps bare post URL."""
|
||||
base = (post_url or "").strip()
|
||||
if total <= 1:
|
||||
return base
|
||||
# index is 0-based; fragment uses 1-based #eN
|
||||
return f"{base}#e{index + 1}"
|
||||
|
||||
|
||||
def _normalize_fields(raw: dict[str, Any], schema: dict[str, str]) -> dict[str, str]:
|
||||
return {key: str(raw.get(key) or "").strip() for key in schema}
|
||||
|
||||
|
||||
def _parse_llm_content(
|
||||
content: str,
|
||||
schema: dict[str, str],
|
||||
*,
|
||||
extract_schema: dict[str, str] | None = None,
|
||||
instruction: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Ask DeepSeek to fill schema fields from free text. Returns dict (+ is_event)."""
|
||||
multi_event: bool,
|
||||
) -> tuple[bool, list[dict[str, str]]]:
|
||||
"""Return (is_event, list of field dicts)."""
|
||||
parsed = json.loads(content)
|
||||
if not isinstance(parsed, dict):
|
||||
return False, []
|
||||
|
||||
is_event = bool(parsed.get("is_event", True))
|
||||
if not is_event:
|
||||
return False, []
|
||||
|
||||
if multi_event and isinstance(parsed.get("events"), list):
|
||||
events: list[dict[str, str]] = []
|
||||
for item in parsed["events"]:
|
||||
if isinstance(item, dict):
|
||||
events.append(_normalize_fields(item, schema))
|
||||
return True, events
|
||||
|
||||
# Single-event fallback (legacy fields / top-level keys)
|
||||
fields_raw = parsed.get("fields") if isinstance(parsed.get("fields"), dict) else parsed
|
||||
if not isinstance(fields_raw, dict):
|
||||
fields_raw = {}
|
||||
# Drop non-schema keys that confuse normalize when falling back to top-level
|
||||
cleaned = {k: fields_raw.get(k) for k in schema}
|
||||
return True, [_normalize_fields(cleaned, schema)]
|
||||
|
||||
|
||||
async def _call_deepseek(user_prompt: str) -> str:
|
||||
settings = llm_settings()
|
||||
if not settings["api_key"]:
|
||||
raise RuntimeError(
|
||||
"DEEPSEEK_API_KEY is not set. Add it to .env for LLM extract_mode."
|
||||
)
|
||||
|
||||
schema = extract_schema or DEFAULT_EXTRACT_SCHEMA
|
||||
instr = instruction or DEFAULT_INSTRUCTION
|
||||
schema_lines = "\n".join(f"- {k}: {v}" for k, v in schema.items())
|
||||
user_prompt = (
|
||||
f"{instr}\n\n"
|
||||
f"Fields to extract:\n{schema_lines}\n\n"
|
||||
'Return JSON: {"is_event": true|false, "fields": {<field>: <string>}}\n\n'
|
||||
f"Text:\n{text[:12000]}"
|
||||
)
|
||||
|
||||
payload = {
|
||||
"model": settings["model"],
|
||||
"messages": [
|
||||
@@ -95,14 +124,72 @@ async def extract_event_fields(
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
content = data["choices"][0]["message"]["content"]
|
||||
parsed = json.loads(content)
|
||||
fields = parsed.get("fields") if isinstance(parsed.get("fields"), dict) else parsed
|
||||
if not isinstance(fields, dict):
|
||||
fields = {}
|
||||
# Normalize to strings for known keys
|
||||
result = {key: str(fields.get(key) or "").strip() for key in schema}
|
||||
result["is_event"] = bool(parsed.get("is_event", True))
|
||||
return data["choices"][0]["message"]["content"]
|
||||
|
||||
|
||||
async def extract_event_list(
|
||||
text: str,
|
||||
*,
|
||||
extract_schema: dict[str, str] | None = None,
|
||||
instruction: str | None = None,
|
||||
multi_event: bool = False,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Extract zero or more event field dicts from free text.
|
||||
|
||||
Each dict has schema keys as strings. Empty list if not an event / no items.
|
||||
"""
|
||||
schema = extract_schema or DEFAULT_EXTRACT_SCHEMA
|
||||
instr = instruction or DEFAULT_INSTRUCTION
|
||||
schema_lines = "\n".join(f"- {k}: {v}" for k, v in schema.items())
|
||||
|
||||
if multi_event:
|
||||
user_prompt = (
|
||||
f"{instr}\n\n"
|
||||
"If the text describes multiple distinct events (different places, "
|
||||
"coords, or dates), return one object per event in \"events\".\n"
|
||||
f"Fields per event:\n{schema_lines}\n\n"
|
||||
'Return JSON: {"is_event": true|false, "events": [{<field>: <string>}, ...]}\n'
|
||||
"If there is no event, return is_event=false and events=[].\n\n"
|
||||
f"Text:\n{text[:12000]}"
|
||||
)
|
||||
else:
|
||||
user_prompt = (
|
||||
f"{instr}\n\n"
|
||||
f"Fields to extract:\n{schema_lines}\n\n"
|
||||
'Return JSON: {"is_event": true|false, "fields": {<field>: <string>}}\n\n'
|
||||
f"Text:\n{text[:12000]}"
|
||||
)
|
||||
|
||||
content = await _call_deepseek(user_prompt)
|
||||
is_event, events = _parse_llm_content(content, schema, multi_event=multi_event)
|
||||
if not is_event:
|
||||
return []
|
||||
return events
|
||||
|
||||
|
||||
async def extract_event_fields(
|
||||
text: str,
|
||||
*,
|
||||
extract_schema: dict[str, str] | None = None,
|
||||
instruction: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Ask DeepSeek to fill schema fields from free text. Returns dict (+ is_event).
|
||||
|
||||
Single-event API kept for crawl4ai and legacy callers.
|
||||
"""
|
||||
events = await extract_event_list(
|
||||
text,
|
||||
extract_schema=extract_schema,
|
||||
instruction=instruction,
|
||||
multi_event=False,
|
||||
)
|
||||
if not events:
|
||||
schema = extract_schema or DEFAULT_EXTRACT_SCHEMA
|
||||
empty = {key: "" for key in schema}
|
||||
empty["is_event"] = False
|
||||
return empty
|
||||
result = dict(events[0])
|
||||
result["is_event"] = True
|
||||
return result
|
||||
|
||||
|
||||
@@ -146,13 +233,6 @@ def fields_to_ingest(
|
||||
}
|
||||
|
||||
|
||||
def parse_coords(raw: str) -> tuple[float | None, float | None]:
|
||||
match = COORDS_RE.search(raw or "")
|
||||
if not match:
|
||||
return None, None
|
||||
return float(match.group(1)), float(match.group(2))
|
||||
|
||||
|
||||
def parse_date(raw: str) -> datetime | None:
|
||||
if not raw:
|
||||
return None
|
||||
|
||||
@@ -81,7 +81,7 @@ class TelegramListener:
|
||||
", ".join(sorted(channels)) or "(none)",
|
||||
)
|
||||
|
||||
def _build_event(self, channel: str, post) -> dict:
|
||||
def _build_event(self, channel: str, post) -> dict | None:
|
||||
sub = self._channels.get(normalize_channel(channel)) or {}
|
||||
cfg = sub.get("source_config") or {}
|
||||
extract_mode = cfg.get("extract_mode") or "heuristic"
|
||||
@@ -107,6 +107,9 @@ class TelegramListener:
|
||||
|
||||
async def _ingest_post(self, channel: str, post) -> None:
|
||||
event = self._build_event(channel, post)
|
||||
if event is None:
|
||||
logger.debug("Listener skip %s (profile required fields missing)", post.url)
|
||||
return
|
||||
sub = self._channels.get(normalize_channel(channel)) or {}
|
||||
job_id = sub.get("job_id")
|
||||
|
||||
|
||||
Reference in New Issue
Block a user