Add reusable LLM parser profiles with multi-event extract.
Support kind=llm profiles (instruction/schema), optional multi-event posts via #eN URLs, and recover stale running/queued parse jobs after worker crashes. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -14,6 +14,7 @@ from contracts.heuristic_profile import (
|
||||
HeuristicProfile,
|
||||
TARGET_FIELDS,
|
||||
apply_profile,
|
||||
match_profile,
|
||||
target_field_specs,
|
||||
)
|
||||
|
||||
@@ -47,6 +48,14 @@ def preview_with_profile(sample_post: str, profile: dict[str, Any] | HeuristicPr
|
||||
return apply_profile(sample_post, profile)
|
||||
|
||||
|
||||
def match_preview(
|
||||
sample_post: str,
|
||||
profile: dict[str, Any] | HeuristicProfile,
|
||||
) -> tuple[dict[str, str], bool, list[str]]:
|
||||
matched, fields, missing = match_profile(sample_post, profile)
|
||||
return fields, matched, missing
|
||||
|
||||
|
||||
def empty_preview_fields(preview: dict[str, str]) -> list[str]:
|
||||
return [name for name, value in preview.items() if not (value or "").strip()]
|
||||
|
||||
@@ -191,3 +200,111 @@ async def generate_profile(
|
||||
|
||||
content = data["choices"][0]["message"]["content"]
|
||||
return _parse_profile_response(content)
|
||||
|
||||
|
||||
async def preview_llm_extract(
|
||||
sample_post: str,
|
||||
llm_profile: dict[str, Any],
|
||||
) -> tuple[list[dict[str, str]], dict[str, str], bool, list[str], bool, int]:
|
||||
"""One-shot DeepSeek extract for CA preview (does not persist).
|
||||
|
||||
Returns (events, fields, matched, missing_required, is_event, matched_count).
|
||||
``fields`` is the first event (or empty) for legacy UI compatibility.
|
||||
"""
|
||||
from contracts.llm_profile import DEFAULT_INSTRUCTION, LlmProfile, match_llm_required
|
||||
|
||||
settings = deepseek_settings()
|
||||
if not settings["api_key"]:
|
||||
raise RuntimeError(
|
||||
"DEEPSEEK_API_KEY is not set on ca-api. Add it to .env for LLM preview."
|
||||
)
|
||||
|
||||
sample = (sample_post or "").strip()
|
||||
if not sample:
|
||||
raise ValueError("sample_post is required")
|
||||
|
||||
profile = LlmProfile.model_validate(llm_profile)
|
||||
schema = profile.extract_schema
|
||||
instr = profile.instruction or DEFAULT_INSTRUCTION
|
||||
schema_lines = "\n".join(f"- {k}: {v}" for k, v in schema.items())
|
||||
multi = bool(profile.multi_event)
|
||||
|
||||
if multi:
|
||||
user_prompt = (
|
||||
f"{instr}\n\n"
|
||||
"If the text describes multiple distinct events (different places, "
|
||||
"coords, or dates), return one object per event in \"events\".\n"
|
||||
f"Fields per event:\n{schema_lines}\n\n"
|
||||
'Return JSON: {"is_event": true|false, "events": [{<field>: <string>}, ...]}\n'
|
||||
"If there is no event, return is_event=false and events=[].\n\n"
|
||||
f"Text:\n{sample[:12000]}"
|
||||
)
|
||||
else:
|
||||
user_prompt = (
|
||||
f"{instr}\n\n"
|
||||
f"Fields to extract:\n{schema_lines}\n\n"
|
||||
'Return JSON: {"is_event": true|false, "fields": {<field>: <string>}}\n\n'
|
||||
f"Text:\n{sample[:12000]}"
|
||||
)
|
||||
|
||||
payload = {
|
||||
"model": settings["model"],
|
||||
"messages": [
|
||||
{
|
||||
"role": "system",
|
||||
"content": (
|
||||
"You extract structured event data for a geoint map. "
|
||||
"Output valid JSON only, no markdown."
|
||||
),
|
||||
},
|
||||
{"role": "user", "content": user_prompt},
|
||||
],
|
||||
"temperature": 0.1,
|
||||
"response_format": {"type": "json_object"},
|
||||
}
|
||||
|
||||
url = f"{settings['base_url']}/chat/completions"
|
||||
async with httpx.AsyncClient(timeout=90.0) as client:
|
||||
response = await client.post(
|
||||
url,
|
||||
headers={
|
||||
"Authorization": f"Bearer {settings['api_key']}",
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
json=payload,
|
||||
)
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
|
||||
content = data["choices"][0]["message"]["content"]
|
||||
parsed = _extract_json_object(content)
|
||||
is_event = bool(parsed.get("is_event", True))
|
||||
empty_fields = {key: "" for key in schema}
|
||||
|
||||
if not is_event:
|
||||
return [], empty_fields, False, [], False, 0
|
||||
|
||||
events: list[dict[str, str]] = []
|
||||
if multi and isinstance(parsed.get("events"), list):
|
||||
for item in parsed["events"]:
|
||||
if isinstance(item, dict):
|
||||
events.append({key: str(item.get(key) or "").strip() for key in schema})
|
||||
else:
|
||||
fields_raw = parsed.get("fields") if isinstance(parsed.get("fields"), dict) else parsed
|
||||
if not isinstance(fields_raw, dict):
|
||||
fields_raw = {}
|
||||
events.append({key: str(fields_raw.get(key) or "").strip() for key in schema})
|
||||
|
||||
if not events:
|
||||
return [], empty_fields, False, [], False, 0
|
||||
|
||||
matched_events = [ev for ev in events if match_llm_required(ev, list(profile.required_fields))]
|
||||
fields = events[0]
|
||||
missing = [
|
||||
name
|
||||
for name in profile.required_fields
|
||||
if not str(fields.get(name) or "").strip()
|
||||
]
|
||||
matched = match_llm_required(fields, list(profile.required_fields))
|
||||
return events, fields, matched, missing, True, len(matched_events)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user