Files
microfish/backend/app/services/memory_extraction.py
Kunthawat Greethong 8b84378fe1 feat: SaaS foundation for CrowdSight
Elevate MiroFish/CrowdSight from single-container dev to a SaaS foundation:

- Local memory backend (Zep-compatible): memory services/models, local graph
  builder + updater, AgentActivity seam, import-boundary isolation; Zep stays
  default, local is opt-in behind MEMORY_BACKEND. Semantic parity not yet proven.
- Durable product persistence: projects/simulations/reports schema (migration
  0007) + tenant/owner-scoped ProductRepository + dual-write + scoped_project
  read-first + ArtifactStore abstraction; durable JobQueue + worker.py.
- SaaS hardening: durable RateLimiter (wired to login), UsageService (LLM
  accounting), redacted AuditService, idempotency, CORS allowlist, safe API
  errors, single-use PasswordResetService + endpoints (covers invite-pending).
- Exactly 3 roles (super_admin/admin/user) with tenant authz policy.
- Admin UI: GET/POST/PATCH /api/admin/users + GET/PUT /api/admin/settings
  (super-admin only, encrypted/masked); AdminView.vue + SettingsView.vue with
  admin/super-admin route guards, th/en i18n.
- Production deploy topology: multi-stage Dockerfile (frontend build + gunicorn
  wsgi + nginx SPA-proxy + supervisord worker), backend/wsgi.py, gunicorn dep.

Backend 197 passed; frontend 10 tests + build green. ruff unavailable (gap).
No commit of credentials; secrets handled via env/.env.example.
Deferred: Zep semantic A/B parity, object storage cutover, mobile QA, EasyPanel
container build of deploy topology.
2026-08-31 13:05:21 +07:00

144 lines
4.7 KiB
Python

"""Strict LLM contract for extracting local graph memory."""
from __future__ import annotations
import json
from typing import Any
from pydantic import BaseModel, ConfigDict, Field, ValidationError
MAX_EPISODE_CHARS = 20_000
MAX_CONTEXT_CHARS = 12_000
class _StrictModel(BaseModel):
model_config = ConfigDict(extra="forbid")
class ExtractedEntity(_StrictModel):
mention: str = Field(min_length=1, max_length=2_000)
canonical_name: str = Field(min_length=1, max_length=512)
labels: list[str] = Field(default_factory=list, max_length=32)
aliases: list[str] = Field(default_factory=list, max_length=32)
attributes: dict[str, Any] = Field(default_factory=dict)
summary: str = Field(default="", max_length=4_000)
confidence: float = Field(default=0.0, ge=0.0, le=1.0)
class ExtractedEdge(_StrictModel):
source_entity_ref: str = Field(min_length=1, max_length=512)
target_entity_ref: str = Field(min_length=1, max_length=512)
relation: str = Field(min_length=1, max_length=128)
fact: str = Field(min_length=1, max_length=4_000)
attributes: dict[str, Any] = Field(default_factory=dict)
valid_at: str | None = Field(default=None, max_length=128)
invalid_at: str | None = Field(default=None, max_length=128)
expired_at: str | None = Field(default=None, max_length=128)
confidence: float = Field(default=0.0, ge=0.0, le=1.0)
evidence: list[str] = Field(default_factory=list, max_length=32)
class MemoryExtractionResult(_StrictModel):
entities: list[ExtractedEntity] = Field(default_factory=list, max_length=500)
edges: list[ExtractedEdge] = Field(default_factory=list, max_length=1_000)
episode_summary: str = Field(default="", max_length=8_000)
unresolved_mentions: list[str] = Field(default_factory=list, max_length=200)
def parse_extraction_response(raw: str | dict[str, Any]) -> MemoryExtractionResult:
"""Parse and validate an LLM response without accepting extra fields."""
if isinstance(raw, str):
text = raw.strip()
if text.startswith("```"):
lines = text.splitlines()
if lines and lines[0].strip().startswith("```"):
lines = lines[1:]
if lines and lines[-1].strip() == "```":
lines = lines[:-1]
text = "\n".join(lines).strip()
try:
raw = json.loads(text)
except json.JSONDecodeError as exc:
raise ValueError("invalid_memory_json") from exc
if not isinstance(raw, dict):
raise ValueError("invalid_memory_payload")
return MemoryExtractionResult.model_validate(raw)
def _language_instruction(language: str) -> str:
if language == "en":
return "IMPORTANT: Write summaries and facts in English. Return JSON only."
return "IMPORTANT: Write summaries and facts in Thai. Return JSON only."
def build_extraction_prompt(
*,
language: str,
ontology: dict[str, Any],
episode_text: str,
context: str = "",
) -> str:
if not isinstance(episode_text, str) or len(episode_text) > MAX_EPISODE_CHARS:
raise ValueError("episode_too_large")
if not isinstance(context, str) or len(context) > MAX_CONTEXT_CHARS:
raise ValueError("context_too_large")
if not isinstance(ontology, dict):
raise ValueError("invalid_ontology")
ontology_json = json.dumps(ontology, ensure_ascii=False, sort_keys=True)
context_block = context if context else "(none)"
return f"""{_language_instruction(language)}
You extract evidence-grounded graph memory from one episode.
Never invent facts. Use only labels and relations allowed by the ontology.
Use stable entity_refs inside this response; never guess database IDs.
Preserve temporal fields as null when the evidence does not support a date.
Keep confidence between 0 and 1 and keep evidence references when available.
Return JSON only with this shape:
{{
"entities": [{{
"mention": "text span",
"canonical_name": "stable name",
"labels": ["Person"],
"aliases": [],
"attributes": {{}},
"summary": "short evidence-grounded summary",
"confidence": 0.0
}}],
"edges": [{{
"source_entity_ref": "entity-ref",
"target_entity_ref": "entity-ref",
"relation": "RELATION_NAME",
"fact": "evidence-grounded fact",
"attributes": {{}},
"valid_at": null,
"invalid_at": null,
"expired_at": null,
"confidence": 0.0,
"evidence": ["episode reference or span"]
}}],
"episode_summary": "short summary",
"unresolved_mentions": []
}}
Ontology:
{ontology_json}
Additional context:
{context_block}
Episode:
{episode_text}
"""
__all__ = [
"ExtractedEdge",
"ExtractedEntity",
"MemoryExtractionResult",
"build_extraction_prompt",
"parse_extraction_response",
]