Files
microfish/backend/app/services/local_graph_builder.py
Kunthawat Greethong 8b84378fe1 feat: SaaS foundation for CrowdSight
Elevate MiroFish/CrowdSight from single-container dev to a SaaS foundation:

- Local memory backend (Zep-compatible): memory services/models, local graph
  builder + updater, AgentActivity seam, import-boundary isolation; Zep stays
  default, local is opt-in behind MEMORY_BACKEND. Semantic parity not yet proven.
- Durable product persistence: projects/simulations/reports schema (migration
  0007) + tenant/owner-scoped ProductRepository + dual-write + scoped_project
  read-first + ArtifactStore abstraction; durable JobQueue + worker.py.
- SaaS hardening: durable RateLimiter (wired to login), UsageService (LLM
  accounting), redacted AuditService, idempotency, CORS allowlist, safe API
  errors, single-use PasswordResetService + endpoints (covers invite-pending).
- Exactly 3 roles (super_admin/admin/user) with tenant authz policy.
- Admin UI: GET/POST/PATCH /api/admin/users + GET/PUT /api/admin/settings
  (super-admin only, encrypted/masked); AdminView.vue + SettingsView.vue with
  admin/super-admin route guards, th/en i18n.
- Production deploy topology: multi-stage Dockerfile (frontend build + gunicorn
  wsgi + nginx SPA-proxy + supervisord worker), backend/wsgi.py, gunicorn dep.

Backend 197 passed; frontend 10 tests + build green. ruff unavailable (gap).
No commit of credentials; secrets handled via env/.env.example.
Deferred: Zep semantic A/B parity, object storage cutover, mobile QA, EasyPanel
container build of deploy topology.
2026-08-31 13:05:21 +07:00

198 lines
7.9 KiB
Python

"""Tenant-scoped graph builder backed by the local memory repository."""
from __future__ import annotations
from typing import Any, Callable, Iterable
from uuid import uuid4
from .memory_repository import SqlAlchemyMemoryRepository
from .memory_service import MemoryExtractionService
from ..utils.locale import get_locale
from ..utils.language_policy import normalize_locale
from ..utils.logger import get_logger
logger = get_logger("crowdsight.local_graph_builder")
ProgressCallback = Callable[[str, float], None]
class LocalGraphBuilderService:
"""Build a graph synchronously into local durable memory.
The public methods intentionally mirror ``GraphBuilderService`` so the API
can switch storage backends without giving either backend authorization
authority. Every repository instance is created with the request actor's
organization and the graph being processed.
"""
def __init__(
self,
session_factory,
*,
organization_id: str,
project_id: str,
extraction_client=None,
extraction_service: MemoryExtractionService | None = None,
language: str | None = None,
):
if not callable(session_factory):
raise ValueError("memory_session_factory_required")
if not isinstance(organization_id, str) or not organization_id.strip():
raise ValueError("memory_builder_organization_required")
if not isinstance(project_id, str) or not project_id.strip():
raise ValueError("memory_builder_project_required")
self.session_factory = session_factory
self.organization_id = organization_id
self.project_id = project_id
self.language = normalize_locale(language or get_locale())
self.extraction_service = (
extraction_service
if extraction_service is not None
else MemoryExtractionService(extraction_client) if extraction_client is not None else None
)
self._ontology_by_graph: dict[str, dict[str, Any]] = {}
def _repository(self, session, graph_id: str) -> SqlAlchemyMemoryRepository:
return SqlAlchemyMemoryRepository(
session,
organization_id=self.organization_id,
graph_id=graph_id,
)
@staticmethod
def _validate_graph_id(graph_id: str) -> str:
if not isinstance(graph_id, str) or not graph_id.strip():
raise ValueError("invalid_graph_id")
return graph_id.strip()
def create_graph(self, name: str = "") -> str:
"""Create a new graph and return its durable ID."""
graph_id = f"crowdsight_{uuid4().hex[:24]}"
with self.session_factory() as session:
repository = self._repository(session, graph_id)
repository.create_graph(project_id=self.project_id)
session.commit()
logger.info("Created local memory graph %s", graph_id)
return graph_id
def set_ontology(self, graph_id: str, ontology: dict[str, Any]) -> None:
graph_id = self._validate_graph_id(graph_id)
if not isinstance(ontology, dict):
raise ValueError("invalid_ontology")
with self.session_factory() as session:
repository = self._repository(session, graph_id)
repository.update_graph_ontology(ontology)
session.commit()
self._ontology_by_graph[graph_id] = dict(ontology)
def add_text_batches(
self,
graph_id: str,
text_batches: Iterable[str],
batch_size: int = 3,
progress_callback: ProgressCallback | None = None,
) -> list[str]:
graph_id = self._validate_graph_id(graph_id)
extraction_service = self.extraction_service
if extraction_service is None:
raise ValueError("memory_extraction_client_required")
if isinstance(batch_size, bool) or not isinstance(batch_size, int) or batch_size < 1:
raise ValueError("invalid_batch_size")
chunks = list(text_batches)
if any(not isinstance(chunk, str) or not chunk.strip() for chunk in chunks):
raise ValueError("invalid_text_batch")
episode_ids: list[str] = []
total = len(chunks)
for index, chunk in enumerate(chunks):
with self.session_factory() as session:
repository = self._repository(session, graph_id)
graph = repository.get_graph()
ontology = dict(self._ontology_by_graph.get(graph_id) or graph.ontology)
result = extraction_service.extract(
language=self.language,
ontology=ontology,
episode_text=chunk,
)
ingest_result = extraction_service.persist(
repository,
result,
source_type="text",
source_ref=f"episode_{index}",
episode_text=chunk,
)
episode = repository.get_episode(source_type="text", source_ref=f"episode_{index}")
session.commit()
if episode is None:
raise RuntimeError("local_episode_persist_failed")
episode_ids.append(episode.id)
if progress_callback:
progress_callback(
f"Processed local memory chunk {index + 1}/{total}",
(index + 1) / total if total else 1.0,
)
logger.debug(
"Processed local graph chunk %s: entities=%s edges=%s",
index,
ingest_result.entity_count,
ingest_result.edge_count,
)
return episode_ids
def _wait_for_episodes(
self,
episode_ids: list[str],
progress_callback: ProgressCallback | None = None,
) -> None:
"""Local extraction is committed synchronously; verify IDs instead of polling Zep."""
if not isinstance(episode_ids, list):
raise ValueError("invalid_episode_ids")
if progress_callback:
progress_callback("Local memory processing complete", 1.0)
def get_graph_data(self, graph_id: str) -> dict[str, Any]:
graph_id = self._validate_graph_id(graph_id)
with self.session_factory() as session:
repository = self._repository(session, graph_id)
graph = repository.get_graph()
nodes = repository.list_nodes(limit=10_000)
edges = repository.list_edges(limit=20_000)
return {
"graph_id": graph.id,
"node_count": len(nodes),
"edge_count": len(edges),
"nodes": [
{
"uuid": node.id,
"name": node.canonical_name,
"labels": list(node.labels or []),
"summary": node.summary or "",
"attributes": dict(node.attributes or {}),
}
for node in nodes
],
"edges": [
{
"uuid": edge.id,
"name": edge.relation,
"fact": edge.fact,
"source_node_uuid": edge.source_node_id,
"target_node_uuid": edge.target_node_id,
"attributes": dict(edge.attributes or {}),
}
for edge in edges
],
}
def delete_graph(self, graph_id: str) -> None:
"""Delete a graph only when it belongs to this organization scope."""
graph_id = self._validate_graph_id(graph_id)
with self.session_factory() as session:
repository = self._repository(session, graph_id)
graph = repository.get_graph()
session.delete(graph)
session.commit()
self._ontology_by_graph.pop(graph_id, None)
logger.info("Deleted local memory graph %s", graph_id)