Elevate MiroFish/CrowdSight from single-container dev to a SaaS foundation: - Local memory backend (Zep-compatible): memory services/models, local graph builder + updater, AgentActivity seam, import-boundary isolation; Zep stays default, local is opt-in behind MEMORY_BACKEND. Semantic parity not yet proven. - Durable product persistence: projects/simulations/reports schema (migration 0007) + tenant/owner-scoped ProductRepository + dual-write + scoped_project read-first + ArtifactStore abstraction; durable JobQueue + worker.py. - SaaS hardening: durable RateLimiter (wired to login), UsageService (LLM accounting), redacted AuditService, idempotency, CORS allowlist, safe API errors, single-use PasswordResetService + endpoints (covers invite-pending). - Exactly 3 roles (super_admin/admin/user) with tenant authz policy. - Admin UI: GET/POST/PATCH /api/admin/users + GET/PUT /api/admin/settings (super-admin only, encrypted/masked); AdminView.vue + SettingsView.vue with admin/super-admin route guards, th/en i18n. - Production deploy topology: multi-stage Dockerfile (frontend build + gunicorn wsgi + nginx SPA-proxy + supervisord worker), backend/wsgi.py, gunicorn dep. Backend 197 passed; frontend 10 tests + build green. ruff unavailable (gap). No commit of credentials; secrets handled via env/.env.example. Deferred: Zep semantic A/B parity, object storage cutover, mobile QA, EasyPanel container build of deploy topology.
83 lines
3.0 KiB
Python
83 lines
3.0 KiB
Python
"""Durable job queue: claim/complete/fail lifecycle over the ``jobs`` table.
|
|
|
|
This is the portable core a worker topology builds on; it does not require a
|
|
broker. For PostgreSQL production this should issue ``SELECT ... FOR UPDATE``
|
|
plus ``UPDATE ... WHERE status='queued'`` to claim atomically; the SQLite
|
|
fallback below uses a synchronized in-process write for local/tests.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from datetime import datetime, timezone
|
|
from typing import Callable, Optional
|
|
|
|
from sqlalchemy.orm import Session
|
|
|
|
from ..models.operations import Job, JobStatus
|
|
|
|
|
|
def _utc_now() -> datetime:
|
|
return datetime.now(timezone.utc)
|
|
|
|
|
|
class JobQueue:
|
|
"""Flush-only queue: caller owns the transaction boundary."""
|
|
|
|
def __init__(self, session: Session):
|
|
self.session = session
|
|
self._handlers: dict[str, Callable] = {}
|
|
|
|
def register_handler(self, operation: str, handler) -> None:
|
|
"""Register a callable handler for an operation (worker plugin point)."""
|
|
self._handlers[operation] = handler
|
|
|
|
def dispatch(self, job: Job, *, payload=None):
|
|
"""Invoke the registered handler for ``job.operation``.
|
|
|
|
Returns the handler result. Raises ``ValueError`` when no handler is
|
|
registered so the worker can fail the job.
|
|
"""
|
|
handler = self._handlers.get(job.operation)
|
|
if handler is None:
|
|
raise ValueError(f"no_handler_for_operation: {job.operation}")
|
|
return handler(payload, job)
|
|
|
|
def claim_next_job(
|
|
self, *, worker_id: str = "worker", organization_id: Optional[str] = None
|
|
) -> Optional[Job]:
|
|
"""Claim the next queued job for a worker.
|
|
|
|
For the given optional organization scope, atomically flip one queued
|
|
job to ``running`` and return it; returns None when nothing is claimable.
|
|
"""
|
|
query = self.session.query(Job).filter(Job.status == JobStatus.QUEUED.value)
|
|
if organization_id is not None:
|
|
query = query.filter(Job.organization_id == organization_id)
|
|
job = query.order_by(Job.created_at.asc()).first()
|
|
if job is None:
|
|
return None
|
|
job.status = JobStatus.RUNNING.value
|
|
job.message = f"claimed by {worker_id}"
|
|
self.session.flush()
|
|
return job
|
|
|
|
def complete_job(self, job_id: str, result=None, message: str = "") -> None:
|
|
job = self.session.get(Job, job_id)
|
|
if job is None:
|
|
raise ValueError("job_not_found")
|
|
job.status = JobStatus.SUCCEEDED.value
|
|
job.result = result
|
|
job.message = message or job.message
|
|
job.finished_at = _utc_now()
|
|
self.session.flush()
|
|
|
|
def fail_job(self, job_id: str, error_code: str, message: str = "") -> None:
|
|
job = self.session.get(Job, job_id)
|
|
if job is None:
|
|
raise ValueError("job_not_found")
|
|
job.status = JobStatus.FAILED.value
|
|
job.error_code = error_code
|
|
job.message = message or job.message
|
|
job.finished_at = _utc_now()
|
|
self.session.flush()
|