3d9b76cea4
Check Cross-Plugin Imports / check (push) Has been cancelled
- SPIKE-E: FTS+Vector+Permission benchmark on 10k records (all <30ms) - E-PROV: supports_fts/vector/rag/graph capability flags on all providers - E-FTS/VEC: All 11 providers refactored to BaseSearchProvider with permission filtering - E-PERM: Over-fetch strategy for vector+permission (15x faster than ANY() filter) - E-FUSE: rrf_fusion_multi() for N-way RRF over FTS+Vector+RAG+Graph - E-LLM: Query understanding cleaned up to use central llm_complete() - E-CHUNK: Document chunking module + document_chunks table with HNSW index - E-EMB: Chunk embedding ARQ jobs (index_file_chunks, reindex_chunks) - E-RAG: RAG retrieval via FileSearchProvider.search_rag() - E-GRAPH: GraphRAG BFS traversal via GraphRAGSearchProvider.search_graph() - E-IX-EVT: Auto-indexing via outbox events + delete/cleanup handlers - E-IX-RE: Batch reindex with progress tracking + reindex_all job - E-DATA-LIFE: Lifecycle module (remove/rebuild/restore/correct) + API endpoints - E-K-MEM: AgentMemorySearchProvider - E-P-AI: AIChatSearchProvider - E-P-WF: WorkflowSearchProvider - E-P-COMM: ConversationSearchProvider verified (already on BaseSearchProvider) - E-API: Filter params (date_from/to, tags, sort) + /facets endpoint - E-TOOL: unified_search AI tool registered in ToolRegistry - E-MCP: Search tool in MCP server with normal RBAC/tenant checks - E-UI-CMD: CommandPalette (Cmd+K) with debounced search + recent searches - E-UI-FAC: SearchFacets, SearchResultCard, SavedSearches components - E-TEST: 40 new tests in test_unified_search_phase_e.py (105 total green) - E-DOC: api-documentation.md, plugin-development-guide.md, test-strategy.md updated 105 tests passing, TypeScript clean.
77 lines
2.7 KiB
Python
77 lines
2.7 KiB
Python
"""SQLAlchemy models for the Unified Search plugin."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import uuid
|
|
from datetime import datetime
|
|
|
|
from sqlalchemy import DateTime, ForeignKey, Index, Integer, String, Text
|
|
from sqlalchemy.dialects.postgresql import JSONB, UUID as PGUUID
|
|
from sqlalchemy.orm import Mapped, mapped_column
|
|
|
|
from app.core.db import Base, TenantMixin
|
|
|
|
|
|
class SearchProviderRegistry(Base, TenantMixin):
|
|
"""Registry of active search providers per tenant."""
|
|
|
|
__tablename__ = "unified_search_providers"
|
|
__table_args__ = (
|
|
Index("ix_usp_tenant", "tenant_id"),
|
|
)
|
|
|
|
id: Mapped[uuid.UUID] = mapped_column(
|
|
PGUUID(as_uuid=True), primary_key=True, default=uuid.uuid4
|
|
)
|
|
entity_type: Mapped[str] = mapped_column(String(50), nullable=False)
|
|
plugin_name: Mapped[str] = mapped_column(String(80), nullable=False)
|
|
is_active: Mapped[bool] = mapped_column(
|
|
nullable=False, default=True
|
|
)
|
|
config: Mapped[dict] = mapped_column(JSONB, nullable=False, default=dict)
|
|
|
|
|
|
class SearchIndexLog(Base, TenantMixin):
|
|
"""Log of indexing actions for audit and debugging."""
|
|
|
|
__tablename__ = "unified_search_index_log"
|
|
__table_args__ = (
|
|
Index("ix_usil_tenant", "tenant_id"),
|
|
Index("ix_usil_entity", "entity_type", "entity_id"),
|
|
)
|
|
|
|
id: Mapped[uuid.UUID] = mapped_column(
|
|
PGUUID(as_uuid=True), primary_key=True, default=uuid.uuid4
|
|
)
|
|
entity_type: Mapped[str] = mapped_column(String(50), nullable=False)
|
|
entity_id: Mapped[uuid.UUID] = mapped_column(PGUUID(as_uuid=True), nullable=False)
|
|
action: Mapped[str] = mapped_column(String(20), nullable=False)
|
|
status: Mapped[str] = mapped_column(String(20), nullable=False, default="pending")
|
|
error_message: Mapped[str | None] = mapped_column(Text, nullable=True)
|
|
|
|
from pgvector.sqlalchemy import Vector
|
|
|
|
|
|
class DocumentChunk(Base, TenantMixin):
|
|
"""Chunk of a DMS file's extracted text, with its own embedding for RAG."""
|
|
|
|
__tablename__ = "document_chunks"
|
|
__table_args__ = (
|
|
Index("ix_document_chunks_tenant", "tenant_id"),
|
|
Index("ix_document_chunks_file", "file_id"),
|
|
Index("ix_document_chunks_tenant_file", "tenant_id", "file_id"),
|
|
)
|
|
|
|
id: Mapped[uuid.UUID] = mapped_column(
|
|
PGUUID(as_uuid=True), primary_key=True, default=uuid.uuid4
|
|
)
|
|
file_id: Mapped[uuid.UUID] = mapped_column(
|
|
PGUUID(as_uuid=True), ForeignKey("files.id", ondelete="CASCADE"), nullable=False
|
|
)
|
|
chunk_index: Mapped[int] = mapped_column(Integer, nullable=False)
|
|
chunk_text: Mapped[str] = mapped_column(Text, nullable=False)
|
|
chunk_hash: Mapped[str] = mapped_column(String(64), nullable=False)
|
|
embedding: Mapped[list[float] | None] = mapped_column(
|
|
Vector(768), nullable=True
|
|
)
|