3d9b76cea4
Check Cross-Plugin Imports / check (push) Has been cancelled
- SPIKE-E: FTS+Vector+Permission benchmark on 10k records (all <30ms) - E-PROV: supports_fts/vector/rag/graph capability flags on all providers - E-FTS/VEC: All 11 providers refactored to BaseSearchProvider with permission filtering - E-PERM: Over-fetch strategy for vector+permission (15x faster than ANY() filter) - E-FUSE: rrf_fusion_multi() for N-way RRF over FTS+Vector+RAG+Graph - E-LLM: Query understanding cleaned up to use central llm_complete() - E-CHUNK: Document chunking module + document_chunks table with HNSW index - E-EMB: Chunk embedding ARQ jobs (index_file_chunks, reindex_chunks) - E-RAG: RAG retrieval via FileSearchProvider.search_rag() - E-GRAPH: GraphRAG BFS traversal via GraphRAGSearchProvider.search_graph() - E-IX-EVT: Auto-indexing via outbox events + delete/cleanup handlers - E-IX-RE: Batch reindex with progress tracking + reindex_all job - E-DATA-LIFE: Lifecycle module (remove/rebuild/restore/correct) + API endpoints - E-K-MEM: AgentMemorySearchProvider - E-P-AI: AIChatSearchProvider - E-P-WF: WorkflowSearchProvider - E-P-COMM: ConversationSearchProvider verified (already on BaseSearchProvider) - E-API: Filter params (date_from/to, tags, sort) + /facets endpoint - E-TOOL: unified_search AI tool registered in ToolRegistry - E-MCP: Search tool in MCP server with normal RBAC/tenant checks - E-UI-CMD: CommandPalette (Cmd+K) with debounced search + recent searches - E-UI-FAC: SearchFacets, SearchResultCard, SavedSearches components - E-TEST: 40 new tests in test_unified_search_phase_e.py (105 total green) - E-DOC: api-documentation.md, plugin-development-guide.md, test-strategy.md updated 105 tests passing, TypeScript clean.
165 lines
5.3 KiB
Python
165 lines
5.3 KiB
Python
"""Mail search provider."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import uuid
|
|
from typing import Any
|
|
|
|
from sqlalchemy import text
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.plugins.builtins.unified_search.base_provider import BaseSearchProvider
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class MailSearchProvider(BaseSearchProvider):
|
|
"""Search provider for Mail entities."""
|
|
|
|
entity_type = "mail"
|
|
|
|
async def _search_fts_filtered(
|
|
self,
|
|
db: AsyncSession,
|
|
tsquery: str,
|
|
tenant_id: uuid.UUID,
|
|
limit: int,
|
|
visible_ids: set[uuid.UUID] | None,
|
|
) -> list[dict[str, Any]]:
|
|
"""Full-text search on mails.body_tsv, filtered by visible_ids.
|
|
|
|
If visible_ids is None, no visibility filter is applied (system admin).
|
|
"""
|
|
if visible_ids is not None:
|
|
sql = text(
|
|
"""
|
|
SELECT m.*, ts_rank(m.body_tsv, to_tsquery('pg_catalog.german', :q)) AS rank
|
|
FROM mails m
|
|
WHERE m.tenant_id = :tid
|
|
AND m.deleted_at IS NULL
|
|
AND m.body_tsv @@ to_tsquery('pg_catalog.german', :q)
|
|
AND m.id = ANY(:visible_ids)
|
|
ORDER BY rank DESC
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{
|
|
"q": tsquery,
|
|
"tid": tenant_id,
|
|
"lim": limit,
|
|
"visible_ids": list(visible_ids),
|
|
},
|
|
)
|
|
else:
|
|
sql = text(
|
|
"""
|
|
SELECT m.*, ts_rank(m.body_tsv, to_tsquery('pg_catalog.german', :q)) AS rank
|
|
FROM mails m
|
|
WHERE m.tenant_id = :tid
|
|
AND m.deleted_at IS NULL
|
|
AND m.body_tsv @@ to_tsquery('pg_catalog.german', :q)
|
|
ORDER BY rank DESC
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{"q": tsquery, "tid": tenant_id, "lim": limit},
|
|
)
|
|
rows = result.mappings().all()
|
|
return [dict(r) for r in rows]
|
|
|
|
async def _search_vector_filtered(
|
|
self,
|
|
db: AsyncSession,
|
|
embedding: list[float],
|
|
tenant_id: uuid.UUID,
|
|
limit: int,
|
|
visible_ids: set[uuid.UUID] | None,
|
|
) -> list[dict[str, Any]]:
|
|
"""Semantic search on mails.embedding, filtered by visible_ids.
|
|
|
|
If visible_ids is None, no visibility filter is applied (system admin).
|
|
"""
|
|
if visible_ids is not None:
|
|
sql = text(
|
|
"""
|
|
SELECT m.*, 1 - (m.embedding <=> cast(:emb AS vector)) AS score
|
|
FROM mails m
|
|
WHERE m.tenant_id = :tid
|
|
AND m.deleted_at IS NULL
|
|
AND m.embedding IS NOT NULL
|
|
AND m.id = ANY(:visible_ids)
|
|
ORDER BY m.embedding <=> cast(:emb AS vector)
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{
|
|
"emb": str(embedding),
|
|
"tid": tenant_id,
|
|
"lim": limit,
|
|
"visible_ids": list(visible_ids),
|
|
},
|
|
)
|
|
else:
|
|
sql = text(
|
|
"""
|
|
SELECT m.*, 1 - (m.embedding <=> cast(:emb AS vector)) AS score
|
|
FROM mails m
|
|
WHERE m.tenant_id = :tid
|
|
AND m.deleted_at IS NULL
|
|
AND m.embedding IS NOT NULL
|
|
ORDER BY m.embedding <=> cast(:emb AS vector)
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{"emb": str(embedding), "tid": tenant_id, "lim": limit},
|
|
)
|
|
rows = result.mappings().all()
|
|
return [dict(r) for r in rows]
|
|
|
|
async def get_embedding_text(
|
|
self, db: AsyncSession, entity_id: uuid.UUID, tenant_id: uuid.UUID
|
|
) -> str:
|
|
"""Get text for embedding generation."""
|
|
sql = text(
|
|
"""
|
|
SELECT subject, body_text
|
|
FROM mails
|
|
WHERE id = :eid AND tenant_id = :tid
|
|
"""
|
|
)
|
|
result = await db.execute(sql, {"eid": entity_id, "tid": tenant_id})
|
|
row = result.mappings().first()
|
|
if not row:
|
|
return ""
|
|
subject = row.get("subject", "") or ""
|
|
body = row.get("body_text", "") or ""
|
|
return f"{subject} {body[:5000]}"
|
|
|
|
def to_search_result(self, entity: object) -> dict[str, Any]:
|
|
"""Convert mail to search result dict."""
|
|
if isinstance(entity, dict):
|
|
subject = entity.get("subject", "")
|
|
body = entity.get("body_text", "") or ""
|
|
entity_id = str(entity.get("id", ""))
|
|
else:
|
|
subject = getattr(entity, "subject", "")
|
|
body = getattr(entity, "body_text", "") or ""
|
|
entity_id = str(getattr(entity, "id", ""))
|
|
return {
|
|
"entity_type": self.entity_type,
|
|
"entity_id": entity_id,
|
|
"title": subject,
|
|
"snippet": body[:200],
|
|
"score": 0.0,
|
|
"data": {},
|
|
}
|