"""DMS File search provider.""" from __future__ import annotations import logging import uuid from typing import Any from sqlalchemy import text from sqlalchemy.ext.asyncio import AsyncSession from app.config import settings from app.plugins.builtins.unified_search.base_provider import BaseSearchProvider logger = logging.getLogger(__name__) class FileSearchProvider(BaseSearchProvider): """Search provider for DMS File entities.""" entity_type = "file" supports_rag: bool = True async def _search_fts_filtered( self, db: AsyncSession, tsquery: str, tenant_id: uuid.UUID, limit: int, visible_ids: set[uuid.UUID] | None, ) -> list[dict[str, Any]]: """Full-text search on files.content_tsv, filtered by visible_ids. If visible_ids is None, no visibility filter is applied (system admin). """ if visible_ids is not None: sql = text( """ SELECT f.*, ts_rank(f.content_tsv, to_tsquery('pg_catalog.german', :q)) AS rank FROM files f WHERE f.tenant_id = :tid AND f.deleted_at IS NULL AND f.content_tsv @@ to_tsquery('pg_catalog.german', :q) AND f.id = ANY(:visible_ids) ORDER BY rank DESC LIMIT :lim """ ) result = await db.execute( sql, { "q": tsquery, "tid": tenant_id, "lim": limit, "visible_ids": list(visible_ids), }, ) else: sql = text( """ SELECT f.*, ts_rank(f.content_tsv, to_tsquery('pg_catalog.german', :q)) AS rank FROM files f WHERE f.tenant_id = :tid AND f.deleted_at IS NULL AND f.content_tsv @@ to_tsquery('pg_catalog.german', :q) ORDER BY rank DESC LIMIT :lim """ ) result = await db.execute( sql, {"q": tsquery, "tid": tenant_id, "lim": limit}, ) rows = result.mappings().all() return [dict(r) for r in rows] async def _search_vector_filtered( self, db: AsyncSession, embedding: list[float], tenant_id: uuid.UUID, limit: int, visible_ids: set[uuid.UUID] | None, ) -> list[dict[str, Any]]: """Semantic search on files.embedding, filtered by visible_ids. If visible_ids is None, no visibility filter is applied (system admin). """ if visible_ids is not None: sql = text( """ SELECT f.*, 1 - (f.embedding <=> cast(:emb AS vector)) AS score FROM files f WHERE f.tenant_id = :tid AND f.deleted_at IS NULL AND f.embedding IS NOT NULL AND f.id = ANY(:visible_ids) ORDER BY f.embedding <=> cast(:emb AS vector) LIMIT :lim """ ) result = await db.execute( sql, { "emb": str(embedding), "tid": tenant_id, "lim": limit, "visible_ids": list(visible_ids), }, ) else: sql = text( """ SELECT f.*, 1 - (f.embedding <=> cast(:emb AS vector)) AS score FROM files f WHERE f.tenant_id = :tid AND f.deleted_at IS NULL AND f.embedding IS NOT NULL ORDER BY f.embedding <=> cast(:emb AS vector) LIMIT :lim """ ) result = await db.execute( sql, {"emb": str(embedding), "tid": tenant_id, "lim": limit}, ) rows = result.mappings().all() return [dict(r) for r in rows] async def get_embedding_text( self, db: AsyncSession, entity_id: uuid.UUID, tenant_id: uuid.UUID ) -> str: """Get text for embedding generation.""" sql = text( """ SELECT name, content_text FROM files WHERE id = :eid AND tenant_id = :tid """ ) result = await db.execute(sql, {"eid": entity_id, "tid": tenant_id}) row = result.mappings().first() if not row: return "" name = row.get("name", "") or "" content = row.get("content_text", "") or "" return f"{name} {content[:5000]}" async def search_rag( self, db: AsyncSession, query_embedding: list[float], tenant_id: uuid.UUID, limit: int, user_id: uuid.UUID | None = None, is_system_admin: bool = False, ) -> list[dict[str, Any]]: """RAG search: find relevant document chunks via vector similarity. Queries the document_chunks table using cosine distance on chunk embeddings, joins to files to exclude soft-deleted files, and applies permission filtering using the over-fetch strategy (same as search_vector). """ await db.execute(text(f"SET LOCAL hnsw.ef_search = {settings.hnsw_ef_search}")) if is_system_admin or not user_id: return await self._search_rag_filtered(db, query_embedding, tenant_id, limit, None) visible_ids = await self._get_visible_ids(db, tenant_id, user_id) if not visible_ids: return [] over_fetch_limit = limit * 3 results = await self._search_rag_filtered(db, query_embedding, tenant_id, over_fetch_limit, None) filtered = [r for r in results if r.get("file_id") in visible_ids] return filtered[:limit] async def _search_rag_filtered( self, db: AsyncSession, embedding: list[float], tenant_id: uuid.UUID, limit: int, visible_ids: set[uuid.UUID] | None, ) -> list[dict[str, Any]]: """RAG vector search on document_chunks.embedding, filtered by visible_ids.""" if visible_ids is not None: sql = text( """ SELECT dc.chunk_text, dc.file_id, dc.chunk_index, 1 - (dc.embedding <=> cast(:emb AS vector)) AS score FROM document_chunks dc JOIN files f ON dc.file_id = f.id WHERE dc.tenant_id = :tid AND dc.deleted_at IS NULL AND dc.embedding IS NOT NULL AND f.deleted_at IS NULL AND f.id = ANY(:visible_ids) ORDER BY dc.embedding <=> cast(:emb AS vector) LIMIT :lim """ ) result = await db.execute( sql, { "emb": str(embedding), "tid": tenant_id, "lim": limit, "visible_ids": list(visible_ids), }, ) else: sql = text( """ SELECT dc.chunk_text, dc.file_id, dc.chunk_index, 1 - (dc.embedding <=> cast(:emb AS vector)) AS score FROM document_chunks dc JOIN files f ON dc.file_id = f.id WHERE dc.tenant_id = :tid AND dc.deleted_at IS NULL AND dc.embedding IS NOT NULL AND f.deleted_at IS NULL ORDER BY dc.embedding <=> cast(:emb AS vector) LIMIT :lim """ ) result = await db.execute( sql, {"emb": str(embedding), "tid": tenant_id, "lim": limit}, ) rows = result.mappings().all() return [ { "id": str(r["file_id"]), "entity_type": "file", "entity_id": str(r["file_id"]), "snippet": r["chunk_text"][:200], "score": float(r.get("score", 0.0)), "data": {"chunk_index": r.get("chunk_index")}, } for r in rows ] def to_search_result(self, entity: object) -> dict[str, Any]: """Convert file to search result dict.""" if isinstance(entity, dict): name = entity.get("name", "") content = entity.get("content_text", "") or "" entity_id = str(entity.get("id", "")) else: name = getattr(entity, "name", "") content = getattr(entity, "content_text", "") or "" entity_id = str(getattr(entity, "id", "")) return { "entity_type": self.entity_type, "entity_id": entity_id, "title": name, "snippet": content[:200], "score": 0.0, "data": {}, }