abbe7a18fc
- P0: hooks.py 3-tuple fix, trigger_dispatcher Contract, contacts/plugin unregister_actions_by_owner - P0: 5 test files — check_permission mocks removed, hardcoded DB credential → env var - P1: attachment_service DmsFile via Contract helper, restore_registry/history_hooks dedup - P1: mail/plugin restore unregister, mcp_client datetime.now(UTC), saved_views/filters patterns - P1: ProtectedRoute fail-closed, 13 test assertion fixes (bcrypt, DB-URLs, SECRET_KEYs) - P2: deprecated notifications → post_system_message (3 files), forgejo Base, report_generator lazy import - P2: webhooks permissions, deps.py/roles.py plugin perms removed, import_export default - P2: address/tags/entity_links patterns removed, worker.py Contract-Umgehungen fixed - P2: 28 frontend TODOs (hardcoded constants, deprecated notification API) - P3: dead code, duplicates, deprecated imports, private attr, __import__ inline - P3: 8 frontend TODOs (LucideIcons, inline styles, XSS, i18n) - ruff: 838 → 0 (612 auto-fix + 246 manual + 27 F821 regression fix) - F821: 30 → 0 (AutomationDefinition, DmsFile, user_id, Path, Any, String) - Contract-Umgehungen: 2 neue gefunden (worker.py:169, worker.py:280) und gefixt
262 lines
8.9 KiB
Python
262 lines
8.9 KiB
Python
"""DMS File search provider."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import logging
|
|
import uuid
|
|
from typing import Any
|
|
|
|
from sqlalchemy import text
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
|
|
|
from app.config import settings
|
|
from app.plugins.builtins.unified_search.base_provider import BaseSearchProvider
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class FileSearchProvider(BaseSearchProvider):
|
|
"""Search provider for DMS File entities."""
|
|
|
|
entity_type = "file"
|
|
supports_rag: bool = True
|
|
|
|
async def _search_fts_filtered(
|
|
self,
|
|
db: AsyncSession,
|
|
tsquery: str,
|
|
tenant_id: uuid.UUID,
|
|
limit: int,
|
|
visible_ids: set[uuid.UUID] | None,
|
|
) -> list[dict[str, Any]]:
|
|
"""Full-text search on files.content_tsv, filtered by visible_ids.
|
|
|
|
If visible_ids is None, no visibility filter is applied (system admin).
|
|
"""
|
|
if visible_ids is not None:
|
|
sql = text(
|
|
"""
|
|
SELECT f.*, ts_rank(f.content_tsv, to_tsquery('pg_catalog.german', :q)) AS rank
|
|
FROM files f
|
|
WHERE f.tenant_id = :tid
|
|
AND f.deleted_at IS NULL
|
|
AND f.content_tsv @@ to_tsquery('pg_catalog.german', :q)
|
|
AND f.id = ANY(:visible_ids)
|
|
ORDER BY rank DESC
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{
|
|
"q": tsquery,
|
|
"tid": tenant_id,
|
|
"lim": limit,
|
|
"visible_ids": list(visible_ids),
|
|
},
|
|
)
|
|
else:
|
|
sql = text(
|
|
"""
|
|
SELECT f.*, ts_rank(f.content_tsv, to_tsquery('pg_catalog.german', :q)) AS rank
|
|
FROM files f
|
|
WHERE f.tenant_id = :tid
|
|
AND f.deleted_at IS NULL
|
|
AND f.content_tsv @@ to_tsquery('pg_catalog.german', :q)
|
|
ORDER BY rank DESC
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{"q": tsquery, "tid": tenant_id, "lim": limit},
|
|
)
|
|
rows = result.mappings().all()
|
|
return [dict(r) for r in rows]
|
|
|
|
async def _search_vector_filtered(
|
|
self,
|
|
db: AsyncSession,
|
|
embedding: list[float],
|
|
tenant_id: uuid.UUID,
|
|
limit: int,
|
|
visible_ids: set[uuid.UUID] | None,
|
|
) -> list[dict[str, Any]]:
|
|
"""Semantic search on files.embedding, filtered by visible_ids.
|
|
|
|
If visible_ids is None, no visibility filter is applied (system admin).
|
|
"""
|
|
if visible_ids is not None:
|
|
sql = text(
|
|
"""
|
|
SELECT f.*, 1 - (f.embedding <=> cast(:emb AS vector)) AS score
|
|
FROM files f
|
|
WHERE f.tenant_id = :tid
|
|
AND f.deleted_at IS NULL
|
|
AND f.embedding IS NOT NULL
|
|
AND f.id = ANY(:visible_ids)
|
|
ORDER BY f.embedding <=> cast(:emb AS vector)
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{
|
|
"emb": str(embedding),
|
|
"tid": tenant_id,
|
|
"lim": limit,
|
|
"visible_ids": list(visible_ids),
|
|
},
|
|
)
|
|
else:
|
|
sql = text(
|
|
"""
|
|
SELECT f.*, 1 - (f.embedding <=> cast(:emb AS vector)) AS score
|
|
FROM files f
|
|
WHERE f.tenant_id = :tid
|
|
AND f.deleted_at IS NULL
|
|
AND f.embedding IS NOT NULL
|
|
ORDER BY f.embedding <=> cast(:emb AS vector)
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{"emb": str(embedding), "tid": tenant_id, "lim": limit},
|
|
)
|
|
rows = result.mappings().all()
|
|
return [dict(r) for r in rows]
|
|
|
|
async def get_embedding_text(
|
|
self, db: AsyncSession, entity_id: uuid.UUID, tenant_id: uuid.UUID
|
|
) -> str:
|
|
"""Get text for embedding generation."""
|
|
sql = text(
|
|
"""
|
|
SELECT name, content_text
|
|
FROM files
|
|
WHERE id = :eid AND tenant_id = :tid
|
|
"""
|
|
)
|
|
result = await db.execute(sql, {"eid": entity_id, "tid": tenant_id})
|
|
row = result.mappings().first()
|
|
if not row:
|
|
return ""
|
|
name = row.get("name", "") or ""
|
|
content = row.get("content_text", "") or ""
|
|
return f"{name} {content[:5000]}"
|
|
|
|
async def search_rag(
|
|
self,
|
|
db: AsyncSession,
|
|
query_embedding: list[float],
|
|
tenant_id: uuid.UUID,
|
|
limit: int,
|
|
user_id: uuid.UUID | None = None,
|
|
is_system_admin: bool = False,
|
|
) -> list[dict[str, Any]]:
|
|
"""RAG search: find relevant document chunks via vector similarity.
|
|
|
|
Queries the document_chunks table using cosine distance on chunk
|
|
embeddings, joins to files to exclude soft-deleted files, and applies
|
|
permission filtering using the over-fetch strategy (same as search_vector).
|
|
"""
|
|
await db.execute(text(f"SET LOCAL hnsw.ef_search = {settings.hnsw_ef_search}"))
|
|
|
|
if is_system_admin or not user_id:
|
|
return await self._search_rag_filtered(db, query_embedding, tenant_id, limit, None)
|
|
|
|
visible_ids = await self._get_visible_ids(db, tenant_id, user_id)
|
|
if not visible_ids:
|
|
return []
|
|
|
|
over_fetch_limit = limit * 3
|
|
results = await self._search_rag_filtered(db, query_embedding, tenant_id, over_fetch_limit, None)
|
|
filtered = [r for r in results if r.get("file_id") in visible_ids]
|
|
return filtered[:limit]
|
|
|
|
async def _search_rag_filtered(
|
|
self,
|
|
db: AsyncSession,
|
|
embedding: list[float],
|
|
tenant_id: uuid.UUID,
|
|
limit: int,
|
|
visible_ids: set[uuid.UUID] | None,
|
|
) -> list[dict[str, Any]]:
|
|
"""RAG vector search on document_chunks.embedding, filtered by visible_ids."""
|
|
if visible_ids is not None:
|
|
sql = text(
|
|
"""
|
|
SELECT dc.chunk_text, dc.file_id, dc.chunk_index,
|
|
1 - (dc.embedding <=> cast(:emb AS vector)) AS score
|
|
FROM document_chunks dc
|
|
JOIN files f ON dc.file_id = f.id
|
|
WHERE dc.tenant_id = :tid
|
|
AND dc.deleted_at IS NULL
|
|
AND dc.embedding IS NOT NULL
|
|
AND f.deleted_at IS NULL
|
|
AND f.id = ANY(:visible_ids)
|
|
ORDER BY dc.embedding <=> cast(:emb AS vector)
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{
|
|
"emb": str(embedding),
|
|
"tid": tenant_id,
|
|
"lim": limit,
|
|
"visible_ids": list(visible_ids),
|
|
},
|
|
)
|
|
else:
|
|
sql = text(
|
|
"""
|
|
SELECT dc.chunk_text, dc.file_id, dc.chunk_index,
|
|
1 - (dc.embedding <=> cast(:emb AS vector)) AS score
|
|
FROM document_chunks dc
|
|
JOIN files f ON dc.file_id = f.id
|
|
WHERE dc.tenant_id = :tid
|
|
AND dc.deleted_at IS NULL
|
|
AND dc.embedding IS NOT NULL
|
|
AND f.deleted_at IS NULL
|
|
ORDER BY dc.embedding <=> cast(:emb AS vector)
|
|
LIMIT :lim
|
|
"""
|
|
)
|
|
result = await db.execute(
|
|
sql,
|
|
{"emb": str(embedding), "tid": tenant_id, "lim": limit},
|
|
)
|
|
rows = result.mappings().all()
|
|
return [
|
|
{
|
|
"id": str(r["file_id"]),
|
|
"entity_type": "file",
|
|
"entity_id": str(r["file_id"]),
|
|
"snippet": r["chunk_text"][:200],
|
|
"score": float(r.get("score", 0.0)),
|
|
"data": {"chunk_index": r.get("chunk_index")},
|
|
}
|
|
for r in rows
|
|
]
|
|
|
|
def to_search_result(self, entity: object) -> dict[str, Any]:
|
|
"""Convert file to search result dict."""
|
|
if isinstance(entity, dict):
|
|
name = entity.get("name", "")
|
|
content = entity.get("content_text", "") or ""
|
|
entity_id = str(entity.get("id", ""))
|
|
else:
|
|
name = getattr(entity, "name", "")
|
|
content = getattr(entity, "content_text", "") or ""
|
|
entity_id = str(getattr(entity, "id", ""))
|
|
return {
|
|
"entity_type": self.entity_type,
|
|
"entity_id": entity_id,
|
|
"title": name,
|
|
"snippet": content[:200],
|
|
"score": 0.0,
|
|
"data": {},
|
|
}
|