115 lines
3.3 KiB
Python
115 lines
3.3 KiB
Python
|
|
"""Company search provider."""
|
||
|
|
|
||
|
|
from __future__ import annotations
|
||
|
|
|
||
|
|
import logging
|
||
|
|
import uuid
|
||
|
|
from typing import Any
|
||
|
|
|
||
|
|
from sqlalchemy import text
|
||
|
|
from sqlalchemy.ext.asyncio import AsyncSession
|
||
|
|
|
||
|
|
logger = logging.getLogger(__name__)
|
||
|
|
|
||
|
|
|
||
|
|
class CompanySearchProvider:
|
||
|
|
"""Search provider for Company entities."""
|
||
|
|
|
||
|
|
entity_type = "company"
|
||
|
|
|
||
|
|
async def search_fts(
|
||
|
|
self,
|
||
|
|
db: AsyncSession,
|
||
|
|
tsquery: str,
|
||
|
|
tenant_id: uuid.UUID,
|
||
|
|
limit: int,
|
||
|
|
) -> list[dict[str, Any]]:
|
||
|
|
"""Full-text search on companies.search_tsv."""
|
||
|
|
sql = text(
|
||
|
|
"""
|
||
|
|
SELECT c.*, ts_rank(c.search_tsv, to_tsquery('pg_catalog.german', :q)) AS rank
|
||
|
|
FROM companies c
|
||
|
|
WHERE c.tenant_id = :tid
|
||
|
|
AND c.deleted_at IS NULL
|
||
|
|
AND c.search_tsv @@ to_tsquery('pg_catalog.german', :q)
|
||
|
|
ORDER BY rank DESC
|
||
|
|
LIMIT :lim
|
||
|
|
"""
|
||
|
|
)
|
||
|
|
result = await db.execute(
|
||
|
|
sql,
|
||
|
|
{"q": tsquery, "tid": tenant_id, "lim": limit},
|
||
|
|
)
|
||
|
|
rows = result.mappings().all()
|
||
|
|
return [dict(r) for r in rows]
|
||
|
|
|
||
|
|
async def search_vector(
|
||
|
|
self,
|
||
|
|
db: AsyncSession,
|
||
|
|
embedding: list[float],
|
||
|
|
tenant_id: uuid.UUID,
|
||
|
|
limit: int,
|
||
|
|
) -> list[dict[str, Any]]:
|
||
|
|
"""Semantic search on companies.embedding."""
|
||
|
|
sql = text(
|
||
|
|
"""
|
||
|
|
SELECT c.*, 1 - (c.embedding <=> cast(:emb AS vector)) AS score
|
||
|
|
FROM companies c
|
||
|
|
WHERE c.tenant_id = :tid
|
||
|
|
AND c.deleted_at IS NULL
|
||
|
|
AND c.embedding IS NOT NULL
|
||
|
|
ORDER BY c.embedding <=> cast(:emb AS vector)
|
||
|
|
LIMIT :lim
|
||
|
|
"""
|
||
|
|
)
|
||
|
|
result = await db.execute(
|
||
|
|
sql,
|
||
|
|
{"emb": str(embedding), "tid": tenant_id, "lim": limit},
|
||
|
|
)
|
||
|
|
rows = result.mappings().all()
|
||
|
|
return [dict(r) for r in rows]
|
||
|
|
|
||
|
|
async def get_embedding_text(
|
||
|
|
self,
|
||
|
|
db: AsyncSession,
|
||
|
|
entity_id: uuid.UUID,
|
||
|
|
tenant_id: uuid.UUID,
|
||
|
|
) -> str:
|
||
|
|
"""Get text for embedding generation."""
|
||
|
|
sql = text(
|
||
|
|
"""
|
||
|
|
SELECT name, description, industry
|
||
|
|
FROM companies
|
||
|
|
WHERE id = :eid AND tenant_id = :tid
|
||
|
|
"""
|
||
|
|
)
|
||
|
|
result = await db.execute(sql, {"eid": entity_id, "tid": tenant_id})
|
||
|
|
row = result.mappings().first()
|
||
|
|
if not row:
|
||
|
|
return ""
|
||
|
|
parts = [
|
||
|
|
row.get("name", ""),
|
||
|
|
row.get("description", ""),
|
||
|
|
row.get("industry", ""),
|
||
|
|
]
|
||
|
|
return " ".join(str(p) for p in parts if p)
|
||
|
|
|
||
|
|
def to_search_result(self, entity: object) -> dict[str, Any]:
|
||
|
|
"""Convert company to search result dict."""
|
||
|
|
if isinstance(entity, dict):
|
||
|
|
name = entity.get("name", "")
|
||
|
|
description = entity.get("description", "") or ""
|
||
|
|
entity_id = str(entity.get("id", ""))
|
||
|
|
else:
|
||
|
|
name = getattr(entity, "name", "")
|
||
|
|
description = getattr(entity, "description", "") or ""
|
||
|
|
entity_id = str(getattr(entity, "id", ""))
|
||
|
|
return {
|
||
|
|
"entity_type": self.entity_type,
|
||
|
|
"entity_id": entity_id,
|
||
|
|
"title": name,
|
||
|
|
"snippet": description[:200],
|
||
|
|
"score": 0.0,
|
||
|
|
"data": {},
|
||
|
|
}
|