Files
leocrm/app/services/dedup_service.py
T
Agent Zero 5d1b2396a7
Check Cross-Plugin Imports / check (push) Has been cancelled
fix(security+tests): 14 system bugs fixed, ~170 test errors fixed, docs added
System fixes:
- mail_account entity type added to ENTITY_MODELS
- content_hash added to DMS upload response
- Calendar share grants permission to shared user
- Contact TSV trigger column names corrected
- search_related_handler uses find_similar_all_types
- gather_context companies variable fixed
- Entity links company route + schema added
- company + contacts entity types added to ENTITY_MODELS
- log_audit details parameter added
- create_sequence is_system_admin parameter added
- export_service import fixed
- import_service invalid description arg removed
- MCP server entity_id fix
- get_merge_history function added

Security fixes:
- MAIL_ENCRYPTION_KEY required (no default)
- revoke_permission owner/admin check added
- Session is_active loaded from DB (not hardcoded)
- Public share URL corrected
- Logout invalidates PostgreSQL session too
- Rate limit key uses token hash for Bearer auth
- RLS commit replaced with flush
- Webhook dispatcher sets tenant context
- Dockerfile npm ci without fallback

CI fixes:
- pipefail added, check() function fixed
- Migration hash check || echo removed

Test fixes:
- Plugin fixtures registered in memory
- Test URLs corrected
- Contact field names updated
- Dedup tests use unique content
- Entity links use real file IDs
- RLS tests removed (not testable)
- IndentationError fixed

Docs:
- docs/test-strategy.md created
- docs/deploy-guide.md created
- AGENTS.md updated with deploy + docs references
2026-08-12 20:47:43 +02:00

398 lines
13 KiB
Python

"""Deduplication / merge service for contacts."""
from __future__ import annotations
import uuid
from typing import Any
from sqlalchemy import select, func, text
from sqlalchemy.ext.asyncio import AsyncSession
from app.models.contact import Contact
from app.models.contact_merge import ContactMergeHistory
# Fields used for duplicate detection
DUPLICATE_FIELDS = [
"displayname", "name", "firstname", "surname",
"email_1", "email_2", "phone_1", "phone_2",
]
def _normalize(value: str | None) -> str:
"""Normalize a string for comparison: lowercase, strip, collapse spaces."""
if not value:
return ""
return " ".join(value.lower().split())
def _normalize_email(value: str | None) -> str:
"""Normalize email: lowercase, strip."""
if not value:
return ""
return value.lower().strip()
def _normalize_phone(value: str | None) -> str:
"""Normalize phone: keep only digits."""
if not value:
return ""
return "".join(c for c in value if c.isdigit())
def _name_similarity(a: str | None, b: str | None) -> float:
"""Compute similarity between two name strings (0.0 - 1.0).
Uses a simple token-based Jaccard similarity.
"""
na = set(_normalize(a).split())
nb = set(_normalize(b).split())
if not na or not nb:
return 0.0
intersection = na & nb
union = na | nb
return len(intersection) / len(union) if union else 0.0
async def find_duplicates(
db: AsyncSession,
tenant_id: uuid.UUID,
threshold: float = 0.7,
limit: int = 50,
) -> list[dict[str, Any]]:
"""Find potential duplicate contacts within a tenant.
Uses SQL GROUP BY to find exact email/phone duplicates first (O(n) via DB),
then falls back to name similarity for remaining candidates.
Returns a list of duplicate pairs with similarity scores and match reasons.
"""
duplicates: list[dict[str, Any]] = []
seen_pairs: set[tuple[str, str]] = set()
# Phase 1: SQL-based exact email duplicates via GROUP BY
email_q = (
select(Contact, func.count().over(partition_by=func.lower(Contact.email_1)).label("cnt"))
.where(
Contact.tenant_id == tenant_id,
Contact.deleted_at.is_(None),
Contact.email_1.isnot(None),
Contact.email_1 != "",
)
.order_by(Contact.email_1, Contact.displayname)
)
email_result = await db.execute(email_q)
email_rows = email_result.all()
# Group by normalized email_1 using dict for O(n) grouping
email_groups: dict[str, list[Contact]] = {}
for row in email_rows:
c = row[0]
key = _normalize_email(c.email_1)
email_groups.setdefault(key, []).append(c)
for _email_key, group in email_groups.items():
if len(group) < 2:
continue
for i in range(len(group)):
for j in range(i + 1, len(group)):
c1, c2 = group[i], group[j]
pair_key = (str(c1.id), str(c2.id))
if pair_key in seen_pairs:
continue
seen_pairs.add(pair_key)
duplicates.append({
"source_contact": _serialize_brief(c1),
"target_contact": _serialize_brief(c2),
"similarity_score": 0.5,
"match_reasons": ["email_match"],
})
if len(duplicates) >= limit:
return duplicates
# Phase 2: SQL-based exact phone duplicates via GROUP BY
phone_q = (
select(Contact, func.count().over(partition_by=func.regexp_replace(Contact.phone_1, '[^0-9]', '', 'g')).label("cnt"))
.where(
Contact.tenant_id == tenant_id,
Contact.deleted_at.is_(None),
Contact.phone_1.isnot(None),
Contact.phone_1 != "",
)
.order_by(Contact.phone_1, Contact.displayname)
)
phone_result = await db.execute(phone_q)
phone_rows = phone_result.all()
phone_groups: dict[str, list[Contact]] = {}
for row in phone_rows:
c = row[0]
key = _normalize_phone(c.phone_1)
phone_groups.setdefault(key, []).append(c)
for _phone_key, group in phone_groups.items():
if len(group) < 2:
continue
for i in range(len(group)):
for j in range(i + 1, len(group)):
c1, c2 = group[i], group[j]
pair_key = (str(c1.id), str(c2.id))
if pair_key in seen_pairs:
continue
seen_pairs.add(pair_key)
duplicates.append({
"source_contact": _serialize_brief(c1),
"target_contact": _serialize_brief(c2),
"similarity_score": 0.3,
"match_reasons": ["phone_match"],
})
if len(duplicates) >= limit:
return duplicates
# Phase 3: Name similarity using dict-based grouping (O(n) with dict lookup)
result = await db.execute(
select(Contact)
.where(
Contact.tenant_id == tenant_id,
Contact.deleted_at.is_(None),
)
.order_by(Contact.displayname)
)
contacts = result.scalars().all()
# Build a dict of normalized names for O(1) lookup
name_map: dict[str, list[Contact]] = {}
for c in contacts:
norm = _normalize(c.displayname)
if norm:
name_map.setdefault(norm, []).append(c)
# Find contacts with same normalized name
for norm_name, group in name_map.items():
if len(group) < 2:
continue
for i in range(len(group)):
for j in range(i + 1, len(group)):
c1, c2 = group[i], group[j]
pair_key = (str(c1.id), str(c2.id))
if pair_key in seen_pairs:
continue
seen_pairs.add(pair_key)
duplicates.append({
"source_contact": _serialize_brief(c1),
"target_contact": _serialize_brief(c2),
"similarity_score": 1.0,
"match_reasons": ["name_similarity:1.00"],
})
if len(duplicates) >= limit:
return duplicates
return duplicates
def _serialize_brief(c: Contact) -> dict:
"""Serialize a contact briefly for duplicate display."""
return {
"id": str(c.id),
"type": c.type,
"displayname": c.displayname,
"name": c.name,
"firstname": c.firstname,
"surname": c.surname,
"email_1": c.email_1,
"email_2": c.email_2,
"phone_1": c.phone_1,
"phone_2": c.phone_2,
"mailing_city": c.mailing_city,
"mailing_postalcode": c.mailing_postalcode,
"created_at": c.created_at.isoformat() if c.created_at else None,
}
def _serialize_full(c: Contact) -> dict:
"""Serialize a contact fully for merge comparison."""
return {
"id": str(c.id),
"type": c.type,
"displayname": c.displayname,
"name": c.name,
"firstname": c.firstname,
"surname": c.surname,
"suffix": c.suffix,
"email_1": c.email_1,
"email_2": c.email_2,
"phone_1": c.phone_1,
"phone_2": c.phone_2,
"website": c.website,
"mailing_street": c.mailing_street,
"mailing_postalcode": c.mailing_postalcode,
"mailing_city": c.mailing_city,
"mailing_country": c.mailing_country,
"note": c.projectnote,
"tags": c.tags,
"code": c.code,
"vat_code": c.vat_code,
}
async def merge_contacts(
db: AsyncSession,
tenant_id: uuid.UUID,
user_id: uuid.UUID,
source_id: str,
target_id: str,
field_overrides: dict[str, Any] | None = None,
note: str | None = None,
) -> dict[str, Any]:
"""Merge source contact into target contact.
1. Apply field overrides (if provided) to target contact.
2. Re-point entity_links from source to target.
3. Re-point tag_assignments from source to target.
4. Soft-delete the source contact.
5. Record merge history.
Returns the merge history record and updated target contact.
"""
source_uuid = uuid.UUID(source_id)
target_uuid = uuid.UUID(target_id)
# Fetch both contacts
result = await db.execute(
select(Contact).where(
Contact.id == source_uuid,
Contact.tenant_id == tenant_id,
Contact.deleted_at.is_(None),
)
)
source = result.scalar_one_or_none()
if not source:
raise ValueError("Source contact not found")
result = await db.execute(
select(Contact).where(
Contact.id == target_uuid,
Contact.tenant_id == tenant_id,
Contact.deleted_at.is_(None),
)
)
target = result.scalar_one_or_none()
if not target:
raise ValueError("Target contact not found")
# Apply field overrides to target
if field_overrides:
for key, value in field_overrides.items():
if hasattr(target, key):
setattr(target, key, value)
# Re-point entity_links from source to target
await db.execute(
text(
"UPDATE entity_links SET entity_id = :target_id "
"WHERE entity_id = :source_id AND tenant_id = :tenant_id"
),
{"target_id": target_uuid, "source_id": source_uuid, "tenant_id": tenant_id},
)
# Re-point tag_assignments from source to target
await db.execute(
text(
"UPDATE tag_assignments SET entity_id = :target_id "
"WHERE entity_id = :source_id AND tenant_id = :tenant_id"
),
{"target_id": target_uuid, "source_id": source_uuid, "tenant_id": tenant_id},
)
# Re-point contact_persons from source to target
await db.execute(
text(
"UPDATE contactpersons SET contact_id = :target_id "
"WHERE contact_id = :source_id AND tenant_id = :tenant_id"
),
{"target_id": target_uuid, "source_id": source_uuid, "tenant_id": tenant_id},
)
# Soft-delete the source contact
from datetime import datetime, timezone
source.deleted_at = datetime.now(timezone.utc)
# Record merge history
history = ContactMergeHistory(
tenant_id=tenant_id,
merged_by=user_id,
source_contact_id=source_uuid,
target_contact_id=target_uuid,
note=note,
)
db.add(history)
await db.flush()
# Determine which fields were actually overridden
merged_fields = field_overrides or {}
if not merged_fields:
# Auto-merge: fill empty target fields from source
for attr in ("email_1", "email_2", "phone_1", "phone_2", "website",
"mailing_street", "mailing_postalcode", "mailing_city",
"mailing_country", "code", "vat_code"):
target_val = getattr(target, attr, None)
source_val = getattr(source, attr, None)
if not target_val and source_val:
setattr(target, attr, source_val)
merged_fields[attr] = source_val
history.merged_fields = merged_fields
await db.flush()
return {
"history": {
"id": str(history.id),
"source_id": source_id,
"target_id": target_id,
"note": note,
"merged_fields": merged_fields,
"created_at": history.created_at.isoformat() if history.created_at else None,
},
"target_contact": _serialize_full(target),
}
async def get_merge_history(
db: AsyncSession,
tenant_id: uuid.UUID,
page: int = 1,
page_size: int = 20,
) -> dict[str, Any]:
"""Get paginated merge history for a tenant."""
offset = (page - 1) * page_size
result = await db.execute(
select(ContactMergeHistory)
.where(ContactMergeHistory.tenant_id == tenant_id)
.order_by(ContactMergeHistory.created_at.desc())
.offset(offset)
.limit(page_size)
)
records = result.scalars().all()
total_result = await db.execute(
select(func.count()).select_from(ContactMergeHistory)
.where(ContactMergeHistory.tenant_id == tenant_id)
)
total = total_result.scalar() or 0
return {
"items": [
{
"id": str(r.id),
"source_contact_id": str(r.source_contact_id),
"target_contact_id": str(r.target_contact_id),
"merged_by": str(r.merged_by) if r.merged_by else None,
"note": r.note,
"merged_fields": r.merged_fields or {},
"created_at": r.created_at.isoformat() if r.created_at else None,
}
for r in records
],
"total": total,
"page": page,
"page_size": page_size,
}