intelligence-api/app/routers/contacts.py
valentinbvro 9956b120d0 feat: point company search and contact match at real ClickHouse data
/companies/search queried a nonexistent 'companies' table; now queries
stg_apollo_organizations (5.26M organizations, typed/normalized layer).
/contacts/match queried a nonexistent 'contacts' table; now queries
apollo_persons_raw with LIKE matching, since emails/phone_numbers are
still raw list-repr text there -- exact matching needs a stg_apollo_persons
layer that doesn't exist yet.
2026-07-21 19:27:06 +02:00

52 lines
1.6 KiB
Python

from fastapi import APIRouter
from rapidfuzz import fuzz
from app.clickhouse import get_client
from app.schemas import ContactMatchQuery
router = APIRouter(prefix="/contacts", tags=["contacts"])
MATCH_COLUMNS = [
"id",
"full_name",
"job_title",
"job_company_name",
"emails",
"work_email",
"phone_numbers",
]
@router.post("/match")
def match_contact(payload: ContactMatchQuery):
"""Entity resolution pe raw layer: apollo_persons_raw pastreaza emails/phone_numbers
ca text needormalizat (list-repr din sursa), deci se cauta cu LIKE, nu match exact.
O potrivire exacta corecta va veni odata cu stratul stg_apollo_persons (nefacut inca)."""
client = get_client()
if payload.email:
result = client.query(
f"""
SELECT {", ".join(MATCH_COLUMNS)}
FROM apollo_persons_raw
WHERE emails ILIKE {{email:String}} OR work_email ILIKE {{email:String}}
LIMIT 5
""",
parameters={"email": f"%{payload.email}%"},
)
return {"matches": [dict(zip(MATCH_COLUMNS, row)) for row in result.result_rows]}
if payload.phone:
result = client.query(
f"""
SELECT {", ".join(MATCH_COLUMNS)}
FROM apollo_persons_raw
WHERE phone_numbers ILIKE {{phone:String}}
LIMIT 5
""",
parameters={"phone": f"%{payload.phone}%"},
)
return {"matches": [dict(zip(MATCH_COLUMNS, row)) for row in result.result_rows]}
return {"matches": []}
def score_name_similarity(a: str, b: str) -> float:
return fuzz.token_sort_ratio(a, b)