feat: point company search and contact match at real ClickHouse data

/companies/search queried a nonexistent 'companies' table; now queries
stg_apollo_organizations (5.26M organizations, typed/normalized layer).
/contacts/match queried a nonexistent 'contacts' table; now queries
apollo_persons_raw with LIKE matching, since emails/phone_numbers are
still raw list-repr text there -- exact matching needs a stg_apollo_persons
layer that doesn't exist yet.
This commit is contained in:
valentinbvro 2026-07-21 19:27:06 +02:00
parent 244c2ef14a
commit 9956b120d0
2 changed files with 69 additions and 7 deletions

View file

@ -5,12 +5,47 @@ from app.schemas import CompanySearchQuery
router = APIRouter(prefix="/companies", tags=["companies"])
SEARCH_COLUMNS = [
"organization_id",
"organization_name",
"normalized_domain",
"hq_city",
"hq_country",
"industries",
"num_current_employees",
"revenue_in_thousands",
]
@router.post("/search")
def search_companies(payload: CompanySearchQuery):
client = get_client()
result = client.query(
"SELECT name, domain, country FROM companies WHERE name ILIKE {q:String} LIMIT {limit:UInt32}",
f"""
SELECT {", ".join(SEARCH_COLUMNS)}
FROM stg_apollo_organizations
WHERE organization_name ILIKE {{q:String}} OR normalized_domain ILIKE {{q:String}}
ORDER BY organization_name
LIMIT {{limit:UInt32}}
""",
parameters={"q": f"%{payload.query}%", "limit": payload.limit},
)
return {"results": result.result_rows}
companies = [dict(zip(SEARCH_COLUMNS, row)) for row in result.result_rows]
return {"results": companies, "count": len(companies)}
@router.get("/{organization_id}")
def get_company(organization_id: str):
client = get_client()
result = client.query(
f"""
SELECT {", ".join(SEARCH_COLUMNS)}
FROM stg_apollo_organizations
WHERE organization_id = {{id:String}}
LIMIT 1
""",
parameters={"id": organization_id},
)
if not result.result_rows:
return {"error": "not_found"}
return dict(zip(SEARCH_COLUMNS, result.result_rows[0]))

View file

@ -6,18 +6,45 @@ from app.schemas import ContactMatchQuery
router = APIRouter(prefix="/contacts", tags=["contacts"])
MATCH_COLUMNS = [
"id",
"full_name",
"job_title",
"job_company_name",
"emails",
"work_email",
"phone_numbers",
]
@router.post("/match")
def match_contact(payload: ContactMatchQuery):
"""Placeholder pentru entity resolution: cauta dupa email/telefon normalizat,
apoi scoreaza candidatii cu rapidfuzz pentru potriviri apropiate."""
"""Entity resolution pe raw layer: apollo_persons_raw pastreaza emails/phone_numbers
ca text needormalizat (list-repr din sursa), deci se cauta cu LIKE, nu match exact.
O potrivire exacta corecta va veni odata cu stratul stg_apollo_persons (nefacut inca)."""
client = get_client()
if payload.email:
result = client.query(
"SELECT * FROM contacts WHERE email = {email:String} LIMIT 5",
parameters={"email": payload.email},
f"""
SELECT {", ".join(MATCH_COLUMNS)}
FROM apollo_persons_raw
WHERE emails ILIKE {{email:String}} OR work_email ILIKE {{email:String}}
LIMIT 5
""",
parameters={"email": f"%{payload.email}%"},
)
return {"matches": result.result_rows}
return {"matches": [dict(zip(MATCH_COLUMNS, row)) for row in result.result_rows]}
if payload.phone:
result = client.query(
f"""
SELECT {", ".join(MATCH_COLUMNS)}
FROM apollo_persons_raw
WHERE phone_numbers ILIKE {{phone:String}}
LIMIT 5
""",
parameters={"phone": f"%{payload.phone}%"},
)
return {"matches": [dict(zip(MATCH_COLUMNS, row)) for row in result.result_rows]}
return {"matches": []}