From 9956b120d04f7fea0f237efdfbf4aaedb771bed3 Mon Sep 17 00:00:00 2001 From: valentinbvro Date: Tue, 21 Jul 2026 19:27:06 +0200 Subject: [PATCH] feat: point company search and contact match at real ClickHouse data /companies/search queried a nonexistent 'companies' table; now queries stg_apollo_organizations (5.26M organizations, typed/normalized layer). /contacts/match queried a nonexistent 'contacts' table; now queries apollo_persons_raw with LIKE matching, since emails/phone_numbers are still raw list-repr text there -- exact matching needs a stg_apollo_persons layer that doesn't exist yet. --- app/routers/companies.py | 39 +++++++++++++++++++++++++++++++++++++-- app/routers/contacts.py | 37 ++++++++++++++++++++++++++++++++----- 2 files changed, 69 insertions(+), 7 deletions(-) diff --git a/app/routers/companies.py b/app/routers/companies.py index 0881e0d..35ccf65 100644 --- a/app/routers/companies.py +++ b/app/routers/companies.py @@ -5,12 +5,47 @@ from app.schemas import CompanySearchQuery router = APIRouter(prefix="/companies", tags=["companies"]) +SEARCH_COLUMNS = [ + "organization_id", + "organization_name", + "normalized_domain", + "hq_city", + "hq_country", + "industries", + "num_current_employees", + "revenue_in_thousands", +] + @router.post("/search") def search_companies(payload: CompanySearchQuery): client = get_client() result = client.query( - "SELECT name, domain, country FROM companies WHERE name ILIKE {q:String} LIMIT {limit:UInt32}", + f""" + SELECT {", ".join(SEARCH_COLUMNS)} + FROM stg_apollo_organizations + WHERE organization_name ILIKE {{q:String}} OR normalized_domain ILIKE {{q:String}} + ORDER BY organization_name + LIMIT {{limit:UInt32}} + """, parameters={"q": f"%{payload.query}%", "limit": payload.limit}, ) - return {"results": result.result_rows} + companies = [dict(zip(SEARCH_COLUMNS, row)) for row in result.result_rows] + return {"results": companies, "count": len(companies)} + + +@router.get("/{organization_id}") +def get_company(organization_id: str): + client = get_client() + result = client.query( + f""" + SELECT {", ".join(SEARCH_COLUMNS)} + FROM stg_apollo_organizations + WHERE organization_id = {{id:String}} + LIMIT 1 + """, + parameters={"id": organization_id}, + ) + if not result.result_rows: + return {"error": "not_found"} + return dict(zip(SEARCH_COLUMNS, result.result_rows[0])) diff --git a/app/routers/contacts.py b/app/routers/contacts.py index c1bf793..fd0e7f9 100644 --- a/app/routers/contacts.py +++ b/app/routers/contacts.py @@ -6,18 +6,45 @@ from app.schemas import ContactMatchQuery router = APIRouter(prefix="/contacts", tags=["contacts"]) +MATCH_COLUMNS = [ + "id", + "full_name", + "job_title", + "job_company_name", + "emails", + "work_email", + "phone_numbers", +] + @router.post("/match") def match_contact(payload: ContactMatchQuery): - """Placeholder pentru entity resolution: cauta dupa email/telefon normalizat, - apoi scoreaza candidatii cu rapidfuzz pentru potriviri apropiate.""" + """Entity resolution pe raw layer: apollo_persons_raw pastreaza emails/phone_numbers + ca text needormalizat (list-repr din sursa), deci se cauta cu LIKE, nu match exact. + O potrivire exacta corecta va veni odata cu stratul stg_apollo_persons (nefacut inca).""" client = get_client() if payload.email: result = client.query( - "SELECT * FROM contacts WHERE email = {email:String} LIMIT 5", - parameters={"email": payload.email}, + f""" + SELECT {", ".join(MATCH_COLUMNS)} + FROM apollo_persons_raw + WHERE emails ILIKE {{email:String}} OR work_email ILIKE {{email:String}} + LIMIT 5 + """, + parameters={"email": f"%{payload.email}%"}, ) - return {"matches": result.result_rows} + return {"matches": [dict(zip(MATCH_COLUMNS, row)) for row in result.result_rows]} + if payload.phone: + result = client.query( + f""" + SELECT {", ".join(MATCH_COLUMNS)} + FROM apollo_persons_raw + WHERE phone_numbers ILIKE {{phone:String}} + LIMIT 5 + """, + parameters={"phone": f"%{payload.phone}%"}, + ) + return {"matches": [dict(zip(MATCH_COLUMNS, row)) for row in result.result_rows]} return {"matches": []}