"""Fiscal enrichment worker for ClientFlow opportunities. v4.9.25 adds a small, autonomous layer that runs before reconciliation: - read open opportunities without a fiscal customer; - use local data/cache first; - consult the external company/contact lookup API when enabled; - create an auditable suggestion; - auto-associate only very strong, non-conflicting matches. The service is intentionally conservative. It enriches the fiscal identity used by Jasmin/Odoo reconciliation; it does not create/close opportunities or alter commercial stages. """ from __future__ import annotations import hashlib import json import re import time import urllib.error import urllib.parse import urllib.request from datetime import datetime, timezone from typing import Any, Dict, Iterable, List, Optional, Tuple from sqlalchemy import text from app.config import settings from app.db import engine from app.commercial_service import ( get_customer_by_tax_id, link_customer_to_opportunity, normalize_fiscal_name, normalize_tax_id, upsert_customer, ) from app.opportunity_service import ensure_opportunity_schema, get_opportunity, list_opportunities from app.email_identity_extraction_service import extract_identity_for_opportunity, is_plausible_company_mention CACHE_SOURCE = "informa_pipeline_api" DEFAULT_AUTO_THRESHOLD = 95.0 PUBLIC_EMAIL_DOMAINS = { "gmail.com", "googlemail.com", "hotmail.com", "hotmail.pt", "outlook.com", "outlook.pt", "live.com", "msn.com", "icloud.com", "me.com", "mac.com", "yahoo.com", "yahoo.pt", "sapo.pt", "mail.com", "proton.me", "protonmail.com", "aol.com", "gmx.com", "gmx.net", "uol.com.br", } VERY_STRONG_MATCH_TYPES = { "nif_exato", "email_exato", "contacto_email_exato", "email_principal_exato", "email_exato_empresa_inferida", "email_exato_empresa_associada", "empresa_email_principal_exato", } STRONG_MATCH_TYPES = VERY_STRONG_MATCH_TYPES | { "nome_exato", "email_principal_dominio", "website_dominio", "contacto_email_dominio", } MEDIUM_MATCH_TYPES = STRONG_MATCH_TYPES | {"prefixo_nome", "parte_nome", "dominio", "email_dominio", "contacto"} DOMAIN_ONLY_MATCH_TYPES = {"email_principal_dominio", "email_dominio_empresa_associada", "contacto_email_dominio", "dominio", "email_dominio", "website_dominio"} _SCHEMA_READY = False def _clean(value: Any) -> str: return str(value or "").strip() def _json(value: Any) -> str: return json.dumps(value or {}, ensure_ascii=False, default=str) def _float(value: Any, default: float = 0.0) -> float: try: return float(value) except (TypeError, ValueError): return default def _now_iso() -> str: return datetime.now(timezone.utc).isoformat() def _sha256_payload(value: Any) -> str: raw = json.dumps(value or {}, ensure_ascii=False, sort_keys=True, default=str) return hashlib.sha256(raw.encode("utf-8")).hexdigest() def _domain_from_email(email: Any) -> str: email_value = _clean(email).lower() if "@" not in email_value: return "" return normalize_domain(email_value.rsplit("@", 1)[1]) def normalize_domain(value: Any) -> str: raw = _clean(value).lower() if not raw: return "" raw = raw.replace("https://", "").replace("http://", "") raw = raw.split("/", 1)[0].split("?", 1)[0].split("#", 1)[0] raw = raw.strip(". ") if raw.startswith("www."): raw = raw[4:] return raw def is_public_email_domain(domain: Any) -> bool: return normalize_domain(domain) in PUBLIC_EMAIL_DOMAINS def ensure_fiscal_enrichment_schema() -> None: """Create the cache/suggestion/audit tables used by the worker.""" global _SCHEMA_READY if _SCHEMA_READY: return ensure_opportunity_schema() with engine.begin() as conn: conn.execute(text(""" CREATE TABLE IF NOT EXISTS external_company_cache ( id UUID PRIMARY KEY DEFAULT gen_random_uuid(), source TEXT NOT NULL DEFAULT 'informa_pipeline_api', lookup_type TEXT NOT NULL, lookup_value TEXT NOT NULL, external_id TEXT, nif TEXT, legal_name TEXT, normalized_name TEXT, cae TEXT, address TEXT, postcode TEXT, city TEXT, district TEXT, country TEXT, website TEXT, phone TEXT, email TEXT, duns TEXT, status TEXT, score NUMERIC(5,2), match_type TEXT, raw_payload JSONB NOT NULL DEFAULT '{}'::jsonb, payload_hash TEXT, fetched_at TIMESTAMPTZ NOT NULL DEFAULT now(), expires_at TIMESTAMPTZ ) """)) for stmt in [ "ALTER TABLE external_company_cache ADD COLUMN IF NOT EXISTS payload_hash TEXT", "ALTER TABLE external_company_cache ADD COLUMN IF NOT EXISTS expires_at TIMESTAMPTZ", "CREATE UNIQUE INDEX IF NOT EXISTS ux_external_company_cache_lookup ON external_company_cache(source, lookup_type, lookup_value)", "CREATE INDEX IF NOT EXISTS idx_external_company_cache_nif ON external_company_cache(nif) WHERE nif IS NOT NULL AND nif <> ''", "CREATE INDEX IF NOT EXISTS idx_external_company_cache_name ON external_company_cache(normalized_name)", ]: conn.execute(text(stmt)) conn.execute(text(""" CREATE TABLE IF NOT EXISTS fiscal_customer_suggestions ( id UUID PRIMARY KEY DEFAULT gen_random_uuid(), opportunity_id UUID REFERENCES opportunities(id) ON DELETE CASCADE, reconciliation_item_id UUID, suggested_customer_id UUID REFERENCES customers(id) ON DELETE SET NULL, suggested_nif TEXT, suggested_name TEXT NOT NULL, source TEXT NOT NULL DEFAULT 'informa_pipeline_api', lookup_type TEXT, lookup_value TEXT, match_type TEXT, confidence NUMERIC(5,2) NOT NULL DEFAULT 0, reason TEXT, status TEXT NOT NULL DEFAULT 'pending', auto_applied BOOLEAN NOT NULL DEFAULT FALSE, raw_payload JSONB NOT NULL DEFAULT '{}'::jsonb, created_at TIMESTAMPTZ NOT NULL DEFAULT now(), updated_at TIMESTAMPTZ NOT NULL DEFAULT now(), resolved_at TIMESTAMPTZ, resolved_by TEXT ) """)) for stmt in [ "ALTER TABLE fiscal_customer_suggestions ADD COLUMN IF NOT EXISTS suggested_customer_id UUID REFERENCES customers(id) ON DELETE SET NULL", "ALTER TABLE fiscal_customer_suggestions ADD COLUMN IF NOT EXISTS auto_applied BOOLEAN NOT NULL DEFAULT FALSE", "ALTER TABLE fiscal_customer_suggestions ADD COLUMN IF NOT EXISTS lookup_type TEXT", "ALTER TABLE fiscal_customer_suggestions ADD COLUMN IF NOT EXISTS lookup_value TEXT", "ALTER TABLE fiscal_customer_suggestions ADD COLUMN IF NOT EXISTS resolved_by TEXT", "CREATE INDEX IF NOT EXISTS idx_fiscal_customer_suggestions_opp ON fiscal_customer_suggestions(opportunity_id, status)", "CREATE INDEX IF NOT EXISTS idx_fiscal_customer_suggestions_status ON fiscal_customer_suggestions(status, created_at DESC)", "CREATE UNIQUE INDEX IF NOT EXISTS ux_fiscal_customer_suggestion_opp_nif_pending ON fiscal_customer_suggestions(opportunity_id, suggested_nif) WHERE suggested_nif IS NOT NULL AND suggested_nif <> '' AND status = 'pending'", ]: conn.execute(text(stmt)) conn.execute(text(""" CREATE TABLE IF NOT EXISTS fiscal_enrichment_runs ( id UUID PRIMARY KEY DEFAULT gen_random_uuid(), mode TEXT NOT NULL DEFAULT 'incremental', status TEXT NOT NULL DEFAULT 'finished', seen INTEGER NOT NULL DEFAULT 0, enriched INTEGER NOT NULL DEFAULT 0, suggested INTEGER NOT NULL DEFAULT 0, auto_applied INTEGER NOT NULL DEFAULT 0, skipped INTEGER NOT NULL DEFAULT 0, errors JSONB NOT NULL DEFAULT '[]'::jsonb, payload JSONB NOT NULL DEFAULT '{}'::jsonb, started_at TIMESTAMPTZ NOT NULL DEFAULT now(), finished_at TIMESTAMPTZ NOT NULL DEFAULT now() ) """)) _SCHEMA_READY = True def _external_enabled() -> bool: return bool(getattr(settings, "external_company_lookup_enabled", False)) and bool(_clean(getattr(settings, "external_company_lookup_base_url", ""))) def _api_base_url() -> str: return _clean(getattr(settings, "external_company_lookup_base_url", "")).rstrip("/") def _api_key() -> str: return _clean(getattr(settings, "external_company_lookup_api_key", "")) def _timeout() -> float: return max(float(getattr(settings, "external_company_lookup_timeout", 10) or 10), 1.0) def _auto_threshold() -> float: return float(getattr(settings, "external_company_lookup_auto_threshold", DEFAULT_AUTO_THRESHOLD) or DEFAULT_AUTO_THRESHOLD) def _http_json(method: str, path: str, *, params: Optional[Dict[str, Any]] = None, payload: Optional[Dict[str, Any]] = None) -> Dict[str, Any]: if not _external_enabled(): return {"_skipped": "EXTERNAL_COMPANY_LOOKUP_ENABLED=false"} url = _api_base_url() + path if params: query = urllib.parse.urlencode({k: v for k, v in params.items() if v not in (None, "")}, doseq=True) if query: url += "?" + query headers = {"Accept": "application/json"} if _api_key(): headers["X-API-Key"] = _api_key() data = None if payload is not None: data = json.dumps(payload, ensure_ascii=False).encode("utf-8") headers["Content-Type"] = "application/json" request = urllib.request.Request(url, data=data, method=method.upper(), headers=headers) try: with urllib.request.urlopen(request, timeout=_timeout()) as response: # noqa: S310 - configured internal API endpoint raw = response.read().decode("utf-8") return json.loads(raw) if raw else {} except urllib.error.HTTPError as exc: if exc.code == 404: return {"_not_found": True, "status_code": 404} body = "" try: body = exc.read().decode("utf-8")[:500] except Exception: body = "" return {"_error": f"HTTP {exc.code}", "body": body} except Exception as exc: return {"_error": str(exc)} def _company_from_contact_item(item: Dict[str, Any]) -> Optional[Dict[str, Any]]: empresa = item.get("empresa") if not isinstance(empresa, dict) or not empresa: return None company = dict(empresa) company["score"] = _float(item.get("confidence"), _float(company.get("score"), 0.0)) company["match_type"] = _clean(item.get("match_type") or item.get("empresa_resolution_type") or company.get("match_type")) company["empresa_resolution_type"] = item.get("empresa_resolution_type") company["contacto_payload"] = item.get("contacto") or {} return company def _best_company_from_response(data: Dict[str, Any]) -> Optional[Dict[str, Any]]: if not isinstance(data, dict) or data.get("_error") or data.get("_skipped"): return None if data.get("nome") or data.get("nif"): company = dict(data) company.setdefault("score", 100.0 if data.get("nif") else 0.0) company.setdefault("match_type", "nif_exato" if data.get("nif") else "") return company contacts = data.get("contactos") if isinstance(contacts, list) and contacts: candidates = [_company_from_contact_item(x) for x in contacts if isinstance(x, dict)] candidates = [x for x in candidates if x] if candidates: return sorted(candidates, key=lambda c: _float(c.get("score")), reverse=True)[0] companies = data.get("empresas") or data.get("matches") if isinstance(companies, list) and companies: candidates = [x for x in companies if isinstance(x, dict)] if candidates: return sorted(candidates, key=lambda c: _float(c.get("score")), reverse=True)[0] return None def _company_to_customer_data(company: Dict[str, Any]) -> Dict[str, Any]: return { "name": _clean(company.get("nome") or company.get("legal_name") or company.get("name")), "tax_id": normalize_tax_id(company.get("nif") or company.get("tax_id")), "email": _clean(company.get("email_principal") or company.get("email")), "phone": _clean(company.get("telefone") or company.get("phone")), "street_name": _clean(company.get("morada") or company.get("address")), "postal_zone": _clean(company.get("codigo_postal") or company.get("postcode")), "city_name": _clean(company.get("localidade") or company.get("city")), "country": _clean(company.get("pais") or company.get("country") or "PT") or "PT", "metadata": { "fiscal_enrichment_source": CACHE_SOURCE, "external_company_id": company.get("id"), "cae": company.get("cae"), "duns": company.get("duns"), "website": company.get("website"), "district": company.get("distrito") or company.get("district"), "einforma_estado": company.get("einforma_estado"), "last_fiscal_enrichment_at": _now_iso(), }, } def _cache_lookup(lookup_type: str, lookup_value: str) -> Optional[Dict[str, Any]]: ensure_fiscal_enrichment_schema() with engine.begin() as conn: row = conn.execute(text(""" SELECT raw_payload FROM external_company_cache WHERE source = :source AND lookup_type = :lookup_type AND lookup_value = :lookup_value AND (expires_at IS NULL OR expires_at > now()) ORDER BY fetched_at DESC LIMIT 1 """), {"source": CACHE_SOURCE, "lookup_type": lookup_type, "lookup_value": lookup_value}).mappings().first() if row and isinstance(row.get("raw_payload"), dict): return dict(row["raw_payload"]) return None def _cache_company(lookup_type: str, lookup_value: str, company: Dict[str, Any]) -> None: ensure_fiscal_enrichment_schema() if not lookup_value or not company: return normalized_name = normalize_fiscal_name(company.get("nome") or company.get("legal_name") or company.get("name")) params = { "source": CACHE_SOURCE, "lookup_type": lookup_type, "lookup_value": lookup_value, "external_id": _clean(company.get("id")), "nif": normalize_tax_id(company.get("nif") or company.get("tax_id")) or None, "legal_name": _clean(company.get("nome") or company.get("legal_name") or company.get("name")) or None, "normalized_name": normalized_name or None, "cae": _clean(company.get("cae")) or None, "address": _clean(company.get("morada") or company.get("address")) or None, "postcode": _clean(company.get("codigo_postal") or company.get("postcode")) or None, "city": _clean(company.get("localidade") or company.get("city")) or None, "district": _clean(company.get("distrito") or company.get("district")) or None, "country": _clean(company.get("pais") or company.get("country") or "PT") or "PT", "website": _clean(company.get("website")) or None, "phone": _clean(company.get("telefone") or company.get("phone")) or None, "email": _clean(company.get("email_principal") or company.get("email")) or None, "duns": _clean(company.get("duns")) or None, "status": _clean(company.get("einforma_estado") or company.get("status")) or None, "score": _float(company.get("score"), 0.0), "match_type": _clean(company.get("match_type")) or None, "raw_payload": _json(company), "payload_hash": _sha256_payload(company), } with engine.begin() as conn: conn.execute(text(""" INSERT INTO external_company_cache ( source, lookup_type, lookup_value, external_id, nif, legal_name, normalized_name, cae, address, postcode, city, district, country, website, phone, email, duns, status, score, match_type, raw_payload, payload_hash, fetched_at, expires_at ) VALUES ( :source, :lookup_type, :lookup_value, :external_id, :nif, :legal_name, :normalized_name, :cae, :address, :postcode, :city, :district, :country, :website, :phone, :email, :duns, :status, :score, :match_type, CAST(:raw_payload AS JSONB), :payload_hash, now(), now() + interval '30 days' ) ON CONFLICT (source, lookup_type, lookup_value) DO UPDATE SET external_id = EXCLUDED.external_id, nif = EXCLUDED.nif, legal_name = EXCLUDED.legal_name, normalized_name = EXCLUDED.normalized_name, cae = EXCLUDED.cae, address = EXCLUDED.address, postcode = EXCLUDED.postcode, city = EXCLUDED.city, district = EXCLUDED.district, country = EXCLUDED.country, website = EXCLUDED.website, phone = EXCLUDED.phone, email = EXCLUDED.email, duns = EXCLUDED.duns, status = EXCLUDED.status, score = EXCLUDED.score, match_type = EXCLUDED.match_type, raw_payload = EXCLUDED.raw_payload, payload_hash = EXCLUDED.payload_hash, fetched_at = now(), expires_at = EXCLUDED.expires_at """), params) def batch_lookup_companies(*, nifs: Iterable[str] = (), names: Iterable[str] = (), emails: Iterable[str] = (), domains: Iterable[str] = (), websites: Iterable[str] = (), limit_per_query: int = 5) -> Dict[str, Any]: """Use POST /empresas/batch for large offline enrichment/reconciliation runs. The incremental worker prefers single lookups so it can stop as soon as it finds a strong match. Batch lookup is exposed for future nightly jobs and tests/manual diagnostics. """ payload = { "nifs": [normalize_tax_id(x) for x in nifs if normalize_tax_id(x)], "names": [_clean(x) for x in names if _clean(x)], "emails": [_clean(x).lower() for x in emails if _clean(x)], "domains": [normalize_domain(x) for x in domains if normalize_domain(x) and not is_public_email_domain(x)], "websites": [_clean(x) for x in websites if _clean(x)], "include_contactos": False, "limit_per_query": max(min(int(limit_per_query or 5), 20), 1), } payload = {k: v for k, v in payload.items() if v not in ([], "", None)} if not any(k in payload for k in ("nifs", "names", "emails", "domains", "websites")): return {"total_inputs": 0, "results": {}} return _http_json("POST", "/empresas/batch", payload=payload) def _lookup_external_by_nif(nif: str) -> Optional[Dict[str, Any]]: nif = normalize_tax_id(nif) if not nif: return None cached = _cache_lookup("nif", nif) if cached: return cached data = _http_json("GET", f"/empresas/nif/{urllib.parse.quote(nif)}", params={"include_contactos": "false"}) company = _best_company_from_response(data) if company: company.setdefault("score", 100.0) company.setdefault("match_type", "nif_exato") _cache_company("nif", nif, company) return company def _lookup_external_by_email(email: str) -> Optional[Dict[str, Any]]: email = _clean(email).lower() if not email or "@" not in email: return None cached = _cache_lookup("email", email) if cached: return cached data = _http_json("GET", "/contactos/search", params={"email": email, "limit": 10}) company = _best_company_from_response(data) if company: _cache_company("email", email, company) return company domain = _domain_from_email(email) if domain and not is_public_email_domain(domain): return _lookup_external_by_domain(domain) return None def _lookup_external_by_domain(domain: str) -> Optional[Dict[str, Any]]: domain = normalize_domain(domain) if not domain or is_public_email_domain(domain): return None cached = _cache_lookup("domain", domain) if cached: return cached data = _http_json("GET", f"/empresas/domain/{urllib.parse.quote(domain)}", params={"limit": 10, "include_contactos": "false"}) company = _best_company_from_response(data) if company: _cache_company("domain", domain, company) return company data = _http_json("GET", "/empresas/website", params={"domain": domain, "limit": 10, "include_contactos": "false"}) company = _best_company_from_response(data) if company: _cache_company("domain", domain, company) return company def _lookup_external_by_name(name: str) -> Optional[Dict[str, Any]]: name = _clean(name) if len(name) < 3 or not _is_valid_company_lookup_value(name) or _looks_like_person_name(name): return None normalized = normalize_fiscal_name(name) cached = _cache_lookup("name", normalized) if cached: return cached data = _http_json("GET", "/empresas/search", params={"q": name, "limit": 10, "include_contactos": "false"}) company = _best_company_from_response(data) if company: _cache_company("name", normalized, company) return company def _source_signals_from_opportunity(opportunity: Dict[str, Any]) -> List[Tuple[str, str]]: metadata = opportunity.get("metadata") if isinstance(opportunity.get("metadata"), dict) else {} signals: List[Tuple[str, str]] = [] for key in ("customer_tax_id", "tax_id", "nif"): value = normalize_tax_id(metadata.get(key) or opportunity.get(key)) if value: signals.append(("nif", value)) email = _clean(opportunity.get("customer_email") or metadata.get("customer_email") or metadata.get("email")).lower() if email and "@" in email: signals.append(("email", email)) domain = _domain_from_email(email) if domain and not is_public_email_domain(domain): signals.append(("domain", domain)) for key in ("website", "url"): domain = normalize_domain(metadata.get(key) or opportunity.get(key)) if domain and not is_public_email_domain(domain): signals.append(("domain", domain)) for value in (opportunity.get("linked_customer_tax_id"),): nif = normalize_tax_id(value) if nif: signals.append(("nif", nif)) name = _clean(opportunity.get("customer_name") or metadata.get("customer_name")) if name and _is_likely_company_name_signal(name): signals.append(("name", name)) # stable de-dup preserving order result: List[Tuple[str, str]] = [] seen = set() for kind, value in signals: key = (kind, value.casefold()) if value and key not in seen: seen.add(key) result.append((kind, value)) return result def _company_from_customer_row(row: Dict[str, Any], *, match_type: str = "email_identity_company_internal", score: float = 96.0) -> Dict[str, Any]: """Represent an existing ClientFlow customer as a company candidate.""" return { "id": _clean(row.get("id")), "nome": _clean(row.get("name")), "nif": normalize_tax_id(row.get("tax_id")), "email_principal": _clean(row.get("email")), "telefone": _clean(row.get("phone")), "morada": _clean(row.get("street_name")), "codigo_postal": _clean(row.get("postal_zone")), "localidade": _clean(row.get("city_name")), "pais": _clean(row.get("country") or "Portugal"), "score": score, "match_type": match_type, "clientflow_customer_id": _clean(row.get("id")), "source": "clientflow_internal_identity", } def _identity_company_mentions(identity: Optional[Dict[str, Any]]) -> List[str]: if not identity or not isinstance(identity.get("company_mentions"), list): return [] out: List[str] = [] for value in identity.get("company_mentions") or []: cleaned = _clean(value) if cleaned and is_plausible_company_mention(cleaned): out.append(cleaned) return out def _normalized_identity_mentions(identity: Optional[Dict[str, Any]]) -> List[str]: return [normalize_fiscal_name(x) for x in _identity_company_mentions(identity) if normalize_fiscal_name(x)] FISCAL_NAME_WEAK_TOKENS = { # Tokens that are common legal/geographic/sector descriptors and must never # be enough to associate a fiscal customer by themselves. Real audits found # false links such as Verifone Portugal -> ERT and Feteira/Torrão Engenharia # -> HUASI when generic tokens were treated as strong evidence. "portugal", "portuguesa", "portugues", "pt", "lda", "ltda", "limitada", "unipessoal", "sociedade", "empresa", "grupo", "group", "holding", "sgps", "sa", "s", "a", "comercial", "comercio", "comércio", "servicos", "servico", "serviço", "serviços", "engenharia", "engineer", "engineering", "construcao", "construção", "construcoes", "construções", "seguros", "seguro", "mediacao", "mediação", "contabilidade", "contabilista", "solucoes", "soluções", "solutions", "sistemas", "systems", "industrial", "industriais", "energy", "energia", "power", "electric", "eletrica", "elétrica", "tecnica", "técnica", } INVALID_COMPANY_LOOKUP_VALUES = { "pt", "com", "www", "mail", "email", "geral", "info", "contacto", "contato", "administrativo", "contabilidade", "financeiro", "support", "suporte", "noreply", "no-reply", } COMPANY_LEGAL_OR_ORG_TOKENS = { "lda", "ltda", "limitada", "unipessoal", "sa", "sgps", "sociedade", "empresa", "grupo", "holding", "associacao", "associação", "fundacao", "fundação", } def _strong_company_tokens(name_norm: str) -> set[str]: """Return meaningful tokens for fiscal-name matching. These tokens are used only as a fallback after exact/substring checks. They intentionally exclude weak legal/geographic words such as Portugal, LDA or S.A. so fuzzy identity extraction cannot auto-link unrelated customers that share only generic descriptors. """ return { token for token in (name_norm or "").split() if len(token) >= 4 and token not in FISCAL_NAME_WEAK_TOKENS } def _has_company_legal_or_org_signal(value: Any) -> bool: normalized = normalize_fiscal_name(value) tokens = set((normalized or "").split()) return bool(tokens & COMPANY_LEGAL_OR_ORG_TOKENS) def _is_valid_company_lookup_value(value: Any) -> bool: """Return False for fragments that should never trigger fiscal lookup. The worker used to create historical suggestions for values like "pt". Those are evidence of an email/domain, not a company identity. """ normalized = normalize_fiscal_name(value) if not normalized: return False tokens = [t for t in normalized.split() if t] if not tokens: return False if len(tokens) == 1: token = tokens[0] if token in INVALID_COMPANY_LOOKUP_VALUES or token in FISCAL_NAME_WEAK_TOKENS or len(token) < 3: return False strong = _strong_company_tokens(normalized) if not strong: return False return True def _looks_like_person_name(value: Any) -> bool: """Heuristic guard: avoid fiscal lookup by ordinary personal names.""" raw = _clean(value) normalized = normalize_fiscal_name(raw) if not normalized: return False tokens = [t for t in normalized.split() if t] if len(tokens) < 2 or len(tokens) > 4: return False if _has_company_legal_or_org_signal(raw): return False if any(t in FISCAL_NAME_WEAK_TOKENS for t in tokens): return False # Names written as normal title-case, e.g. "Nuno Silva" or "Bárbara Gonçalves", # are more likely contacts than companies. Uppercase fiscal names still pass. letters = [ch for ch in raw if ch.isalpha()] upper_ratio = (sum(1 for ch in letters if ch.isupper()) / len(letters)) if letters else 0.0 if upper_ratio < 0.75: return True return False def _is_likely_company_name_signal(value: Any) -> bool: """Return True only when a name is safe enough to use as fiscal-name signal.""" raw = _clean(value) if not _is_valid_company_lookup_value(raw): return False if _looks_like_person_name(raw): return False return _has_company_legal_or_org_signal(raw) or len(_strong_company_tokens(normalize_fiscal_name(raw))) >= 2 def _company_name_matches_mention(mention_norm: str, candidate_norm: str) -> bool: if not mention_norm or not candidate_norm: return False if len(mention_norm) < 4 or len(candidate_norm) < 4: return False mention_tokens = _strong_company_tokens(mention_norm) candidate_tokens = _strong_company_tokens(candidate_norm) if not mention_tokens or not candidate_tokens: return False if mention_norm == candidate_norm: return True if mention_tokens == candidate_tokens: return True shared = mention_tokens & candidate_tokens if shared: return True # Substring matching is useful for variants like "Dietimport S.A" vs # "DIETIMPORT, S.A.", but it must never bypass strong-token evidence. # This prevents generic sector overlaps such as Engenharia/Construções. return False def _has_manually_rejected_suggestion(opportunity_id: str, *, suggested_nif: str = "", lookup_value: str = "") -> bool: """Avoid re-applying the same fiscal suggestion that an operator rejected. A rejected wrong NIF must not block a future correct NIF for the same company mention. Therefore, when a NIF exists, the guard is scoped to that NIF only. Lookup-value matching is used only for suggestions without NIF. """ if not _clean(opportunity_id): return False nif = normalize_tax_id(suggested_nif) lookup = _clean(lookup_value) if not nif and not lookup: return False params = {"opportunity_id": opportunity_id, "nif": nif, "lookup_pattern": f"%{lookup}%"} if nif: where_clause = "suggested_nif = :nif" else: where_clause = "lookup_value ILIKE :lookup_pattern" with engine.begin() as conn: row = conn.execute(text(f""" SELECT 1 FROM fiscal_customer_suggestions WHERE opportunity_id = CAST(:opportunity_id AS UUID) AND status = 'rejected' AND ({where_clause}) LIMIT 1 """), params).first() return bool(row) def _find_internal_customer_by_identity(identity: Optional[Dict[str, Any]], opportunity: Dict[str, Any]) -> Optional[Dict[str, Any]]: """Prefer explicit company evidence found in the email body/signature. This prevents a weak external domain match from winning when the email text itself mentions a company already known by ClientFlow/Odoo/Jasmin, e.g. Dietimport S.A. in the legal disclaimer. """ if not identity: return None mentions = _identity_company_mentions(identity) normalized_mentions = _normalized_identity_mentions(identity) raw_domain = normalize_domain(identity.get("domain") or _domain_from_email(opportunity.get("customer_email"))) # v4928.1.5.76: public/ISP domains (sapo.pt, gmail.com, outlook.com, ...) # are valid contact channels but must never be treated as company identity # evidence. Two unrelated Portuguese companies can both use sapo.pt; using # that as an internal-customer domain match caused false confirmations. domain = "" if is_public_email_domain(raw_domain) else raw_domain if not normalized_mentions and not domain: return None with engine.begin() as conn: rows = conn.execute(text(""" SELECT id::text, name, tax_id, email, phone, street_name, postal_zone, city_name, country, metadata FROM customers WHERE (tax_id IS NOT NULL AND tax_id <> '') OR (email IS NOT NULL AND email <> '') OR name IS NOT NULL ORDER BY updated_at DESC LIMIT 2000 """)).mappings().all() best: Optional[Dict[str, Any]] = None best_score = 0.0 best_match_type = "email_identity_company_internal" for row in rows: name_norm = normalize_fiscal_name(row.get("name")) email_domain = _domain_from_email(row.get("email")) score = 0.0 match_type = "email_identity_company_internal" for mention in normalized_mentions: if mention and name_norm: if mention == name_norm: score = max(score, 98.0) match_type = "email_identity_company_internal_exact" elif _company_name_matches_mention(mention, name_norm): # Name-only overlap is useful for operator suggestions, but # it is not safe enough for auto-application. score = max(score, 88.0) match_type = "email_identity_company_internal" if domain and email_domain and not is_public_email_domain(email_domain) and domain == email_domain: score = max(score + 6.0 if score else 96.0, 96.0) match_type = "email_identity_company_internal_domain" # same address reinforces an explicit company mention if score >= 90 and _clean(identity.get("address")) and _clean(row.get("street_name")): if normalize_fiscal_name(row.get("street_name")) in normalize_fiscal_name(identity.get("address")): score += 2.0 if score > best_score: best_score = min(score, 99.0) best_match_type = match_type best = dict(row) if not best or best_score < 88: return None return _company_from_customer_row(best, match_type=best_match_type, score=best_score) def _identity_company_conflict(identity: Optional[Dict[str, Any]], company: Dict[str, Any]) -> bool: """Detect when an external candidate conflicts with explicit email identity.""" if not identity: return False mentions = _identity_company_mentions(identity) normalized_mentions = _normalized_identity_mentions(identity) if not normalized_mentions: return False candidate_name = normalize_fiscal_name(company.get("nome") or company.get("legal_name") or company.get("name")) if not candidate_name: return False for mention in normalized_mentions: if _company_name_matches_mention(mention, candidate_name): return False return True def _identity_mentions_match_name(identity: Optional[Dict[str, Any]], name: Any) -> bool: if not identity or not _clean(name): return False candidate_name = normalize_fiscal_name(name) mentions = _identity_company_mentions(identity) for mention in mentions: mention_norm = normalize_fiscal_name(mention) if _company_name_matches_mention(mention_norm, candidate_name): return True return False def _identity_conflicts_with_linked_customer(identity: Optional[Dict[str, Any]], opportunity: Dict[str, Any]) -> bool: linked_name = _clean(opportunity.get("linked_customer_name")) mentions = _identity_company_mentions(identity) if not linked_name or not mentions: return False return not _identity_mentions_match_name(identity, linked_name) def email_identity_review_for_opportunity(opportunity_id: str, *, refresh: bool = False) -> Dict[str, Any]: """Return the operator-facing identity review for an opportunity. This is read-only except when refresh=True, where it stores a new extraction. It intentionally runs even if a fiscal customer is already linked so the UI can show cases like: email mentions Dietimport S.A. but the opportunity is linked to Fmrl - Imobiliária S.A. """ ensure_fiscal_enrichment_schema() opportunity = get_opportunity(opportunity_id) if not opportunity: return {"ok": False, "reason": "opportunity_not_found"} try: identity = extract_identity_for_opportunity( opportunity_id, refresh=refresh, use_llm=bool(getattr(settings, "email_identity_extraction_use_llm", True)), ) except Exception as exc: return {"ok": False, "reason": f"email_identity_extraction_failed: {exc}"} valid_mentions = _identity_company_mentions(identity) identity_company = _find_internal_customer_by_identity(identity, opportunity) if valid_mentions else None linked_customer_id = _clean(opportunity.get("linked_customer_id") or opportunity.get("local_customer_id")) suggested_customer_id = _clean(identity_company.get("clientflow_customer_id")) if identity_company else "" conflict = _identity_conflicts_with_linked_customer(identity, opportunity) if valid_mentions else False if conflict and suggested_customer_id and linked_customer_id and suggested_customer_id == linked_customer_id: conflict = False return { "ok": True, "opportunity_id": opportunity_id, "identity": identity, "valid_company_mentions": valid_mentions, "linked_customer_id": linked_customer_id, "linked_customer_name": opportunity.get("linked_customer_name"), "linked_customer_tax_id": opportunity.get("linked_customer_tax_id"), "suggested_internal_customer": identity_company, "conflict": conflict, "status": "conflict" if conflict else ("suggestion" if identity_company else "identity_only"), } def assist_email_identity_enrichment(opportunity_id: str, *, refresh: bool = True, apply_safe: bool = False) -> Dict[str, Any]: """Use extracted email identity to create an assisted fiscal suggestion. v4.9.26.4 deliberately defaults to apply_safe=False. The goal is to show evidence and create pending suggestions for the operator, not to auto-link fiscal customers based only on an LLM extraction. """ ensure_fiscal_enrichment_schema() review = email_identity_review_for_opportunity(opportunity_id, refresh=refresh) if not review.get("ok"): return {"seen": 0, "suggested": 0, "auto_applied": 0, "skipped": 1, "reason": review.get("reason")} identity = review.get("identity") or {} opportunity = get_opportunity(opportunity_id) or {} valid_mentions = _identity_company_mentions(identity) identity_company = review.get("suggested_internal_customer") if valid_mentions else None linked_customer_id = _clean(review.get("linked_customer_id")) conflict = bool(review.get("conflict")) if valid_mentions else False event_type = "email_identity_review" event_note = "Identidade extraída do email para revisão fiscal assistida." suggestion = None if identity_company: confidence = _confidence_for_company(identity_company, lookup_type="email_identity") suggested_customer_id = _clean(identity_company.get("clientflow_customer_id")) or None # v4.9.26.5 / v4.9.26.4.1: # If the extracted identity points to the same fiscal customer already # linked to the opportunity, do not create another pending suggestion. # The operator needs a validation event, not duplicate work. if suggested_customer_id and linked_customer_id and suggested_customer_id == linked_customer_id and not conflict: with engine.begin() as conn: conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), 'email_identity_matches_current_fiscal_customer', :note, CAST(:payload AS JSONB), 'email_identity_assisted_enrichment') """), { "opportunity_id": opportunity_id, "note": f"Identidade extraída confirma o cliente fiscal atual: {identity_company.get('nome')} / {identity_company.get('nif') or 'sem NIF'}.", "payload": _json({"identity": identity, "customer_id": suggested_customer_id, "conflict": False}), }) conn.execute(text(""" UPDATE opportunities SET metadata = COALESCE(metadata, '{}'::jsonb) || jsonb_build_object( 'last_email_identity_validation', jsonb_build_object( 'status', 'matches_current_fiscal_customer', 'customer_id', CAST(:customer_id AS TEXT), 'customer_name', CAST(:customer_name AS TEXT), 'customer_tax_id', CAST(:customer_tax_id AS TEXT), 'validated_at', now() ) ), updated_at = now() WHERE id = CAST(:opportunity_id AS UUID) """), { "opportunity_id": opportunity_id, "customer_id": suggested_customer_id, "customer_name": identity_company.get("nome"), "customer_tax_id": identity_company.get("nif"), }) return { "seen": 1, "suggested": 0, "auto_applied": 0, "skipped": 1, "identity_used": True, "conflict": False, "suggestion_id": None, "status": "email_identity_matches_current_fiscal_customer", } lookup_value = ", ".join(identity.get("company_mentions") or []) if _has_manually_rejected_suggestion( opportunity_id, suggested_nif=identity_company.get("nif"), lookup_value=lookup_value, ): return { "seen": 1, "suggested": 0, "auto_applied": 0, "skipped": 1, "identity_used": True, "conflict": conflict, "reason": "manual_rejection_exists", } status = "pending" auto_applied = False if apply_safe and not conflict and not linked_customer_id and suggested_customer_id and _should_auto_apply(opportunity, identity_company, confidence=confidence): link_customer_to_opportunity(suggested_customer_id, opportunity_id) status = "accepted" auto_applied = True suggestion = _upsert_suggestion( opportunity_id, identity_company, lookup_type="email_identity", lookup_value=lookup_value, confidence=confidence, status=status, suggested_customer_id=suggested_customer_id, auto_applied=auto_applied, ) event_type = "email_identity_fiscal_conflict" if conflict else "email_identity_fiscal_suggestion" event_note = ( f"Possível conflito fiscal: email menciona {', '.join(identity.get('company_mentions') or [])}; " f"cliente atual {review.get('linked_customer_name') or '—'}." if conflict else f"Sugestão fiscal por identidade extraída do email: {identity_company.get('nome')} / {identity_company.get('nif') or 'sem NIF'}." ) with engine.begin() as conn: conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), :event_type, :note, CAST(:payload AS JSONB), 'email_identity_assisted_enrichment') """), { "opportunity_id": opportunity_id, "event_type": event_type, "note": event_note, "payload": _json({"identity": identity, "suggestion_id": suggestion.get("id") if suggestion else None, "conflict": conflict}), }) return { "seen": 1, "suggested": 1, "auto_applied": 1 if auto_applied else 0, "skipped": 0, "identity_used": True, "conflict": conflict, "suggestion_id": suggestion.get("id") if suggestion else None, "status": event_type, } # No internal customer match. If a company was explicitly mentioned, ask the # external lookup by name and create a pending suggestion, still assisted. mentions = valid_mentions for mention in mentions[:3]: try: company = _lookup_external_by_name(mention) except Exception: company = None if not company: continue confidence = _apply_identity_confidence_guard(company, identity=identity, confidence=_confidence_for_company(company, lookup_type="name")) if confidence < 75: continue if _has_manually_rejected_suggestion( opportunity_id, suggested_nif=company.get("nif") or company.get("tax_id"), lookup_value=mention, ): return { "seen": 1, "suggested": 0, "auto_applied": 0, "skipped": 1, "identity_used": True, "reason": "manual_rejection_exists", } suggestion = _upsert_suggestion( opportunity_id, company, lookup_type="email_identity_name", lookup_value=mention, confidence=confidence, status="pending", suggested_customer_id=None, auto_applied=False, ) with engine.begin() as conn: conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), 'email_identity_external_suggestion', :note, CAST(:payload AS JSONB), 'email_identity_assisted_enrichment') """), { "opportunity_id": opportunity_id, "note": f"Sugestão externa por empresa extraída do email: {mention}.", "payload": _json({"identity": identity, "suggestion_id": suggestion.get("id") if suggestion else None}), }) return {"seen": 1, "suggested": 1, "auto_applied": 0, "skipped": 0, "identity_used": True, "suggestion_id": suggestion.get("id") if suggestion else None, "status": "email_identity_external_suggestion"} with engine.begin() as conn: conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), :event_type, :note, CAST(:payload AS JSONB), 'email_identity_assisted_enrichment') """), { "opportunity_id": opportunity_id, "event_type": event_type, "note": event_note, "payload": _json({"identity": identity, "conflict": conflict}), }) return {"seen": 1, "suggested": 0, "auto_applied": 0, "skipped": 0, "identity_used": bool(identity), "status": "identity_only"} def _apply_source_name_confidence_guard( opportunity: Dict[str, Any], company: Dict[str, Any], *, lookup_type: str, confidence: float, ) -> float: """Block unsafe auto-application when an exact NIF maps to another company name. The external pipeline is a candidate source, not the fiscal source of truth. A syntactically exact NIF response can still be attached to the wrong row. When the operational source already carries a company-like legal name and the returned company has no strong name overlap, keep the result only as a manual suggestion. """ adjusted = float(confidence or 0.0) if str(lookup_type or "").strip().lower() != "nif": return adjusted source_name = _clean(opportunity.get("customer_name") or opportunity.get("title")) candidate_name = _clean(company.get("nome") or company.get("legal_name") or company.get("name")) if not (_is_likely_company_name_signal(source_name) and candidate_name): return adjusted source_norm = normalize_fiscal_name(source_name) candidate_norm = normalize_fiscal_name(candidate_name) if _company_name_matches_mention(source_norm, candidate_norm): return adjusted company["source_name_conflict"] = True company["source_company_name"] = source_name company["external_company_name"] = candidate_name return min(adjusted, 70.0) def _apply_identity_confidence_guard(company: Dict[str, Any], *, identity: Optional[Dict[str, Any]], confidence: float) -> float: """Lower confidence when the endpoint only matched a domain and the email mentions another company.""" match_type = _clean(company.get("match_type") or company.get("empresa_resolution_type")) adjusted = float(confidence or 0.0) if match_type in DOMAIN_ONLY_MATCH_TYPES: adjusted = min(adjusted, 85.0) if _identity_company_conflict(identity, company): adjusted = min(adjusted, 70.0) company["identity_conflict"] = True company["identity_company_mentions"] = identity.get("company_mentions") if identity else [] return adjusted def _lookup_company_for_signal(kind: str, value: str) -> Optional[Dict[str, Any]]: if kind == "nif": return _lookup_external_by_nif(value) if kind == "email": return _lookup_external_by_email(value) if kind == "domain": return _lookup_external_by_domain(value) if kind == "name": return _lookup_external_by_name(value) return None def _confidence_for_company(company: Dict[str, Any], *, lookup_type: str) -> float: score = _float(company.get("score"), 0.0) match_type = _clean(company.get("match_type") or company.get("empresa_resolution_type")) if match_type in VERY_STRONG_MATCH_TYPES: return max(score, 95.0) if match_type in STRONG_MATCH_TYPES: return max(score, 85.0) if match_type in MEDIUM_MATCH_TYPES: return max(score, 65.0) if lookup_type == "nif" and normalize_tax_id(company.get("nif")): return max(score, 100.0) if score: return score return 50.0 if lookup_type == "name" else 70.0 def _reason_for_company(company: Dict[str, Any], *, lookup_type: str, lookup_value: str) -> str: match_type = _clean(company.get("match_type") or company.get("empresa_resolution_type") or "match") name = _clean(company.get("nome") or company.get("legal_name") or company.get("name")) nif = normalize_tax_id(company.get("nif") or company.get("tax_id")) return f"{match_type} por {lookup_type}={lookup_value}; empresa={name}; nif={nif or '—'}" def _upsert_suggestion( opportunity_id: str, company: Dict[str, Any], *, lookup_type: str, lookup_value: str, confidence: float, status: str = "pending", suggested_customer_id: Optional[str] = None, auto_applied: bool = False, ) -> Dict[str, Any]: ensure_fiscal_enrichment_schema() name = _clean(company.get("nome") or company.get("legal_name") or company.get("name")) nif = normalize_tax_id(company.get("nif") or company.get("tax_id")) reason = _reason_for_company(company, lookup_type=lookup_type, lookup_value=lookup_value) match_type = _clean(company.get("match_type") or company.get("empresa_resolution_type")) params = { "opportunity_id": opportunity_id, "suggested_customer_id": suggested_customer_id, "suggested_nif": nif or None, "suggested_name": name, "source": CACHE_SOURCE, "lookup_type": lookup_type, "lookup_value": lookup_value, "match_type": match_type, "confidence": confidence, "reason": reason, "status": status, "auto_applied": auto_applied, "raw_payload": _json(company), } with engine.begin() as conn: row = conn.execute(text(""" INSERT INTO fiscal_customer_suggestions ( opportunity_id, suggested_customer_id, suggested_nif, suggested_name, source, lookup_type, lookup_value, match_type, confidence, reason, status, auto_applied, raw_payload, updated_at, resolved_at, resolved_by ) VALUES ( CAST(:opportunity_id AS UUID), CAST(:suggested_customer_id AS UUID), :suggested_nif, :suggested_name, :source, :lookup_type, :lookup_value, :match_type, :confidence, :reason, :status, :auto_applied, CAST(:raw_payload AS JSONB), now(), CASE WHEN :status <> 'pending' THEN now() ELSE NULL END, CASE WHEN :status <> 'pending' THEN 'fiscal_enrichment_worker' ELSE NULL END ) ON CONFLICT (opportunity_id, suggested_nif) WHERE suggested_nif IS NOT NULL AND suggested_nif <> '' AND status = 'pending' DO UPDATE SET suggested_customer_id = COALESCE(EXCLUDED.suggested_customer_id, fiscal_customer_suggestions.suggested_customer_id), suggested_name = EXCLUDED.suggested_name, lookup_type = EXCLUDED.lookup_type, lookup_value = EXCLUDED.lookup_value, match_type = EXCLUDED.match_type, confidence = GREATEST(EXCLUDED.confidence, fiscal_customer_suggestions.confidence), reason = EXCLUDED.reason, raw_payload = EXCLUDED.raw_payload, updated_at = now() RETURNING id::text, opportunity_id::text, suggested_customer_id::text, suggested_nif, suggested_name, source, lookup_type, lookup_value, match_type, confidence, reason, status, auto_applied, raw_payload, created_at, updated_at, resolved_at """), params).mappings().first() return dict(row or {}) def _has_conflicting_customer(opportunity: Dict[str, Any], suggested_tax_id: str) -> bool: linked_tax_id = normalize_tax_id(opportunity.get("linked_customer_tax_id")) return bool(linked_tax_id and suggested_tax_id and linked_tax_id != suggested_tax_id) def _should_auto_apply(opportunity: Dict[str, Any], company: Dict[str, Any], *, confidence: float) -> bool: if opportunity.get("linked_customer_id") or opportunity.get("local_customer_id"): return False tax_id = normalize_tax_id(company.get("nif") or company.get("tax_id")) name = _clean(company.get("nome") or company.get("legal_name") or company.get("name")) if not (tax_id and name): return False if _has_conflicting_customer(opportunity, tax_id): return False if company.get("identity_conflict") or company.get("source_name_conflict"): return False match_type = _clean(company.get("match_type") or company.get("empresa_resolution_type")) if match_type in DOMAIN_ONLY_MATCH_TYPES: return False if match_type == "email_identity_company_internal": return False if match_type == "email_identity_company_internal_exact": return False if match_type == "email_identity_company_internal_domain": return confidence >= _auto_threshold() return confidence >= _auto_threshold() and match_type in VERY_STRONG_MATCH_TYPES def apply_fiscal_suggestion(suggestion_id: str, *, actor: str = "operator") -> Dict[str, Any]: """Accept a suggestion and link/create the fiscal customer idempotently.""" ensure_fiscal_enrichment_schema() with engine.begin() as conn: row = conn.execute(text(""" SELECT * FROM fiscal_customer_suggestions WHERE id = CAST(:id AS UUID) LIMIT 1 """), {"id": suggestion_id}).mappings().first() if not row: return {"applied": False, "reason": "suggestion_not_found"} suggestion = dict(row) company = suggestion.get("raw_payload") if isinstance(suggestion.get("raw_payload"), dict) else {} opportunity_id = str(suggestion.get("opportunity_id") or "") opportunity = get_opportunity(opportunity_id) if not opportunity: return {"applied": False, "reason": "opportunity_not_found"} if opportunity.get("linked_customer_id") and normalize_tax_id(opportunity.get("linked_customer_tax_id")) != normalize_tax_id(suggestion.get("suggested_nif")): return {"applied": False, "reason": "opportunity_has_conflicting_customer"} customer = upsert_customer(_company_to_customer_data(company)) customer_id = str(customer.get("id") or "") if not customer_id: return {"applied": False, "reason": "customer_upsert_failed"} link_customer_to_opportunity(customer_id, opportunity_id) with engine.begin() as conn: conn.execute(text(""" UPDATE fiscal_customer_suggestions SET status = 'accepted', suggested_customer_id = CAST(:customer_id AS UUID), auto_applied = COALESCE(auto_applied, FALSE), resolved_at = now(), resolved_by = :actor, updated_at = now() WHERE id = CAST(:id AS UUID) """), {"id": suggestion_id, "customer_id": customer_id, "actor": actor}) conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), 'fiscal_enrichment_applied', :note, CAST(:payload AS JSONB), :actor) """), { "opportunity_id": opportunity_id, "note": f"Cliente fiscal associado por enriquecimento: {customer.get('name')} / {customer.get('tax_id') or 'sem NIF'}", "payload": _json({"suggestion_id": suggestion_id, "customer_id": customer_id, "source": CACHE_SOURCE}), "actor": actor, }) return {"applied": True, "customer_id": customer_id, "opportunity_id": opportunity_id} def reject_fiscal_suggestion(suggestion_id: str, *, actor: str = "operator", reason: str = "rejected_by_operator") -> Dict[str, Any]: ensure_fiscal_enrichment_schema() with engine.begin() as conn: row = conn.execute(text(""" UPDATE fiscal_customer_suggestions SET status = 'rejected', reason = COALESCE(reason, '') || ' | ' || :reason, resolved_at = now(), resolved_by = :actor, updated_at = now() WHERE id = CAST(:id AS UUID) RETURNING id::text """), {"id": suggestion_id, "actor": actor, "reason": reason}).mappings().first() return {"rejected": bool(row)} def list_fiscal_suggestions_for_opportunity(opportunity_id: str, *, limit: int = 5) -> List[Dict[str, Any]]: ensure_fiscal_enrichment_schema() with engine.begin() as conn: rows = conn.execute(text(""" SELECT s.id::text, s.opportunity_id::text, s.suggested_customer_id::text, COALESCE(NULLIF(s.suggested_nif, ''), c.tax_id) AS suggested_nif, COALESCE(NULLIF(s.suggested_name, ''), c.name) AS suggested_name, s.source, s.lookup_type, s.lookup_value, s.match_type, s.confidence, s.reason, s.status, s.auto_applied, s.created_at, s.updated_at, s.resolved_at FROM fiscal_customer_suggestions s LEFT JOIN customers c ON c.id = s.suggested_customer_id WHERE s.opportunity_id = CAST(:opportunity_id AS UUID) ORDER BY CASE s.status WHEN 'pending' THEN 0 WHEN 'accepted' THEN 1 ELSE 2 END, s.confidence DESC, s.updated_at DESC LIMIT :limit """), {"opportunity_id": opportunity_id, "limit": int(limit)}).mappings().all() return [dict(row) for row in rows] def enrich_opportunity(opportunity_id: str, *, apply_safe: bool = True) -> Dict[str, Any]: """Enrich a single opportunity with a fiscal customer suggestion/association.""" ensure_fiscal_enrichment_schema() opportunity = get_opportunity(opportunity_id) if not opportunity: return {"seen": 0, "enriched": 0, "suggested": 0, "auto_applied": 0, "skipped": 1, "reason": "opportunity_not_found"} if opportunity.get("linked_customer_id") or opportunity.get("local_customer_id"): return {"seen": 1, "enriched": 0, "suggested": 0, "auto_applied": 0, "skipped": 1, "reason": "already_has_fiscal_customer"} errors: List[str] = [] identity: Optional[Dict[str, Any]] = None if bool(getattr(settings, "email_identity_extraction_enabled", True)): try: identity = extract_identity_for_opportunity( opportunity_id, refresh=False, use_llm=bool(getattr(settings, "email_identity_extraction_use_llm", True)), ) except Exception as exc: errors.append(f"email_identity_extraction_failed: {exc}") identity = None identity_company = _find_internal_customer_by_identity(identity, opportunity) if identity_company: confidence = _confidence_for_company(identity_company, lookup_type="email_identity") existing_customer_id = _clean(identity_company.get("clientflow_customer_id")) or None lookup_value = ", ".join(identity.get("company_mentions") or []) if identity else "" if _has_manually_rejected_suggestion( opportunity_id, suggested_nif=identity_company.get("nif"), lookup_value=lookup_value, ): return { "seen": 1, "enriched": 0, "suggested": 0, "auto_applied": 0, "skipped": 1, "reason": "manual_rejection_exists", "errors": errors, "identity_used": True, } auto_applied = False status = "pending" if apply_safe and existing_customer_id and _should_auto_apply(opportunity, identity_company, confidence=confidence): link_customer_to_opportunity(existing_customer_id, opportunity_id) status = "accepted" auto_applied = True suggestion = _upsert_suggestion( opportunity_id, identity_company, lookup_type="email_identity", lookup_value=lookup_value, confidence=confidence, status=status, suggested_customer_id=existing_customer_id, auto_applied=auto_applied, ) if auto_applied: with engine.begin() as conn: conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), 'email_identity_fiscal_auto_applied', :note, CAST(:payload AS JSONB), 'email_identity_extraction_service') """), { "opportunity_id": opportunity_id, "note": f"Cliente fiscal auto-associado por empresa explícita no email: {identity_company.get('nome')} / {identity_company.get('nif') or 'sem NIF'}", "payload": _json({"suggestion_id": suggestion.get("id"), "identity": identity, "confidence": confidence}), }) return {"seen": 1, "enriched": 1, "suggested": 1, "auto_applied": 1, "skipped": 0, "customer_id": existing_customer_id, "suggestion_id": suggestion.get("id"), "identity_used": True} return {"seen": 1, "enriched": 0, "suggested": 1, "auto_applied": 0, "skipped": 0, "suggestion_id": suggestion.get("id"), "confidence": confidence, "errors": errors, "identity_used": True} for lookup_type, lookup_value in _source_signals_from_opportunity(opportunity): try: company = _lookup_company_for_signal(lookup_type, lookup_value) except Exception as exc: # keep worker resilient errors.append(str(exc)) continue if not company: continue name = _clean(company.get("nome") or company.get("legal_name") or company.get("name")) tax_id = normalize_tax_id(company.get("nif") or company.get("tax_id")) if not name: continue confidence = _confidence_for_company(company, lookup_type=lookup_type) confidence = _apply_identity_confidence_guard(company, identity=identity, confidence=confidence) confidence = _apply_source_name_confidence_guard( opportunity, company, lookup_type=lookup_type, confidence=confidence ) match_type = _clean(company.get("match_type") or company.get("empresa_resolution_type")) # v4.9.25.1: keep the enrichment queue operationally clean. # Fuzzy/name-only matches below 75 or explicit approximate-name matches # are too noisy for the normal workflow and should not create pending # suggestions. They can still be inspected by querying the external # source directly when needed. if match_type == "nome_aproximado" or confidence < 75.0: continue existing_customer_id: Optional[str] = None if tax_id: existing = get_customer_by_tax_id(tax_id) if existing: existing_customer_id = str(existing.get("id") or "") or None if _has_manually_rejected_suggestion(opportunity_id, suggested_nif=tax_id, lookup_value=lookup_value): return { "seen": 1, "enriched": 0, "suggested": 0, "auto_applied": 0, "skipped": 1, "reason": "manual_rejection_exists", "errors": errors, } auto_applied = False status = "pending" if apply_safe and _should_auto_apply(opportunity, company, confidence=confidence): customer = upsert_customer(_company_to_customer_data(company)) existing_customer_id = str(customer.get("id") or "") or existing_customer_id if existing_customer_id: link_customer_to_opportunity(existing_customer_id, opportunity_id) status = "accepted" auto_applied = True suggestion = _upsert_suggestion( opportunity_id, company, lookup_type=lookup_type, lookup_value=lookup_value, confidence=confidence, status=status, suggested_customer_id=existing_customer_id, auto_applied=auto_applied, ) if auto_applied: with engine.begin() as conn: conn.execute(text(""" INSERT INTO opportunity_events (id, opportunity_id, event_type, note, payload, created_by) VALUES (gen_random_uuid(), CAST(:opportunity_id AS UUID), 'fiscal_enrichment_auto_applied', :note, CAST(:payload AS JSONB), 'fiscal_enrichment_worker') """), { "opportunity_id": opportunity_id, "note": f"Cliente fiscal auto-associado: {name} / {tax_id}", "payload": _json({"suggestion_id": suggestion.get("id"), "lookup_type": lookup_type, "lookup_value": lookup_value, "confidence": confidence}), }) return {"seen": 1, "enriched": 1, "suggested": 1, "auto_applied": 1, "skipped": 0, "customer_id": existing_customer_id, "suggestion_id": suggestion.get("id")} return {"seen": 1, "enriched": 0, "suggested": 1, "auto_applied": 0, "skipped": 0, "suggestion_id": suggestion.get("id"), "confidence": confidence, "errors": errors} return {"seen": 1, "enriched": 0, "suggested": 0, "auto_applied": 0, "skipped": 1, "reason": "no_company_match", "errors": errors} def _open_opportunities_without_fiscal_customer(*, limit: int = 100) -> List[Dict[str, Any]]: # list_opportunities carries linked_customer_id and contact fields and uses # the same query path as the UI. Filter in Python to avoid duplicating a # large SQL projection here. opportunities = list_opportunities(status="open", limit=max(int(limit or 100), 1)) return [o for o in opportunities if not (o.get("linked_customer_id") or o.get("local_customer_id"))] def enrich_open_opportunities(*, limit: int = 100, apply_safe: bool = True, mode: str = "incremental") -> Dict[str, Any]: """Worker entrypoint: enrich open opportunities missing fiscal customer.""" ensure_fiscal_enrichment_schema() started = time.time() seen = enriched = suggested = auto_applied = skipped = 0 errors: List[str] = [] opportunities = _open_opportunities_without_fiscal_customer(limit=limit) for opportunity in opportunities[: int(limit or 100)]: result = enrich_opportunity(str(opportunity.get("id")), apply_safe=apply_safe) seen += int(result.get("seen") or 0) enriched += int(result.get("enriched") or 0) suggested += int(result.get("suggested") or 0) auto_applied += int(result.get("auto_applied") or 0) skipped += int(result.get("skipped") or 0) errors.extend(str(e) for e in (result.get("errors") or []) if e) summary = { "mode": mode, "enabled": _external_enabled(), "seen": seen, "enriched": enriched, "suggested": suggested, "auto_applied": auto_applied, "skipped": skipped, "errors": errors[:10], "duration_seconds": round(time.time() - started, 3), } with engine.begin() as conn: conn.execute(text(""" INSERT INTO fiscal_enrichment_runs (mode, status, seen, enriched, suggested, auto_applied, skipped, errors, payload, finished_at) VALUES (:mode, 'finished', :seen, :enriched, :suggested, :auto_applied, :skipped, CAST(:errors AS JSONB), CAST(:payload AS JSONB), now()) """), { "mode": mode, "seen": seen, "enriched": enriched, "suggested": suggested, "auto_applied": auto_applied, "skipped": skipped, "errors": _json(errors[:10]), "payload": _json(summary), }) return summary def fiscal_enrichment_summary() -> Dict[str, Any]: ensure_fiscal_enrichment_schema() with engine.begin() as conn: row = conn.execute(text(""" SELECT (SELECT COUNT(*) FROM opportunities WHERE status = 'open' AND local_customer_id IS NULL)::int AS open_without_fiscal_customer, (SELECT COUNT(*) FROM fiscal_customer_suggestions WHERE status = 'pending')::int AS pending_suggestions, (SELECT COUNT(*) FROM fiscal_customer_suggestions WHERE status = 'accepted' AND auto_applied = TRUE)::int AS auto_applied_suggestions, (SELECT MAX(finished_at) FROM fiscal_enrichment_runs)::text AS last_run_at """)).mappings().first() return dict(row or {})