perf: optimize reconciliation candidate lookup

This commit is contained in:
plx
2026-08-14 12:56:03 +00:00
parent 89017f71a6
commit 07dc9db198
8 changed files with 373 additions and 31 deletions

View File

@@ -0,0 +1,216 @@
from contextlib import nullcontext
from inspect import getsource
from pathlib import Path
import app.commercial_service as commercial_service
import app.jasmin_backfill_service as service
import app.operation_service as operation_service
import app.opportunity_service as opportunity_service
import app.product_service as product_service
class _Result:
def __init__(self, rows=()):
self.rows = list(rows)
def mappings(self):
return self
def all(self):
return self.rows
def first(self):
return self.rows[0] if self.rows else None
class _Connection:
def __init__(self, structured_rows, payload_rows=(), *, fiscal_email="buyer@example.test", fiscal_tax_id="501000000"):
self.structured_rows = structured_rows
self.payload_rows = payload_rows
self.fiscal_email = fiscal_email
self.fiscal_tax_id = fiscal_tax_id
self.statements = []
def execute(self, statement, params=None):
sql = str(statement)
self.statements.append(sql)
if "FROM opportunities o" in sql:
return _Result([{
"id": "11111111-1111-1111-1111-111111111111",
"customer_name": "Example",
"customer_email": "buyer@example.test",
"local_customer_id": "22222222-2222-2222-2222-222222222222",
"fiscal_customer_name": "Example Lda",
"fiscal_customer_tax_id": self.fiscal_tax_id,
"fiscal_customer_email": self.fiscal_email,
}])
if "FROM commercial_documents" in sql and "SELECT\n id::text" in sql:
return _Result()
if "WITH candidate_ids AS" in sql:
return _Result(self.structured_rows)
if "ri.payload::text ILIKE" in sql:
return _Result(self.payload_rows)
raise AssertionError(sql)
class _Engine:
def __init__(self, connection):
self.connection = connection
def begin(self):
return nullcontext(self.connection)
def _candidate(candidate_id, score, reason, *, document_date="2026-08-01"):
return {
"id": candidate_id,
"source_system": "jasmin",
"external_type": "jasmin_invoice",
"external_id": candidate_id,
"title": candidate_id,
"status": "open",
"opportunity_id": None,
"customer_id": None,
"customer_name": "Example Lda",
"customer_email": "buyer@example.test",
"customer_tax_id": "501000000",
"document_number": candidate_id,
"document_date": document_date,
"amount": 10,
"currency": "EUR",
"payload": {},
"created_at": "2026-08-01",
"updated_at": "2026-08-02",
"match_score": score,
"match_reason": reason,
}
def test_structured_full_page_skips_payload_statement(monkeypatch):
structured = [_candidate("customer", 100, "customer_id"), _candidate("nif", 98, "nif")]
connection = _Connection(structured)
monkeypatch.setattr(service, "engine", _Engine(connection))
result = service.find_jasmin_document_candidates_for_opportunity(
"11111111-1111-1111-1111-111111111111", limit=2,
)
assert [row["match_reason"] for row in result] == ["customer_id", "nif"]
assert len(connection.statements) == 3 # opportunity, current docs, structured
assert sum("payload::text ILIKE" in sql for sql in connection.statements) == 0
def test_payload_legacy_fallback_preserves_precedence_and_order(monkeypatch):
structured = [_candidate("email", 90, "email", document_date="2026-08-10")]
payload = [
_candidate("payload-nif", 96, "payload_nif", document_date="2026-07-01"),
_candidate("payload-domain", 78, "domain_payload", document_date="2026-08-12"),
]
connection = _Connection(structured, payload)
monkeypatch.setattr(service, "engine", _Engine(connection))
result = service.find_jasmin_document_candidates_for_opportunity(
"11111111-1111-1111-1111-111111111111", limit=5,
)
assert {row["match_reason"] for row in result} == {"payload_nif", "email", "domain_payload"}
assert len(connection.statements) == 4
assert sum("payload::text ILIKE" in sql for sql in connection.statements) == 1
def test_public_domain_never_adds_domain_payload_predicate(monkeypatch):
connection = _Connection([], [_candidate("payload-nif", 96, "payload_nif")], fiscal_email="geral.am.energy@gmail.com")
monkeypatch.setattr(service, "engine", _Engine(connection))
result = service.find_jasmin_document_candidates_for_opportunity(
"11111111-1111-1111-1111-111111111111", limit=8,
)
assert [row["match_reason"] for row in result] == ["payload_nif"]
payload_sql = connection.statements[-1]
assert "CAST(:domain AS TEXT)" not in payload_sql
assert "domain_payload" not in payload_sql
def test_public_domain_without_tax_has_no_payload_statement(monkeypatch):
for domain in ("gmail.com", "outlook.com", "hotmail.com", "yahoo.com"):
connection = _Connection([], fiscal_email=f"person@{domain}", fiscal_tax_id="")
monkeypatch.setattr(service, "engine", _Engine(connection))
assert service.find_jasmin_document_candidates_for_opportunity(
"11111111-1111-1111-1111-111111111111", limit=8,
) == []
assert len(connection.statements) == 3
assert all("payload::text ILIKE" not in sql for sql in connection.statements)
def test_private_business_domain_keeps_domain_payload(monkeypatch):
candidate = _candidate("private-domain", 78, "domain_payload")
candidate["customer_tax_id"] = ""
connection = _Connection([], [candidate], fiscal_email="sales@private-business.pt", fiscal_tax_id="")
monkeypatch.setattr(service, "engine", _Engine(connection))
result = service.find_jasmin_document_candidates_for_opportunity(
"11111111-1111-1111-1111-111111111111", limit=8,
)
assert [row["match_reason"] for row in result] == ["domain_payload"]
assert "CAST(:domain AS TEXT)" in connection.statements[-1]
def test_explicit_conflicting_candidate_nif_is_not_actionable(monkeypatch):
conflict = _candidate("conflict", 90, "email")
conflict["customer_tax_id"] = "999999999"
connection = _Connection([conflict])
monkeypatch.setattr(service, "engine", _Engine(connection))
assert service.find_jasmin_document_candidates_for_opportunity(
"11111111-1111-1111-1111-111111111111", limit=8,
) == []
def test_required_public_domain_provider_policy():
for domain in (
"gmail.com", "googlemail.com", "hotmail.com", "outlook.com", "live.com",
"yahoo.com", "icloud.com", "me.com", "sapo.pt",
):
assert service.is_public_email_domain(domain)
def test_candidate_sql_preserves_exclusion_schema_and_ordering():
source = getsource(service.find_jasmin_document_candidates_for_opportunity)
assert source.count("NOT EXISTS") >= 2
assert "cd.document_number = ri.document_number" in source
assert "cd.external_id = ri.external_id" in source
assert "ORDER BY match_score DESC, ri.document_date DESC NULLS LAST, ri.updated_at DESC" in source
for reason in ("customer_id", "nif", "payload_nif", "email", "domain_payload", "exact_name"):
assert reason in source
def test_opportunity_detail_read_helpers_do_not_initialize_schema():
readers = (
opportunity_service.get_opportunity,
opportunity_service.list_opportunity_tasks,
opportunity_service.list_opportunity_events,
product_service.list_products,
product_service.list_opportunity_items,
commercial_service.list_commercial_documents,
commercial_service.get_customer_for_opportunity,
operation_service.get_operation_links,
)
for reader in readers:
assert "ensure_" not in getsource(reader), reader.__name__
def test_only_plan_specific_indexes_are_migrated():
migration = Path("migrations/009_reconciliation_candidate_lookup_indexes.sql").read_text()
for expression in (
"reconciliation_items(customer_id)",
"reconciliation_items(lower(customer_email))",
"reconciliation_items(lower(customer_name))",
"commercial_documents(opportunity_id, document_number)",
"commercial_documents(opportunity_id, external_id)",
):
assert expression in migration
assert "customer_tax_id" not in migration
assert "reconciliation_items(opportunity_id)" not in migration