168 lines
6.0 KiB
Python
168 lines
6.0 KiB
Python
"""BLIF business knowledge retrieval for reply generation.
|
|
|
|
The goal is deliberately pragmatic: keep mutable business facts outside the LLM,
|
|
retrieve only relevant approved knowledge, and let the LLM (when enabled) write
|
|
an editable draft. This is not fine-tuning; it is controlled business context.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
from dataclasses import dataclass, field
|
|
from functools import lru_cache
|
|
from pathlib import Path
|
|
from typing import Any, Dict, Iterable, List, Optional
|
|
|
|
|
|
KNOWLEDGE_PATH = Path(__file__).resolve().parent / "business_knowledge" / "blif_knowledge.json"
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class KnowledgeTopic:
|
|
id: str
|
|
title: str
|
|
summary: str
|
|
keywords: tuple[str, ...] = field(default_factory=tuple)
|
|
facts: tuple[str, ...] = field(default_factory=tuple)
|
|
forbidden: tuple[str, ...] = field(default_factory=tuple)
|
|
reply_type: str = "answer_without_attachment"
|
|
default_template_code: str = "BUSINESS_KNOWLEDGE_REPLY"
|
|
score: float = 0.0
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class KnowledgeMatch:
|
|
version: str
|
|
source: str
|
|
topics: tuple[KnowledgeTopic, ...]
|
|
products: tuple[Dict[str, Any], ...]
|
|
accessories: tuple[Dict[str, Any], ...]
|
|
|
|
@property
|
|
def has_topics(self) -> bool:
|
|
return bool(self.topics)
|
|
|
|
@property
|
|
def primary_topic(self) -> Optional[KnowledgeTopic]:
|
|
return self.topics[0] if self.topics else None
|
|
|
|
@property
|
|
def default_template_code(self) -> str:
|
|
if self.primary_topic:
|
|
return self.primary_topic.default_template_code
|
|
return "BUSINESS_KNOWLEDGE_REPLY"
|
|
|
|
@property
|
|
def reply_type(self) -> str:
|
|
if self.primary_topic:
|
|
return self.primary_topic.reply_type
|
|
return "answer_without_attachment"
|
|
|
|
def to_prompt_context(self, *, max_topics: int = 4) -> Dict[str, Any]:
|
|
return {
|
|
"knowledge_version": self.version,
|
|
"source": self.source,
|
|
"topics": [
|
|
{
|
|
"id": topic.id,
|
|
"title": topic.title,
|
|
"summary": topic.summary,
|
|
"facts": list(topic.facts),
|
|
"forbidden": list(topic.forbidden),
|
|
"reply_type": topic.reply_type,
|
|
"default_template_code": topic.default_template_code,
|
|
}
|
|
for topic in self.topics[:max_topics]
|
|
],
|
|
}
|
|
|
|
|
|
def _normalize(text: str) -> str:
|
|
text = str(text or "").lower()
|
|
text = text.replace("á", "a").replace("à", "a").replace("ã", "a").replace("â", "a")
|
|
text = text.replace("é", "e").replace("ê", "e")
|
|
text = text.replace("í", "i")
|
|
text = text.replace("ó", "o").replace("õ", "o").replace("ô", "o")
|
|
text = text.replace("ú", "u")
|
|
text = text.replace("ç", "c")
|
|
return re.sub(r"\s+", " ", text).strip()
|
|
|
|
|
|
@lru_cache(maxsize=1)
|
|
def load_business_knowledge() -> Dict[str, Any]:
|
|
with KNOWLEDGE_PATH.open("r", encoding="utf-8") as fh:
|
|
return json.load(fh)
|
|
|
|
|
|
def _topic_from_dict(data: Dict[str, Any], *, score: float = 0.0) -> KnowledgeTopic:
|
|
return KnowledgeTopic(
|
|
id=str(data.get("id") or ""),
|
|
title=str(data.get("title") or ""),
|
|
summary=str(data.get("summary") or ""),
|
|
keywords=tuple(str(item) for item in data.get("keywords") or []),
|
|
facts=tuple(str(item) for item in data.get("facts") or []),
|
|
forbidden=tuple(str(item) for item in data.get("forbidden") or []),
|
|
reply_type=str(data.get("reply_type") or "answer_without_attachment"),
|
|
default_template_code=str(data.get("default_template_code") or "BUSINESS_KNOWLEDGE_REPLY"),
|
|
score=score,
|
|
)
|
|
|
|
|
|
def retrieve_business_knowledge(text: str, *, limit: int = 4) -> KnowledgeMatch:
|
|
"""Return approved BLIF knowledge relevant to a customer message.
|
|
|
|
This intentionally uses deterministic keyword scoring first. It is cheap,
|
|
transparent and good enough for high-frequency commercial topics. A future
|
|
vector search can replace this function without changing the reply pipeline.
|
|
"""
|
|
data = load_business_knowledge()
|
|
normalized = _normalize(text)
|
|
scored: List[KnowledgeTopic] = []
|
|
for raw_topic in data.get("topics") or []:
|
|
score = 0.0
|
|
keywords = list(raw_topic.get("keywords") or [])
|
|
for keyword in keywords:
|
|
kw = _normalize(keyword)
|
|
if not kw:
|
|
continue
|
|
if kw in normalized:
|
|
# Longer/multi-word matches carry a bit more intent signal.
|
|
score += 2.0 if " " in kw else 1.0
|
|
# If the title appears in text, boost lightly.
|
|
title = _normalize(raw_topic.get("title") or "")
|
|
if title and title in normalized:
|
|
score += 1.5
|
|
if score > 0:
|
|
scored.append(_topic_from_dict(raw_topic, score=score))
|
|
|
|
scored.sort(key=lambda item: item.score, reverse=True)
|
|
return KnowledgeMatch(
|
|
version=str(data.get("version") or ""),
|
|
source=str(data.get("source") or ""),
|
|
topics=tuple(scored[:limit]),
|
|
products=tuple(data.get("products") or []),
|
|
accessories=tuple(data.get("accessories") or []),
|
|
)
|
|
|
|
|
|
def format_catalog_prices(match: Optional[KnowledgeMatch] = None) -> str:
|
|
data = load_business_knowledge()
|
|
products: Iterable[Dict[str, Any]] = data.get("products") or []
|
|
accessories: Iterable[Dict[str, Any]] = data.get("accessories") or []
|
|
lines = ["Carregadores BLIF (s/IVA):"]
|
|
for product in products:
|
|
lines.append(f"- {product.get('name')}: {product.get('price_without_vat')} €")
|
|
lines.append("Acessórios principais (s/IVA):")
|
|
for accessory in accessories:
|
|
lines.append(f"- {accessory.get('name')}: {accessory.get('price_without_vat')} €")
|
|
return "\n".join(lines)
|
|
|
|
|
|
def knowledge_summary_lines(match: KnowledgeMatch) -> List[str]:
|
|
lines: List[str] = []
|
|
for topic in match.topics:
|
|
lines.append(f"{topic.title}: {topic.summary}")
|
|
for fact in topic.facts[:4]:
|
|
lines.append(f"- {fact}")
|
|
return lines
|