Projeto do Agent Contas ORACLE
This commit is contained in:
135
app/domain/contas/item_matcher.py
Normal file
135
app/domain/contas/item_matcher.py
Normal file
@@ -0,0 +1,135 @@
|
||||
"""Matcher conservador de nomes de item para o InvoiceResolver.
|
||||
|
||||
Porta o scoring determinístico do Contas original (grafia + fonética), mas não
|
||||
mantém LangChain/gateway próprio. O fallback conversacional é responsabilidade do
|
||||
agent_framework: quando mais de um candidato permanece plausível, o resultado é
|
||||
ambíguo e o runtime pede clarificação ao cliente.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import unicodedata
|
||||
from typing import Any
|
||||
|
||||
try:
|
||||
import jellyfish # type: ignore
|
||||
except Exception: # pragma: no cover - fallback usado em ambientes mínimos
|
||||
jellyfish = None
|
||||
from difflib import SequenceMatcher
|
||||
from . import string_metrics as _fallback_metrics
|
||||
|
||||
from .invoice_resolver import ItemMatcherError
|
||||
|
||||
_TOP_K = 10
|
||||
_MIN_TOKEN_LEN = 3
|
||||
_GENERIC_TOKENS = frozenset({"app", "premium", "plus", "light", "mensal", "mes", "dados", "sva"})
|
||||
|
||||
|
||||
def _normalize(text: str) -> str:
|
||||
decomposed = unicodedata.normalize("NFKD", str(text).lower())
|
||||
return "".join(c for c in decomposed if not unicodedata.combining(c))
|
||||
|
||||
|
||||
def _phrase_sim(a: str, b: str) -> float:
|
||||
if jellyfish is not None:
|
||||
return jellyfish.jaro_winkler_similarity(a, b)
|
||||
return _fallback_metrics.jaro_winkler_similarity(a, b)
|
||||
|
||||
|
||||
def _significant_tokens(text: str) -> list[str]:
|
||||
return [t for t in text.split() if len(t) >= _MIN_TOKEN_LEN and t not in _GENERIC_TOKENS]
|
||||
|
||||
|
||||
def _token_sim(mention: str, candidate: str) -> float:
|
||||
tokens = _significant_tokens(candidate)
|
||||
if not tokens:
|
||||
return _phrase_sim(mention, candidate)
|
||||
return max(_phrase_sim(mention, t) for t in tokens)
|
||||
|
||||
|
||||
def _code_sim(a: str, b: str) -> float:
|
||||
if not a or not b:
|
||||
return 0.0
|
||||
if jellyfish is not None:
|
||||
distance = jellyfish.levenshtein_distance(a, b)
|
||||
return 1.0 - distance / max(len(a), len(b))
|
||||
distance = _fallback_metrics.levenshtein_distance(a, b)
|
||||
return 1.0 - distance / max(len(a), len(b))
|
||||
|
||||
|
||||
def _phon_sim(mention: str, candidate: str) -> float:
|
||||
code_m = jellyfish.metaphone(mention) if jellyfish is not None else _fallback_metrics.metaphone(mention)
|
||||
if not code_m:
|
||||
return 0.0
|
||||
tokens = _significant_tokens(candidate) or [candidate]
|
||||
return max(_code_sim(code_m, jellyfish.metaphone(token) if jellyfish is not None else _fallback_metrics.metaphone(token)) for token in tokens)
|
||||
|
||||
|
||||
|
||||
|
||||
def _token_pair_sim(a: str, b: str) -> float:
|
||||
"""Combina evidência ortográfica e fonética para um par de tokens.
|
||||
|
||||
O pequeno bônus pela segunda evidência resolve transcrições curtas como
|
||||
``apou`` -> ``apple`` sem transformar prefixos puramente gráficos em match.
|
||||
"""
|
||||
jw = _phrase_sim(a, b)
|
||||
code_a = jellyfish.metaphone(a) if jellyfish is not None else _fallback_metrics.metaphone(a)
|
||||
code_b = jellyfish.metaphone(b) if jellyfish is not None else _fallback_metrics.metaphone(b)
|
||||
ph = _code_sim(code_a, code_b)
|
||||
return min(1.0, max(jw, ph) + 0.15 * min(jw, ph))
|
||||
|
||||
|
||||
def _token_alignment_sim(mention: str, candidate: str) -> float:
|
||||
mention_tokens = [t for t in mention.split() if len(t) >= 2]
|
||||
candidate_tokens = [t for t in candidate.split() if len(t) >= _MIN_TOKEN_LEN and t not in _GENERIC_TOKENS]
|
||||
if not mention_tokens or not candidate_tokens:
|
||||
return 0.0
|
||||
# Cada token reconhecido pelo ASR precisa encontrar seu melhor correspondente.
|
||||
# A média impede que um token genérico perfeito (ex.: ``tim``) esconda o
|
||||
# discriminante errado (``miusic`` vs ``games``).
|
||||
return sum(max(_token_pair_sim(mt, ct) for ct in candidate_tokens) for mt in mention_tokens) / len(mention_tokens)
|
||||
|
||||
def _grafia_sim(mention: str, candidate: str) -> float:
|
||||
return max(_phrase_sim(mention, candidate), _token_sim(mention, candidate))
|
||||
|
||||
|
||||
class SimilarityItemMatcher:
|
||||
"""Matcher síncrono compatível com ``InvoiceResolver.ItemMatcherLLM``.
|
||||
|
||||
A implementação é deliberadamente fail-closed: só resolve automaticamente
|
||||
quando o melhor candidato tem score alto e margem suficiente. Empates
|
||||
plausíveis são devolvidos juntos para o framework pedir clarificação.
|
||||
"""
|
||||
|
||||
def __init__(self, *, accept_threshold: float = 0.78, ambiguity_margin: float = 0.06, top_k: int = _TOP_K) -> None:
|
||||
self.accept_threshold = float(accept_threshold)
|
||||
self.ambiguity_margin = float(ambiguity_margin)
|
||||
self.top_k = int(top_k)
|
||||
|
||||
def score(self, mention: str, candidate: str) -> float:
|
||||
nm, nd = _normalize(mention), _normalize(candidate)
|
||||
return max(_grafia_sim(nm, nd), _phon_sim(nm, nd), _token_alignment_sim(nm, nd))
|
||||
|
||||
def best_similarity(self, mention: str, candidates: list[str]) -> float:
|
||||
if not candidates:
|
||||
return 0.0
|
||||
return max(self.score(mention, candidate) for candidate in candidates)
|
||||
|
||||
def ranked(self, mention: str, candidates: list[str]) -> list[tuple[str, float]]:
|
||||
ranked = [(candidate, self.score(mention, candidate)) for candidate in candidates]
|
||||
ranked.sort(key=lambda pair: pair[1], reverse=True)
|
||||
return ranked[: self.top_k]
|
||||
|
||||
def match(self, mention: str, candidates: list[str], *, callbacks: list[Any] | None = None) -> list[str]:
|
||||
del callbacks
|
||||
if not candidates:
|
||||
return []
|
||||
ranked = self.ranked(mention, candidates)
|
||||
if not ranked or ranked[0][1] < self.accept_threshold:
|
||||
return []
|
||||
best = ranked[0][1]
|
||||
plausible = [name for name, score in ranked if score >= self.accept_threshold and best - score <= self.ambiguity_margin]
|
||||
return plausible or [ranked[0][0]]
|
||||
|
||||
|
||||
__all__ = ["SimilarityItemMatcher"]
|
||||
Reference in New Issue
Block a user