Strict structured-output mode rejects uniqueItems, so every request to a gpt-5.6-family zero-data-retention endpoint failed with HTTP 400 behind a generic error; duplicates were already rejected server-side, so the keyword leaves the wire schemas, pinned by a strict-keyword allowlist test built from the ledger that hit this. The bare-BIC redaction pattern deleted every 8- and 11-letter word — Openbank, BAUMARKT, RACETRACKER — blinding the model to the payee it was asked to classify and tripping the unsafe-merchant check on honest answers. BICs now die only labeled or attached to their IBAN, account labels join the redaction secrets, an identifier-shaped merchant name degrades to a merchant-less proposal instead of failing the row, and a provider error inside an HTTP 200 envelope is reported as such (numeric code only) instead of as envelope corruption.
123 lines
4.2 KiB
Go
123 lines
4.2 KiB
Go
package classification
|
|
|
|
import (
|
|
"regexp"
|
|
"sort"
|
|
"strings"
|
|
"unicode"
|
|
"unicode/utf8"
|
|
|
|
"finance-duck/internal/domain"
|
|
)
|
|
|
|
var bankingPatterns = []*regexp.Regexp{
|
|
// Apply before tokenization to capture formatted identifiers as a unit.
|
|
// An IBAN may carry its BIC as the next token; both go as one unit. A
|
|
// *bare* BIC-shaped token is deliberately not redacted: the shape matches
|
|
// every 8- or 11-letter word ("Openbank", "BAUMARKT", "RACETRACKER"),
|
|
// which blinded the model to the very payee it should classify, and a
|
|
// bank code reveals nothing the prompt's institution field does not.
|
|
// Labeled forms ("BIC ...", "SWIFT ...") die with the label below.
|
|
regexp.MustCompile(`(?i)\b[a-z]{2}\s*\d{2}(?:[ -]?[a-z0-9]){11,30}\b(?:\s+[a-z]{6}[a-z0-9]{2}(?:[a-z0-9]{3})?\b)?`),
|
|
regexp.MustCompile(`(?i)\b[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b`),
|
|
regexp.MustCompile(`(?i)\b(?:iban|bic|swift|account(?:\s*(?:number|no))?|konto(?:nummer)?|reference|ref|payment\s*(?:id|reference)|end\s*to\s*end(?:\s*id)?|e2e|eref|mref|kref|cred|mandate|mandat(?:sreferenz)?|kunden(?:nummer|referenz)|kreditornummer|glaeubiger\s*id|gläubiger\s*id)\b[^;\n|]*`),
|
|
regexp.MustCompile(`(?i)\b(?:https?://|www\.)\S+|\b[^\s@]+@[^\s@]+\b`),
|
|
}
|
|
|
|
var identifierPatterns = append(append([]*regexp.Regexp{}, bankingPatterns...),
|
|
regexp.MustCompile(`\b\d{4,6}[\*x]{4,}\d{2,4}\b`),
|
|
regexp.MustCompile(`\b\d{4}-\d{2}-\d{2}T[\d:]+\b`),
|
|
)
|
|
|
|
// countDigits counts decimal digits in a token. The redaction rule drops a
|
|
// token with four or more, or with three among letters, so the count has to be
|
|
// over runes rather than bytes.
|
|
func countDigits(text string) int {
|
|
digits := 0
|
|
for _, r := range text {
|
|
if unicode.IsDigit(r) {
|
|
digits++
|
|
}
|
|
}
|
|
return digits
|
|
}
|
|
|
|
func addSecret(secrets map[string]bool, value string) {
|
|
normalized := normalize(value)
|
|
if normalized == "" {
|
|
return
|
|
}
|
|
secrets[normalized] = true
|
|
}
|
|
|
|
// redactor builds one text filter per request from the account registry, the
|
|
// facts being classified, and configured private names. Counterparties and
|
|
// stored transaction facts are deliberately not secrets.
|
|
func redactor(d domain.Dataset, f domain.Facts, private []string) func(string) string {
|
|
secrets := map[string]bool{}
|
|
for _, a := range d.Accounts {
|
|
addSecret(secrets, a.ID)
|
|
addSecret(secrets, a.IBAN)
|
|
addSecret(secrets, a.ExternalAccountID)
|
|
// People put their own name in the account label; the label is never
|
|
// sent as a field and its text is own-identity data, like PrivateNames.
|
|
addSecret(secrets, a.DisplayName)
|
|
}
|
|
for _, value := range []string{f.ID, f.ExternalID, f.Fingerprint, f.CounterpartyIBAN} {
|
|
addSecret(secrets, value)
|
|
}
|
|
for _, name := range private {
|
|
addSecret(secrets, name)
|
|
}
|
|
values := make([]string, 0, len(secrets))
|
|
for value := range secrets {
|
|
values = append(values, value)
|
|
}
|
|
sort.Slice(values, func(i, j int) bool {
|
|
if len(values[i]) != len(values[j]) {
|
|
return len(values[i]) > len(values[j])
|
|
}
|
|
return values[i] < values[j]
|
|
})
|
|
return func(text string) string {
|
|
if !utf8.ValidString(text) {
|
|
return ""
|
|
}
|
|
for _, pattern := range identifierPatterns {
|
|
text = pattern.ReplaceAllString(text, " ")
|
|
}
|
|
text = " " + normalize(text) + " "
|
|
for _, value := range values {
|
|
needle := " " + value + " "
|
|
for strings.Contains(text, needle) {
|
|
text = strings.ReplaceAll(text, needle, " ")
|
|
}
|
|
}
|
|
kept, length := make([]string, 0, 16), 0
|
|
for _, token := range strings.Fields(text) {
|
|
digits := countDigits(token)
|
|
if digits >= 4 || (digits >= 3 && digits < utf8.RuneCountInString(token)) || utf8.RuneCountInString(token) > 40 {
|
|
continue
|
|
}
|
|
if length+len(token) > 500 {
|
|
break
|
|
}
|
|
kept = append(kept, token)
|
|
length += len(token) + 1
|
|
}
|
|
return strings.Join(kept, " ")
|
|
}
|
|
}
|
|
|
|
// redact is the stateless dataset-only form used when no current Facts object
|
|
// is available. Classification uses redactor so the current row's own ids are
|
|
// also removed.
|
|
func redact(text string, d domain.Dataset, private []string) string {
|
|
return redactor(d, domain.Facts{}, private)(text)
|
|
}
|
|
|
|
// Redact applies the identifier-only policy to one text field.
|
|
func Redact(text string, data domain.Dataset, facts domain.Facts, private []string) string {
|
|
return redactor(data, facts, private)(text)
|
|
}
|