feat(search): Elasticsearch-клиент (имя + контент) + oo search (#37) #41

Merged
eSlider merged 13 commits from feat/es-search#37 into main 2026-09-16 17:43:21 +01:00
2 changed files with 191 additions and 34 deletions
Showing only changes of commit d95985e6d0 - Show all commits
+146 -34
View File
@@ -18,36 +18,52 @@ import (
onlyoffice "github.com/eslider/go-onlyoffice" onlyoffice "github.com/eslider/go-onlyoffice"
) )
// amountLabels are the payable-amount labels in priority order: the first // amountPat is the amount capture shared by every amount regex.
// label present in a document wins. Within one label the last amount is taken, const amountPat = `([0-9]+(?:[.,][0-9]+)*)`
// because totals usually come last.
// // amountRE builds "<label> [optional (comment)] [: -] <number>".
// "gesamtsumme" is not in the original list but is the real label on Diashop func amountRE(label string) *regexp.Regexp {
// invoices: "Gesamtsumme (inkl. Steuern)". return regexp.MustCompile(
var amountLabels = []string{ `(?i)\b` + regexp.QuoteMeta(label) + `\b\s*(?:\([^)]*\))?\s*[:\-]?\s*` + amountPat)
"zu zahlender betrag",
"rechnungsbetrag",
"rechnungsendbetrag",
"endbetrag",
"zahlbetrag",
"bruttobetrag",
"gesamtbetrag",
"gesamtsumme",
"betrag",
"total",
"summe",
} }
// amountRes matches "<label> [optional (comment)] [: -] <number>" for every // amountRes lists the payable-amount patterns in strict priority order: the
// label, in the same priority order as amountLabels. // first pattern with a usable amount wins, and a lower-priority label can never
var amountRes = func() []*regexp.Regexp { // override a higher-priority one ("zu zahlender betrag" > "rechnungsbetrag" >
res := make([]*regexp.Regexp, 0, len(amountLabels)) // "rechnungsendbetrag" > "gesamtbetrag" > "gesamtsumme (inkl. steuern)").
for _, label := range amountLabels { //
res = append(res, regexp.MustCompile( // "gesamtbetrag" and "gesamtsumme" are not in the original set but are the real
`(?i)\b`+regexp.QuoteMeta(label)+`\b\s*(?:\([^)]*\))?\s*[:\-]?\s*([0-9]+(?:[.,][0-9]+)*)`)) // labels on Diashop invoices ("Gesamtsumme (inkl. Steuern)"). The inclusive
} // variant is matched before a plain "gesamtsumme". Everything after those
return res // primary labels is the broader fallback set, consulted only when no primary
}() // label yields an amount. Within one pattern the last usable amount is taken,
// because totals usually come last.
var amountRes = []*regexp.Regexp{
amountRE("zu zahlender betrag"),
amountRE("rechnungsbetrag"),
amountRE("rechnungsendbetrag"),
amountRE("gesamtbetrag"),
regexp.MustCompile(`(?i)\bgesamtsumme\b\s*\(\s*inkl\.?\s*steuern\s*\)\s*[:\-]?\s*` + amountPat),
amountRE("gesamtsumme"),
amountRE("endbetrag"),
amountRE("zahlbetrag"),
amountRE("bruttobetrag"),
amountRE("betrag"),
amountRE("total"),
amountRE("summe"),
}
// taxLineRe marks a line whose number is a tax rate/percentage: an explicit
// percent sign or a VAT/tax keyword. "Steuern" (plural, as in "inkl. Steuern")
// is handled separately so the inclusive total stays usable.
var taxLineRe = regexp.MustCompile(`(?i)%|\bMwSt\b|\bUSt\b|\bProzent\b`)
// steuerRe finds "Steuer"/"Umsatzsteuer" etc. RE2 has no lookahead, so the
// plural "Steuern" is excluded in isTaxLine.
var steuerRe = regexp.MustCompile(`(?i)steuer`)
// percentAfterRe detects a percent sign directly after a number (spaces ok).
var percentAfterRe = regexp.MustCompile(`^\s*%`)
func main() { func main() {
if len(os.Args) < 2 { if len(os.Args) < 2 {
@@ -138,20 +154,116 @@ func pdfAmount(ctx context.Context, c *onlyoffice.Client, id string) (string, er
} }
// extractAmount returns the normalised ("1234.56") payable amount found in // extractAmount returns the normalised ("1234.56") payable amount found in
// text, or "" if no known label matches. // text, or "" if no usable amount matches.
//
// DKV invoices are special-cased first: they repeat a per-vehicle "TOTAL:" line
// and carry the real total only in the "Gesamtsummenaufstellung" section.
func extractAmount(text string) string { func extractAmount(text string) string {
if v, ok := dkvGrandTotal(text); ok {
return v
}
for _, re := range amountRes { for _, re := range amountRes {
ms := re.FindAllStringSubmatch(text, -1) if v, ok := lastUsableAmount(text, re); ok {
if len(ms) == 0 {
continue
}
if v, ok := normalizeAmount(ms[len(ms)-1][1]); ok {
return v return v
} }
} }
return "" return ""
} }
// dkvGrandTotal extracts the total of a DKV "Gesamtsummenaufstellung" section.
//
// Rule: DKV invoices repeat a per-vehicle "TOTAL:" line, so the last TOTAL is
// not the invoice total. When a "Gesamtsummenaufstellung" section exists, its
// total wins over every "TOTAL:" line: the first amount after the "»" marker,
// or, if there is none, the last amount in the section. The section ends at the
// page break (form feed) or end of text.
func dkvGrandTotal(text string) (string, bool) {
idx := strings.Index(strings.ToLower(text), "gesamtsummenaufstellung")
if idx < 0 {
return "", false
}
section := text[idx:]
if ff := strings.IndexByte(section, '\f'); ff >= 0 {
section = section[:ff]
}
if m := strings.Index(section, "»"); m >= 0 {
if v, ok := firstAmount(section[m:]); ok {
return v, true
}
}
return lastAmount(section)
}
// lastUsableAmount returns the last amount matched by re that is not a tax rate
// or percentage. Within one label the last usable amount wins.
func lastUsableAmount(text string, re *regexp.Regexp) (string, bool) {
ms := re.FindAllStringSubmatchIndex(text, -1)
for i := len(ms) - 1; i >= 0; i-- {
m := ms[i]
if isTaxRate(text, m[2], m[3]) {
continue
}
if v, ok := normalizeAmount(text[m[2]:m[3]]); ok {
return v, true
}
}
return "", false
}
// isTaxRate reports whether the number at text[start:end] is a tax rate or a
// percentage instead of a payable amount. A candidate is rejected when the
// token right after the number is "%" or the number's line carries a percent
// sign or a tax keyword. Rejecting is deliberate: office matching treats a
// known-but-different amount as a hard disqualifier, so an empty result is
// safer than the VAT rate.
func isTaxRate(text string, start, end int) bool {
if percentAfterRe.MatchString(text[end:]) {
return true
}
lineStart := strings.LastIndexByte(text[:start], '\n') + 1
line := text[lineStart:]
if n := strings.IndexByte(text[end:], '\n'); n >= 0 {
line = text[lineStart : end+n]
}
return isTaxLine(line)
}
// isTaxLine reports whether a line looks like a tax rate rather than a payable
// amount. "Steuern" is treated as a qualifier ("inkl. Steuern"), not a rate.
func isTaxLine(line string) bool {
if taxLineRe.MatchString(line) {
return true
}
for _, loc := range steuerRe.FindAllStringIndex(line, -1) {
if loc[1] >= len(line) || (line[loc[1]] != 'n' && line[loc[1]] != 'N') {
return true
}
}
return false
}
// numberRe finds bare numbers (with optional thousands/decimal separators).
var numberRe = regexp.MustCompile(`[0-9]+(?:[.,][0-9]+)*`)
func firstAmount(s string) (string, bool) {
for _, m := range numberRe.FindAllString(s, -1) {
if v, ok := normalizeAmount(m); ok {
return v, true
}
}
return "", false
}
func lastAmount(s string) (string, bool) {
ms := numberRe.FindAllString(s, -1)
for i := len(ms) - 1; i >= 0; i-- {
if v, ok := normalizeAmount(ms[i]); ok {
return v, true
}
}
return "", false
}
// normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into // normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into
// "1234.56". The rightmost separator is decimal only when followed by one or // "1234.56". The rightmost separator is decimal only when followed by one or
// two digits; otherwise every separator is a thousands separator. // two digits; otherwise every separator is a thousands separator.
+45
View File
@@ -76,6 +76,51 @@ func TestExtractAmount(t *testing.T) {
text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n", text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n",
want: "123.45", want: "123.45",
}, },
{
name: "dkv gesamtsummenaufstellung grand total after marker",
text: "» TOTAL: 111,11 100,00 11,11 111,11\n" +
"» TOTAL: 222,22 200,00 22,22 222,22\n" +
"Gesamtsummenaufstellung\n" +
"Netto 240,00\n" +
"MwSt 47,25\n" +
"» 287,25\n",
want: "287.25",
},
{
name: "dkv gesamtsummenaufstellung total on next line",
text: "» TOTAL: 111,11\nGesamtsummenaufstellung\n»\n287,25\n",
want: "287.25",
},
{
name: "tax rate with percent sign is not an amount",
text: "Betrag: 19,00 % MwSt",
want: "",
},
{
name: "mehrwertsteuer rate is not an amount",
text: "Gesamtsumme: 19,00% MwSt",
want: "",
},
{
name: "steuer word on the number line rejects it",
text: "Betrag: 2,83 Steuer",
want: "",
},
{
name: "rejected primary falls back to a usable label",
text: "Gesamtsumme: 19,00 % MwSt\nEndbetrag: 42,00",
want: "42.00",
},
{
name: "labeled zu zahlender betrag beats unlabeled larger number",
text: "unlabeled 999,99\nZu zahlender Betrag: 10,00",
want: "10.00",
},
{
name: "labeled zu zahlender betrag beats lower label larger number",
text: "Endbetrag: 999,99\nZu zahlender Betrag: 10,00",
want: "10.00",
},
{ {
name: "diashop style gesamtsumme with comment", name: "diashop style gesamtsumme with comment",
text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n", text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n",