fix(pdfamount): не считать ставку НДС суммой; итог DKV (#29)
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 20s
Tests / Test (Go stable) (pull_request) Successful in 23s
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 4s
Tests / Test (Go 1.25) (pull_request) Successful in 20s
Tests / Test (Go stable) (pull_request) Successful in 23s
- Строки с %/MwSt/USt/Prozent/Steuer больше не дают сумму (было: Storchen 19,00 % -> amount 19.00 и ложная отбраковка кандидатов в матчере). - Приоритет меток: zu zahlender betrag > rechnungsbetrag > rechnungsendbetrag > gesamtbetrag > gesamtsumme (inkl. Steuern); low-priority не перебивает. - DKV: секция Gesamtsummenaufstellung (значение после «»») важнее повторяющихся TOTAL-строк по машинам. - Тесты: 19,00 % не сумма; fallback на usable-метку; DKV-итог.
This commit is contained in:
+146
-34
@@ -18,36 +18,52 @@ import (
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
)
|
||||
|
||||
// amountLabels are the payable-amount labels in priority order: the first
|
||||
// label present in a document wins. Within one label the last amount is taken,
|
||||
// because totals usually come last.
|
||||
//
|
||||
// "gesamtsumme" is not in the original list but is the real label on Diashop
|
||||
// invoices: "Gesamtsumme (inkl. Steuern)".
|
||||
var amountLabels = []string{
|
||||
"zu zahlender betrag",
|
||||
"rechnungsbetrag",
|
||||
"rechnungsendbetrag",
|
||||
"endbetrag",
|
||||
"zahlbetrag",
|
||||
"bruttobetrag",
|
||||
"gesamtbetrag",
|
||||
"gesamtsumme",
|
||||
"betrag",
|
||||
"total",
|
||||
"summe",
|
||||
// amountPat is the amount capture shared by every amount regex.
|
||||
const amountPat = `([0-9]+(?:[.,][0-9]+)*)`
|
||||
|
||||
// amountRE builds "<label> [optional (comment)] [: -] <number>".
|
||||
func amountRE(label string) *regexp.Regexp {
|
||||
return regexp.MustCompile(
|
||||
`(?i)\b` + regexp.QuoteMeta(label) + `\b\s*(?:\([^)]*\))?\s*[:\-]?\s*` + amountPat)
|
||||
}
|
||||
|
||||
// amountRes matches "<label> [optional (comment)] [: -] <number>" for every
|
||||
// label, in the same priority order as amountLabels.
|
||||
var amountRes = func() []*regexp.Regexp {
|
||||
res := make([]*regexp.Regexp, 0, len(amountLabels))
|
||||
for _, label := range amountLabels {
|
||||
res = append(res, regexp.MustCompile(
|
||||
`(?i)\b`+regexp.QuoteMeta(label)+`\b\s*(?:\([^)]*\))?\s*[:\-]?\s*([0-9]+(?:[.,][0-9]+)*)`))
|
||||
}
|
||||
return res
|
||||
}()
|
||||
// amountRes lists the payable-amount patterns in strict priority order: the
|
||||
// first pattern with a usable amount wins, and a lower-priority label can never
|
||||
// override a higher-priority one ("zu zahlender betrag" > "rechnungsbetrag" >
|
||||
// "rechnungsendbetrag" > "gesamtbetrag" > "gesamtsumme (inkl. steuern)").
|
||||
//
|
||||
// "gesamtbetrag" and "gesamtsumme" are not in the original set but are the real
|
||||
// labels on Diashop invoices ("Gesamtsumme (inkl. Steuern)"). The inclusive
|
||||
// variant is matched before a plain "gesamtsumme". Everything after those
|
||||
// primary labels is the broader fallback set, consulted only when no primary
|
||||
// label yields an amount. Within one pattern the last usable amount is taken,
|
||||
// because totals usually come last.
|
||||
var amountRes = []*regexp.Regexp{
|
||||
amountRE("zu zahlender betrag"),
|
||||
amountRE("rechnungsbetrag"),
|
||||
amountRE("rechnungsendbetrag"),
|
||||
amountRE("gesamtbetrag"),
|
||||
regexp.MustCompile(`(?i)\bgesamtsumme\b\s*\(\s*inkl\.?\s*steuern\s*\)\s*[:\-]?\s*` + amountPat),
|
||||
amountRE("gesamtsumme"),
|
||||
amountRE("endbetrag"),
|
||||
amountRE("zahlbetrag"),
|
||||
amountRE("bruttobetrag"),
|
||||
amountRE("betrag"),
|
||||
amountRE("total"),
|
||||
amountRE("summe"),
|
||||
}
|
||||
|
||||
// taxLineRe marks a line whose number is a tax rate/percentage: an explicit
|
||||
// percent sign or a VAT/tax keyword. "Steuern" (plural, as in "inkl. Steuern")
|
||||
// is handled separately so the inclusive total stays usable.
|
||||
var taxLineRe = regexp.MustCompile(`(?i)%|\bMwSt\b|\bUSt\b|\bProzent\b`)
|
||||
|
||||
// steuerRe finds "Steuer"/"Umsatzsteuer" etc. RE2 has no lookahead, so the
|
||||
// plural "Steuern" is excluded in isTaxLine.
|
||||
var steuerRe = regexp.MustCompile(`(?i)steuer`)
|
||||
|
||||
// percentAfterRe detects a percent sign directly after a number (spaces ok).
|
||||
var percentAfterRe = regexp.MustCompile(`^\s*%`)
|
||||
|
||||
func main() {
|
||||
if len(os.Args) < 2 {
|
||||
@@ -138,20 +154,116 @@ func pdfAmount(ctx context.Context, c *onlyoffice.Client, id string) (string, er
|
||||
}
|
||||
|
||||
// extractAmount returns the normalised ("1234.56") payable amount found in
|
||||
// text, or "" if no known label matches.
|
||||
// text, or "" if no usable amount matches.
|
||||
//
|
||||
// DKV invoices are special-cased first: they repeat a per-vehicle "TOTAL:" line
|
||||
// and carry the real total only in the "Gesamtsummenaufstellung" section.
|
||||
func extractAmount(text string) string {
|
||||
for _, re := range amountRes {
|
||||
ms := re.FindAllStringSubmatch(text, -1)
|
||||
if len(ms) == 0 {
|
||||
continue
|
||||
if v, ok := dkvGrandTotal(text); ok {
|
||||
return v
|
||||
}
|
||||
if v, ok := normalizeAmount(ms[len(ms)-1][1]); ok {
|
||||
for _, re := range amountRes {
|
||||
if v, ok := lastUsableAmount(text, re); ok {
|
||||
return v
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// dkvGrandTotal extracts the total of a DKV "Gesamtsummenaufstellung" section.
|
||||
//
|
||||
// Rule: DKV invoices repeat a per-vehicle "TOTAL:" line, so the last TOTAL is
|
||||
// not the invoice total. When a "Gesamtsummenaufstellung" section exists, its
|
||||
// total wins over every "TOTAL:" line: the first amount after the "»" marker,
|
||||
// or, if there is none, the last amount in the section. The section ends at the
|
||||
// page break (form feed) or end of text.
|
||||
func dkvGrandTotal(text string) (string, bool) {
|
||||
idx := strings.Index(strings.ToLower(text), "gesamtsummenaufstellung")
|
||||
if idx < 0 {
|
||||
return "", false
|
||||
}
|
||||
section := text[idx:]
|
||||
if ff := strings.IndexByte(section, '\f'); ff >= 0 {
|
||||
section = section[:ff]
|
||||
}
|
||||
if m := strings.Index(section, "»"); m >= 0 {
|
||||
if v, ok := firstAmount(section[m:]); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return lastAmount(section)
|
||||
}
|
||||
|
||||
// lastUsableAmount returns the last amount matched by re that is not a tax rate
|
||||
// or percentage. Within one label the last usable amount wins.
|
||||
func lastUsableAmount(text string, re *regexp.Regexp) (string, bool) {
|
||||
ms := re.FindAllStringSubmatchIndex(text, -1)
|
||||
for i := len(ms) - 1; i >= 0; i-- {
|
||||
m := ms[i]
|
||||
if isTaxRate(text, m[2], m[3]) {
|
||||
continue
|
||||
}
|
||||
if v, ok := normalizeAmount(text[m[2]:m[3]]); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// isTaxRate reports whether the number at text[start:end] is a tax rate or a
|
||||
// percentage instead of a payable amount. A candidate is rejected when the
|
||||
// token right after the number is "%" or the number's line carries a percent
|
||||
// sign or a tax keyword. Rejecting is deliberate: office matching treats a
|
||||
// known-but-different amount as a hard disqualifier, so an empty result is
|
||||
// safer than the VAT rate.
|
||||
func isTaxRate(text string, start, end int) bool {
|
||||
if percentAfterRe.MatchString(text[end:]) {
|
||||
return true
|
||||
}
|
||||
lineStart := strings.LastIndexByte(text[:start], '\n') + 1
|
||||
line := text[lineStart:]
|
||||
if n := strings.IndexByte(text[end:], '\n'); n >= 0 {
|
||||
line = text[lineStart : end+n]
|
||||
}
|
||||
return isTaxLine(line)
|
||||
}
|
||||
|
||||
// isTaxLine reports whether a line looks like a tax rate rather than a payable
|
||||
// amount. "Steuern" is treated as a qualifier ("inkl. Steuern"), not a rate.
|
||||
func isTaxLine(line string) bool {
|
||||
if taxLineRe.MatchString(line) {
|
||||
return true
|
||||
}
|
||||
for _, loc := range steuerRe.FindAllStringIndex(line, -1) {
|
||||
if loc[1] >= len(line) || (line[loc[1]] != 'n' && line[loc[1]] != 'N') {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// numberRe finds bare numbers (with optional thousands/decimal separators).
|
||||
var numberRe = regexp.MustCompile(`[0-9]+(?:[.,][0-9]+)*`)
|
||||
|
||||
func firstAmount(s string) (string, bool) {
|
||||
for _, m := range numberRe.FindAllString(s, -1) {
|
||||
if v, ok := normalizeAmount(m); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
func lastAmount(s string) (string, bool) {
|
||||
ms := numberRe.FindAllString(s, -1)
|
||||
for i := len(ms) - 1; i >= 0; i-- {
|
||||
if v, ok := normalizeAmount(ms[i]); ok {
|
||||
return v, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into
|
||||
// "1234.56". The rightmost separator is decimal only when followed by one or
|
||||
// two digits; otherwise every separator is a thousands separator.
|
||||
|
||||
@@ -76,6 +76,51 @@ func TestExtractAmount(t *testing.T) {
|
||||
text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n",
|
||||
want: "123.45",
|
||||
},
|
||||
{
|
||||
name: "dkv gesamtsummenaufstellung grand total after marker",
|
||||
text: "» TOTAL: 111,11 100,00 11,11 111,11\n" +
|
||||
"» TOTAL: 222,22 200,00 22,22 222,22\n" +
|
||||
"Gesamtsummenaufstellung\n" +
|
||||
"Netto 240,00\n" +
|
||||
"MwSt 47,25\n" +
|
||||
"» 287,25\n",
|
||||
want: "287.25",
|
||||
},
|
||||
{
|
||||
name: "dkv gesamtsummenaufstellung total on next line",
|
||||
text: "» TOTAL: 111,11\nGesamtsummenaufstellung\n»\n287,25\n",
|
||||
want: "287.25",
|
||||
},
|
||||
{
|
||||
name: "tax rate with percent sign is not an amount",
|
||||
text: "Betrag: 19,00 % MwSt",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "mehrwertsteuer rate is not an amount",
|
||||
text: "Gesamtsumme: 19,00% MwSt",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "steuer word on the number line rejects it",
|
||||
text: "Betrag: 2,83 Steuer",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "rejected primary falls back to a usable label",
|
||||
text: "Gesamtsumme: 19,00 % MwSt\nEndbetrag: 42,00",
|
||||
want: "42.00",
|
||||
},
|
||||
{
|
||||
name: "labeled zu zahlender betrag beats unlabeled larger number",
|
||||
text: "unlabeled 999,99\nZu zahlender Betrag: 10,00",
|
||||
want: "10.00",
|
||||
},
|
||||
{
|
||||
name: "labeled zu zahlender betrag beats lower label larger number",
|
||||
text: "Endbetrag: 999,99\nZu zahlender Betrag: 10,00",
|
||||
want: "10.00",
|
||||
},
|
||||
{
|
||||
name: "diashop style gesamtsumme with comment",
|
||||
text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n",
|
||||
|
||||
Reference in New Issue
Block a user