fix(pdfamount): ставка НДС не сумма; итог DKV (#29) #33
+146
-34
@@ -18,36 +18,52 @@ import (
|
|||||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||||
)
|
)
|
||||||
|
|
||||||
// amountLabels are the payable-amount labels in priority order: the first
|
// amountPat is the amount capture shared by every amount regex.
|
||||||
// label present in a document wins. Within one label the last amount is taken,
|
const amountPat = `([0-9]+(?:[.,][0-9]+)*)`
|
||||||
// because totals usually come last.
|
|
||||||
//
|
// amountRE builds "<label> [optional (comment)] [: -] <number>".
|
||||||
// "gesamtsumme" is not in the original list but is the real label on Diashop
|
func amountRE(label string) *regexp.Regexp {
|
||||||
// invoices: "Gesamtsumme (inkl. Steuern)".
|
return regexp.MustCompile(
|
||||||
var amountLabels = []string{
|
`(?i)\b` + regexp.QuoteMeta(label) + `\b\s*(?:\([^)]*\))?\s*[:\-]?\s*` + amountPat)
|
||||||
"zu zahlender betrag",
|
|
||||||
"rechnungsbetrag",
|
|
||||||
"rechnungsendbetrag",
|
|
||||||
"endbetrag",
|
|
||||||
"zahlbetrag",
|
|
||||||
"bruttobetrag",
|
|
||||||
"gesamtbetrag",
|
|
||||||
"gesamtsumme",
|
|
||||||
"betrag",
|
|
||||||
"total",
|
|
||||||
"summe",
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// amountRes matches "<label> [optional (comment)] [: -] <number>" for every
|
// amountRes lists the payable-amount patterns in strict priority order: the
|
||||||
// label, in the same priority order as amountLabels.
|
// first pattern with a usable amount wins, and a lower-priority label can never
|
||||||
var amountRes = func() []*regexp.Regexp {
|
// override a higher-priority one ("zu zahlender betrag" > "rechnungsbetrag" >
|
||||||
res := make([]*regexp.Regexp, 0, len(amountLabels))
|
// "rechnungsendbetrag" > "gesamtbetrag" > "gesamtsumme (inkl. steuern)").
|
||||||
for _, label := range amountLabels {
|
//
|
||||||
res = append(res, regexp.MustCompile(
|
// "gesamtbetrag" and "gesamtsumme" are not in the original set but are the real
|
||||||
`(?i)\b`+regexp.QuoteMeta(label)+`\b\s*(?:\([^)]*\))?\s*[:\-]?\s*([0-9]+(?:[.,][0-9]+)*)`))
|
// labels on Diashop invoices ("Gesamtsumme (inkl. Steuern)"). The inclusive
|
||||||
}
|
// variant is matched before a plain "gesamtsumme". Everything after those
|
||||||
return res
|
// primary labels is the broader fallback set, consulted only when no primary
|
||||||
}()
|
// label yields an amount. Within one pattern the last usable amount is taken,
|
||||||
|
// because totals usually come last.
|
||||||
|
var amountRes = []*regexp.Regexp{
|
||||||
|
amountRE("zu zahlender betrag"),
|
||||||
|
amountRE("rechnungsbetrag"),
|
||||||
|
amountRE("rechnungsendbetrag"),
|
||||||
|
amountRE("gesamtbetrag"),
|
||||||
|
regexp.MustCompile(`(?i)\bgesamtsumme\b\s*\(\s*inkl\.?\s*steuern\s*\)\s*[:\-]?\s*` + amountPat),
|
||||||
|
amountRE("gesamtsumme"),
|
||||||
|
amountRE("endbetrag"),
|
||||||
|
amountRE("zahlbetrag"),
|
||||||
|
amountRE("bruttobetrag"),
|
||||||
|
amountRE("betrag"),
|
||||||
|
amountRE("total"),
|
||||||
|
amountRE("summe"),
|
||||||
|
}
|
||||||
|
|
||||||
|
// taxLineRe marks a line whose number is a tax rate/percentage: an explicit
|
||||||
|
// percent sign or a VAT/tax keyword. "Steuern" (plural, as in "inkl. Steuern")
|
||||||
|
// is handled separately so the inclusive total stays usable.
|
||||||
|
var taxLineRe = regexp.MustCompile(`(?i)%|\bMwSt\b|\bUSt\b|\bProzent\b`)
|
||||||
|
|
||||||
|
// steuerRe finds "Steuer"/"Umsatzsteuer" etc. RE2 has no lookahead, so the
|
||||||
|
// plural "Steuern" is excluded in isTaxLine.
|
||||||
|
var steuerRe = regexp.MustCompile(`(?i)steuer`)
|
||||||
|
|
||||||
|
// percentAfterRe detects a percent sign directly after a number (spaces ok).
|
||||||
|
var percentAfterRe = regexp.MustCompile(`^\s*%`)
|
||||||
|
|
||||||
func main() {
|
func main() {
|
||||||
if len(os.Args) < 2 {
|
if len(os.Args) < 2 {
|
||||||
@@ -138,20 +154,116 @@ func pdfAmount(ctx context.Context, c *onlyoffice.Client, id string) (string, er
|
|||||||
}
|
}
|
||||||
|
|
||||||
// extractAmount returns the normalised ("1234.56") payable amount found in
|
// extractAmount returns the normalised ("1234.56") payable amount found in
|
||||||
// text, or "" if no known label matches.
|
// text, or "" if no usable amount matches.
|
||||||
|
//
|
||||||
|
// DKV invoices are special-cased first: they repeat a per-vehicle "TOTAL:" line
|
||||||
|
// and carry the real total only in the "Gesamtsummenaufstellung" section.
|
||||||
func extractAmount(text string) string {
|
func extractAmount(text string) string {
|
||||||
|
if v, ok := dkvGrandTotal(text); ok {
|
||||||
|
return v
|
||||||
|
}
|
||||||
for _, re := range amountRes {
|
for _, re := range amountRes {
|
||||||
ms := re.FindAllStringSubmatch(text, -1)
|
if v, ok := lastUsableAmount(text, re); ok {
|
||||||
if len(ms) == 0 {
|
|
||||||
continue
|
|
||||||
}
|
|
||||||
if v, ok := normalizeAmount(ms[len(ms)-1][1]); ok {
|
|
||||||
return v
|
return v
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return ""
|
return ""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// dkvGrandTotal extracts the total of a DKV "Gesamtsummenaufstellung" section.
|
||||||
|
//
|
||||||
|
// Rule: DKV invoices repeat a per-vehicle "TOTAL:" line, so the last TOTAL is
|
||||||
|
// not the invoice total. When a "Gesamtsummenaufstellung" section exists, its
|
||||||
|
// total wins over every "TOTAL:" line: the first amount after the "»" marker,
|
||||||
|
// or, if there is none, the last amount in the section. The section ends at the
|
||||||
|
// page break (form feed) or end of text.
|
||||||
|
func dkvGrandTotal(text string) (string, bool) {
|
||||||
|
idx := strings.Index(strings.ToLower(text), "gesamtsummenaufstellung")
|
||||||
|
if idx < 0 {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
section := text[idx:]
|
||||||
|
if ff := strings.IndexByte(section, '\f'); ff >= 0 {
|
||||||
|
section = section[:ff]
|
||||||
|
}
|
||||||
|
if m := strings.Index(section, "»"); m >= 0 {
|
||||||
|
if v, ok := firstAmount(section[m:]); ok {
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return lastAmount(section)
|
||||||
|
}
|
||||||
|
|
||||||
|
// lastUsableAmount returns the last amount matched by re that is not a tax rate
|
||||||
|
// or percentage. Within one label the last usable amount wins.
|
||||||
|
func lastUsableAmount(text string, re *regexp.Regexp) (string, bool) {
|
||||||
|
ms := re.FindAllStringSubmatchIndex(text, -1)
|
||||||
|
for i := len(ms) - 1; i >= 0; i-- {
|
||||||
|
m := ms[i]
|
||||||
|
if isTaxRate(text, m[2], m[3]) {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if v, ok := normalizeAmount(text[m[2]:m[3]]); ok {
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
// isTaxRate reports whether the number at text[start:end] is a tax rate or a
|
||||||
|
// percentage instead of a payable amount. A candidate is rejected when the
|
||||||
|
// token right after the number is "%" or the number's line carries a percent
|
||||||
|
// sign or a tax keyword. Rejecting is deliberate: office matching treats a
|
||||||
|
// known-but-different amount as a hard disqualifier, so an empty result is
|
||||||
|
// safer than the VAT rate.
|
||||||
|
func isTaxRate(text string, start, end int) bool {
|
||||||
|
if percentAfterRe.MatchString(text[end:]) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
lineStart := strings.LastIndexByte(text[:start], '\n') + 1
|
||||||
|
line := text[lineStart:]
|
||||||
|
if n := strings.IndexByte(text[end:], '\n'); n >= 0 {
|
||||||
|
line = text[lineStart : end+n]
|
||||||
|
}
|
||||||
|
return isTaxLine(line)
|
||||||
|
}
|
||||||
|
|
||||||
|
// isTaxLine reports whether a line looks like a tax rate rather than a payable
|
||||||
|
// amount. "Steuern" is treated as a qualifier ("inkl. Steuern"), not a rate.
|
||||||
|
func isTaxLine(line string) bool {
|
||||||
|
if taxLineRe.MatchString(line) {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
for _, loc := range steuerRe.FindAllStringIndex(line, -1) {
|
||||||
|
if loc[1] >= len(line) || (line[loc[1]] != 'n' && line[loc[1]] != 'N') {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
|
// numberRe finds bare numbers (with optional thousands/decimal separators).
|
||||||
|
var numberRe = regexp.MustCompile(`[0-9]+(?:[.,][0-9]+)*`)
|
||||||
|
|
||||||
|
func firstAmount(s string) (string, bool) {
|
||||||
|
for _, m := range numberRe.FindAllString(s, -1) {
|
||||||
|
if v, ok := normalizeAmount(m); ok {
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
|
func lastAmount(s string) (string, bool) {
|
||||||
|
ms := numberRe.FindAllString(s, -1)
|
||||||
|
for i := len(ms) - 1; i >= 0; i-- {
|
||||||
|
if v, ok := normalizeAmount(ms[i]); ok {
|
||||||
|
return v, true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
|
||||||
// normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into
|
// normalizeAmount turns "1.234,56" (DE), "1,234.56" (EN) or "1234.56" into
|
||||||
// "1234.56". The rightmost separator is decimal only when followed by one or
|
// "1234.56". The rightmost separator is decimal only when followed by one or
|
||||||
// two digits; otherwise every separator is a thousands separator.
|
// two digits; otherwise every separator is a thousands separator.
|
||||||
|
|||||||
@@ -76,6 +76,51 @@ func TestExtractAmount(t *testing.T) {
|
|||||||
text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n",
|
text: "Kundenbezogene Daten\n» TOTAL: 123,45 100,00 23,45 123,45\n",
|
||||||
want: "123.45",
|
want: "123.45",
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
name: "dkv gesamtsummenaufstellung grand total after marker",
|
||||||
|
text: "» TOTAL: 111,11 100,00 11,11 111,11\n" +
|
||||||
|
"» TOTAL: 222,22 200,00 22,22 222,22\n" +
|
||||||
|
"Gesamtsummenaufstellung\n" +
|
||||||
|
"Netto 240,00\n" +
|
||||||
|
"MwSt 47,25\n" +
|
||||||
|
"» 287,25\n",
|
||||||
|
want: "287.25",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "dkv gesamtsummenaufstellung total on next line",
|
||||||
|
text: "» TOTAL: 111,11\nGesamtsummenaufstellung\n»\n287,25\n",
|
||||||
|
want: "287.25",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "tax rate with percent sign is not an amount",
|
||||||
|
text: "Betrag: 19,00 % MwSt",
|
||||||
|
want: "",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "mehrwertsteuer rate is not an amount",
|
||||||
|
text: "Gesamtsumme: 19,00% MwSt",
|
||||||
|
want: "",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "steuer word on the number line rejects it",
|
||||||
|
text: "Betrag: 2,83 Steuer",
|
||||||
|
want: "",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "rejected primary falls back to a usable label",
|
||||||
|
text: "Gesamtsumme: 19,00 % MwSt\nEndbetrag: 42,00",
|
||||||
|
want: "42.00",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "labeled zu zahlender betrag beats unlabeled larger number",
|
||||||
|
text: "unlabeled 999,99\nZu zahlender Betrag: 10,00",
|
||||||
|
want: "10.00",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "labeled zu zahlender betrag beats lower label larger number",
|
||||||
|
text: "Endbetrag: 999,99\nZu zahlender Betrag: 10,00",
|
||||||
|
want: "10.00",
|
||||||
|
},
|
||||||
{
|
{
|
||||||
name: "diashop style gesamtsumme with comment",
|
name: "diashop style gesamtsumme with comment",
|
||||||
text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n",
|
text: "Zwischensumme\n12,34 €\nZwischensumme\n12,34 €\nVersand & Bearbeitung\n4,95 €\nGesamtsumme (inkl. Steuern)\n17,29 €\n",
|
||||||
|
|||||||
Reference in New Issue
Block a user