Files
go-onlyoffice/catalog/names.go
T
eSlider 82da45dc4d
Release Please / Release Please (push) Skipped
Release / GoReleaser (push) Skipped
Tests / Secret scan (gitleaks) (push) Skipped
Tests / Test (Go 1.25) (push) Skipped
Tests / Test (Go stable) (push) Skipped
Tests / Secret scan (gitleaks) (pull_request) Successful in 5s
Tests / Test (Go 1.25) (pull_request) Successful in 1m16s
Tests / Test (Go stable) (pull_request) Successful in 1m36s
feat: upstream generic workspace tooling (board-sync, crm audit, catalog names)
Promote the generic, reusable parts of the private oo-workspace into the
public library/CLI, so oo-workspace can shrink to business glue.

- board.go: Board types + (c *Client) SyncBoard — upsert project
  milestones/tasks by exact title from a YAML board (dry-run when
  apply=false). CLI: oo projects board-sync.
- crm_audit.go: OpportunityAudit + (c *Client) AuditOpportunities —
  file/task/member counts with a generic class (ok|dup|empty|junk-title).
  CLI: oo crm audit [--out].
- catalog/names.go: CleanPersonNames / GuessNameFromEmail /
  FormatProjectTitle (ported from oo-workspace).
- catalog: adopt the newer apply/match logic — name repair (never encode
  company in lastName), company grouping key, contact-info type
  normalization, preserve an already-applied oo_id. Keeps the local
  Address/OOProjects fields and the config-driven classifier.

Business-only oo-workspace commands (crm clean/migrate, folders, search)
deliberately stay private.
2026-09-22 23:13:40 +01:00

153 lines
4.1 KiB
Go
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
package catalog
import (
"regexp"
"strings"
"unicode"
)
var (
parenSuffixRE = regexp.MustCompile(`(?i)\s*[\(\[\{][^)\]\}]*[\)\]\}]\s*$`)
dashCompanyRE = regexp.MustCompile(`(?i)\s+[-–—]\s+[A-Za-z0-9].*$`)
emailLocalRE = regexp.MustCompile(`(?i)^[a-z0-9._%+\-]+@[a-z0-9.\-]+\.[a-z]{2,}$`)
nonNameTokenRE = regexp.MustCompile(`[^a-zA-ZÀ-öø-ÿĀ-ž0-9'’.\-]+`)
)
// CleanPersonNames strips company annotations from display names and fills
// first/last from the email local-part when the source used an address as the
// name. Company affiliation belongs on Org / the CRM companyId — never in LastName.
func CleanPersonNames(first, last, display, org string, emails []string) (cleanFirst, cleanLast string) {
first = strings.TrimSpace(first)
last = strings.TrimSpace(last)
display = strings.TrimSpace(display)
org = strings.TrimSpace(org)
if looksLikeEmail(first) {
ef, el := GuessNameFromEmail(first)
first, last = ef, el
}
if looksLikeEmail(display) && first == "" && last == "" {
display = ""
}
if first == "" && last == "" && display != "" {
first, last = SplitDisplayName(display)
}
first = stripCompanyAnnotation(first, org)
last = stripCompanyAnnotation(last, org)
// "Smith - Acme" / "Jones (Acme)" landed in last.
last = stripCompanyAnnotation(last, org)
if i := strings.IndexAny(first, "(["); i > 0 {
first = strings.TrimSpace(first[:i])
}
// Entire last name is just the company (e.g. last="Acme").
if org != "" && personLastIsOrg(last, org) {
last = ""
}
if (first == "" || looksLikeEmail(first)) && len(emails) > 0 {
ef, el := GuessNameFromEmail(emails[0])
if first == "" || looksLikeEmail(first) {
first = ef
}
if last == "" || last == "-" {
last = el
}
}
first = strings.TrimSpace(first)
last = strings.TrimSpace(last)
if last == "" {
last = "-"
}
return first, last
}
func stripCompanyAnnotation(s, org string) string {
s = strings.TrimSpace(s)
if s == "" {
return ""
}
s = parenSuffixRE.ReplaceAllString(s, "")
s = strings.TrimSpace(s)
s = dashCompanyRE.ReplaceAllString(s, "")
s = strings.TrimSpace(s)
if org != "" {
for _, sep := range []string{" - ", " – ", " — ", " / "} {
if i := strings.LastIndex(strings.ToLower(s), strings.ToLower(sep+org)); i >= 0 {
s = strings.TrimSpace(s[:i])
}
}
suf := " (" + org + ")"
if strings.HasSuffix(strings.ToLower(s), strings.ToLower(suf)) {
s = strings.TrimSpace(s[:len(s)-len(suf)])
}
}
return strings.TrimSpace(s)
}
func personLastIsOrg(last, org string) bool {
last = NormalizeName(last)
org = NormalizeName(org)
if last == "" || org == "" {
return false
}
if last == org {
return true
}
// "Acme" vs "Acme GmbH & Co. KG"
return strings.HasPrefix(org, last+" ") || strings.HasPrefix(org, last+",")
}
func looksLikeEmail(s string) bool {
return emailLocalRE.MatchString(strings.TrimSpace(s))
}
// GuessNameFromEmail turns local@domain into Title-Case first/last when the
// local part looks like first.last / first_last / first-last.
func GuessNameFromEmail(email string) (first, last string) {
email = NormalizeEmail(email)
local, _, ok := strings.Cut(email, "@")
if !ok || local == "" {
return "", ""
}
local = strings.Split(local, "+")[0]
parts := strings.FieldsFunc(local, func(r rune) bool {
return r == '.' || r == '_' || r == '-'
})
if len(parts) == 0 {
return titleToken(local), ""
}
if len(parts) == 1 {
return titleToken(parts[0]), ""
}
return titleToken(parts[0]), titleToken(strings.Join(parts[1:], " "))
}
func titleToken(s string) string {
s = nonNameTokenRE.ReplaceAllString(s, " ")
s = strings.TrimSpace(s)
if s == "" {
return ""
}
runes := []rune(strings.ToLower(s))
runes[0] = unicode.ToTitle(runes[0])
return string(runes)
}
// FormatProjectTitle builds "CC | Company | Title" (spaces around |).
// Country should be a short code (DE, TF, UA, …). Empty segments are dropped.
func FormatProjectTitle(country, company, title string) string {
parts := make([]string, 0, 3)
for _, p := range []string{country, company, title} {
p = strings.TrimSpace(p)
p = strings.ReplaceAll(p, "|", "/")
if p != "" {
parts = append(parts, p)
}
}
return strings.Join(parts, " | ")
}