From ac06d863630faf0cf653a40e4fb754042187fc1d Mon Sep 17 00:00:00 2001 From: Andriy Oblivantsev Date: Thu, 27 Aug 2026 23:40:07 +0100 Subject: [PATCH] feat(files): dedupe by stem|ext + oo projects files dedupe - FileDedupKey groups true duplicates (same name+extension) - put-md --replace deletes matching stem|ext only (not other formats) - oo projects files dedupe PROJECT_ID [--apply] [--cross] - Cross-folder mode prefers non-_trash folders, keeps newest file Co-authored-by: Cursor --- AGENTS.md | 2 +- README.md | 3 + cmd/oo/projects_files.go | 57 +++++++ files_dedupe.go | 337 +++++++++++++++++++++++++++++++++++++++ files_dedupe_test.go | 81 ++++++++++ files_stem.go | 8 +- 6 files changed, 484 insertions(+), 4 deletions(-) create mode 100644 files_dedupe.go create mode 100644 files_dedupe_test.go diff --git a/AGENTS.md b/AGENTS.md index 39ba441..83cc7e5 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -28,7 +28,7 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the - Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them. - Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason. - CLI commands follow **subject → verb** structure (`oo `), never `oo -`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/.go` for a genuinely new domain. -- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`. Use `oo docs put-md` (md→docx upload, `--replace` default upserts by stem) and `oo docs as-md` (download→OCR if needed→markdown). Local converters live in `internal/docpipe` (pandoc / ocrmypdf / pdftotext / tesseract+go-hocr). For layout-aware extracts: `oo docs hocr` (or `as-md --hocr`). OO UI may show `title`+`fileExst` as double extension (e.g. `.docx.docx`) — cosmetic server quirk. +- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`. Use `oo docs put-md` (md→docx upload, `--replace` default upserts by stem|ext) and `oo docs as-md` (download→OCR if needed→markdown). `oo projects files dedupe PROJECT_ID` reports/removes duplicate stem|ext copies (`--apply`, `--cross`). Local converters live in `internal/docpipe` (pandoc / ocrmypdf / pdftotext / tesseract+go-hocr). For layout-aware extracts: `oo docs hocr` (or `as-md --hocr`). OO UI may show `title`+`fileExst` as double extension (e.g. `.docx.docx`) — cosmetic server quirk. - Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command. - No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only. - Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`. diff --git a/README.md b/README.md index a288d56..ba1b0b7 100644 --- a/README.md +++ b/README.md @@ -611,6 +611,9 @@ oo projects files upload 33 ./notes.docx oo projects files download 12345 --to ./copy.docx oo projects files rename 12345 notes-v2.docx oo projects files delete 12345 +oo projects files dedupe 7 # dry-run duplicate report +oo projects files dedupe 7 --apply # remove older stem|ext copies per folder +oo projects files dedupe 7 --cross --apply # cross-folder; keep non-_trash # Agent document pipeline (md in git ↔ docx in OO; OCR scans) oo docs tools diff --git a/cmd/oo/projects_files.go b/cmd/oo/projects_files.go index f92a105..dad12db 100644 --- a/cmd/oo/projects_files.go +++ b/cmd/oo/projects_files.go @@ -24,6 +24,7 @@ func projectFilesCmd() *cobra.Command { cmd.AddCommand(prjFilesDownloadCmd()) cmd.AddCommand(prjFilesRenameCmd()) cmd.AddCommand(prjFilesDeleteCmd()) + cmd.AddCommand(prjFilesDedupeCmd()) // Convenience aliases into oo docs (md↔docx / OCR pipeline). cmd.AddCommand(aliasDocsAsMD()) cmd.AddCommand(aliasDocsPutMD()) @@ -201,6 +202,62 @@ func prjFilesDeleteCmd() *cobra.Command { } } +func prjFilesDedupeCmd() *cobra.Command { + var apply, cross bool + cmd := &cobra.Command{ + Use: "dedupe PROJECT_ID", + Short: "Find (and optionally remove) duplicate files in project Documents folders", + Long: `Duplicates share the same logical name: stem|ext (OO title+fileExst). + +Default: dry-run report. Pass --apply to delete older copies (keeps newest; --cross prefers non-trash folders).`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + c, err := newOO(cmd) + if err != nil { + return err + } + groups, deleted, err := c.DedupeProject(cmd.Context(), args[0], onlyoffice.DedupOptions{ + CrossFolder: cross, + }, apply) + if err != nil { + return err + } + rows := make([]map[string]any, 0, len(groups)) + for _, g := range groups { + row := map[string]any{ + "key": g.Key, + "folder_id": g.FolderID, + "folder_title": g.FolderTitle, + "keep_id": fileIDStr(g.Keep), + "keep_title": onlyoffice.FileEntryTitle(g.Keep), + "remove_count": len(g.Remove), + } + removeIDs := make([]string, 0, len(g.Remove)) + for _, f := range g.Remove { + removeIDs = append(removeIDs, fileIDStr(f)) + } + row["remove_ids"] = removeIDs + rows = append(rows, row) + } + out := map[string]any{ + "project_id": args[0], + "dry_run": !apply, + "cross": cross, + "groups": len(groups), + "duplicates": rows, + } + if apply { + out["deleted_ids"] = deleted + } + printObject(out) + return nil + }, + } + cmd.Flags().BoolVar(&apply, "apply", false, "delete duplicate files (default: report only)") + cmd.Flags().BoolVar(&cross, "cross", false, "also dedupe same stem|ext across folders (prefers non-_trash)") + return cmd +} + func fileEntryRows(files []*onlyoffice.FileEntry) []map[string]any { rows := make([]map[string]any, 0, len(files)) for _, f := range files { diff --git a/files_dedupe.go b/files_dedupe.go new file mode 100644 index 0000000..c0161a2 --- /dev/null +++ b/files_dedupe.go @@ -0,0 +1,337 @@ +package onlyoffice + +import ( + "context" + "path/filepath" + "sort" + "strings" +) + +// FileEntryExt returns a normalized extension (lowercase, with leading dot). +func FileEntryExt(f *FileEntry) string { + if f == nil { + return "" + } + exst := "" + if f.FileExst != nil { + exst = strings.TrimSpace(*f.FileExst) + } + if exst != "" { + if !strings.HasPrefix(exst, ".") { + exst = "." + exst + } + return strings.ToLower(exst) + } + if f.Title != nil { + if ext := filepath.Ext(*f.Title); ext != "" { + return strings.ToLower(ext) + } + } + return "" +} + +// FileDedupKey is stem|ext — two files with the same key are duplicates. +func FileDedupKey(f *FileEntry) string { + st := FileEntryStem(f) + ext := FileEntryExt(f) + if st == "" { + return "" + } + if ext == "" { + return st + } + return st + "|" + strings.TrimPrefix(ext, ".") +} + +// FindFilesByDedupKey returns folder files matching stem and extension. +func FindFilesByDedupKey(files []*FileEntry, stem, ext string) []*FileEntry { + key := dedupKeyFromParts(stem, ext) + if key == "" { + return nil + } + var out []*FileEntry + for _, f := range files { + if FileDedupKey(f) == key { + out = append(out, f) + } + } + return out +} + +func dedupKeyFromParts(stem, ext string) string { + stem = strings.TrimSpace(stem) + if stem == "" { + return "" + } + ext = strings.ToLower(strings.TrimSpace(ext)) + if ext != "" && !strings.HasPrefix(ext, ".") { + ext = "." + ext + } + if ext == "" { + return stem + } + return stem + "|" + strings.TrimPrefix(ext, ".") +} + +// UploadExtFromLocal returns the lowercase extension from a local path. +func UploadExtFromLocal(localPath string) string { + ext := filepath.Ext(localPath) + if ext == "" { + return "" + } + return strings.ToLower(ext) +} + +// IsTrashFolderTitle reports staging/trash folders (e.g. _trash-md). +func IsTrashFolderTitle(title string) bool { + t := strings.ToLower(strings.TrimSpace(title)) + return strings.HasPrefix(t, "_") || strings.Contains(t, "trash") +} + +// ProjectFolderFile ties a file to its project Documents subfolder. +type ProjectFolderFile struct { + FolderID string + FolderTitle string + File *FileEntry +} + +// DedupGroup is one duplicate set: keep the newest (or non-trash) file. +type DedupGroup struct { + Key string + FolderID string + FolderTitle string + Keep *FileEntry + Remove []*FileEntry +} + +// DedupOptions controls project-wide duplicate scans. +type DedupOptions struct { + CrossFolder bool +} + +// FindProjectDuplicates scans project folders for duplicate files. +func FindProjectDuplicates(folders []*FolderEntry, filesByFolder map[string][]*FileEntry, opts DedupOptions) []DedupGroup { + var indexed []ProjectFolderFile + for _, folder := range folders { + if folder == nil || folder.ID == nil { + continue + } + fid := folder.ID.String() + title := "" + if folder.Title != nil { + title = *folder.Title + } + for _, f := range filesByFolder[fid] { + if f == nil { + continue + } + indexed = append(indexed, ProjectFolderFile{ + FolderID: fid, FolderTitle: title, File: f, + }) + } + } + if opts.CrossFolder { + return findCrossFolderDuplicates(indexed) + } + return findWithinFolderDuplicates(indexed) +} + +func findWithinFolderDuplicates(indexed []ProjectFolderFile) []DedupGroup { + byFolder := map[string][]ProjectFolderFile{} + for _, it := range indexed { + byFolder[it.FolderID] = append(byFolder[it.FolderID], it) + } + var out []DedupGroup + for fid, items := range byFolder { + title := "" + if len(items) > 0 { + title = items[0].FolderTitle + } + byKey := map[string][]*FileEntry{} + for _, it := range items { + k := FileDedupKey(it.File) + byKey[k] = append(byKey[k], it.File) + } + for k, group := range byKey { + if len(group) < 2 { + continue + } + keep, remove := pickDuplicateKeeper(group, false) + if keep == nil || len(remove) == 0 { + continue + } + out = append(out, DedupGroup{ + Key: k, FolderID: fid, FolderTitle: title, Keep: keep, Remove: remove, + }) + } + } + sortDedupGroups(out) + return out +} + +func findCrossFolderDuplicates(indexed []ProjectFolderFile) []DedupGroup { + byKey := map[string][]ProjectFolderFile{} + for _, it := range indexed { + k := FileDedupKey(it.File) + byKey[k] = append(byKey[k], it) + } + var out []DedupGroup + for k, items := range byKey { + if len(items) < 2 { + continue + } + files := make([]*FileEntry, len(items)) + folders := make([]string, len(items)) + folderTitles := make([]string, len(items)) + for i, it := range items { + files[i] = it.File + folders[i] = it.FolderID + folderTitles[i] = it.FolderTitle + } + keep, remove := pickDuplicateKeeperWithFolders(files, folders, folderTitles) + if keep == nil || len(remove) == 0 { + continue + } + fid, ftitle := "", "" + for _, it := range items { + if it.File == keep { + fid, ftitle = it.FolderID, it.FolderTitle + break + } + } + out = append(out, DedupGroup{ + Key: k, FolderID: fid, FolderTitle: ftitle, Keep: keep, Remove: remove, + }) + } + sortDedupGroups(out) + return out +} + +func pickDuplicateKeeper(files []*FileEntry, _ bool) (*FileEntry, []*FileEntry) { + return pickDuplicateKeeperWithFolders(files, nil, nil) +} + +func pickDuplicateKeeperWithFolders(files []*FileEntry, folderIDs, folderTitles []string) (*FileEntry, []*FileEntry) { + if len(files) == 0 { + return nil, nil + } + type ranked struct { + file *FileEntry + trash bool + } + rankedFiles := make([]ranked, len(files)) + for i, f := range files { + trash := false + if folderTitles != nil && i < len(folderTitles) { + trash = IsTrashFolderTitle(folderTitles[i]) + } + rankedFiles[i] = ranked{file: f, trash: trash} + } + sort.SliceStable(rankedFiles, func(i, j int) bool { + ri, rj := rankedFiles[i], rankedFiles[j] + if ri.trash != rj.trash { + return !ri.trash // non-trash first + } + ti, tj := rankedFiles[i].file.Updated, rankedFiles[j].file.Updated + if ti == nil { + return false + } + if tj == nil { + return true + } + return ti.After(*tj) // newest first + }) + keep := rankedFiles[0].file + var remove []*FileEntry + for _, r := range rankedFiles[1:] { + remove = append(remove, r.file) + } + return keep, remove +} + +func sortDedupGroups(groups []DedupGroup) { + sort.Slice(groups, func(i, j int) bool { + if groups[i].FolderTitle != groups[j].FolderTitle { + return groups[i].FolderTitle < groups[j].FolderTitle + } + return groups[i].Key < groups[j].Key + }) +} + +// ApplyDedupGroups deletes Remove files from each group. +func (c *Client) ApplyDedupGroups(ctx context.Context, groups []DedupGroup) ([]int, error) { + seen := map[int]struct{}{} + var ids []int + for _, g := range groups { + for _, f := range g.Remove { + n := int(FileEntryNumericID(f)) + if n == 0 { + continue + } + if _, ok := seen[n]; ok { + continue + } + seen[n] = struct{}{} + ids = append(ids, n) + } + } + if len(ids) == 0 { + return nil, nil + } + if err := c.DeleteFiles(ctx, ids); err != nil { + return ids, err + } + return ids, nil +} + +// DeleteFilesByDedupKey removes all files in folderID matching stem+ext. +func (c *Client) DeleteFilesByDedupKey(ctx context.Context, folderID, stem, ext string) ([]int, error) { + files, err := c.FolderFiles(ctx, folderID) + if err != nil { + return nil, err + } + matches := FindFilesByDedupKey(files, stem, ext) + if len(matches) == 0 { + return nil, nil + } + ids := make([]int, 0, len(matches)) + for _, f := range matches { + n := int(FileEntryNumericID(f)) + if n != 0 { + ids = append(ids, n) + } + } + if len(ids) == 0 { + return nil, nil + } + if err := c.DeleteFiles(ctx, ids); err != nil { + return nil, err + } + return ids, nil +} + +// DedupeProject scans project folders and optionally deletes duplicates. +func (c *Client) DedupeProject(ctx context.Context, projectID string, opts DedupOptions, apply bool) ([]DedupGroup, []int, error) { + pf, err := c.GetProjectFiles(ctx, projectID) + if err != nil { + return nil, nil, err + } + filesByFolder := make(map[string][]*FileEntry, len(pf.Folders)) + for _, folder := range pf.Folders { + if folder == nil || folder.ID == nil { + continue + } + fid := folder.ID.String() + files, err := c.FolderFiles(ctx, fid) + if err != nil { + return nil, nil, err + } + filesByFolder[fid] = files + } + groups := FindProjectDuplicates(pf.Folders, filesByFolder, opts) + if !apply || len(groups) == 0 { + return groups, nil, nil + } + deleted, err := c.ApplyDedupGroups(ctx, groups) + return groups, deleted, err +} diff --git a/files_dedupe_test.go b/files_dedupe_test.go new file mode 100644 index 0000000..87a51d1 --- /dev/null +++ b/files_dedupe_test.go @@ -0,0 +1,81 @@ +package onlyoffice + +import ( + "encoding/json" + "testing" + "time" +) + +func TestFileDedupKey(t *testing.T) { + title := "OO-HONDA-7-INDEX.docx" + exst := ".docx" + f := &FileEntry{Title: &title, FileExst: &exst} + if got := FileDedupKey(f); got != "OO-HONDA-7-INDEX|docx" { + t.Fatalf("got %q", got) + } +} + +func TestFindFilesByDedupKey(t *testing.T) { + a := &FileEntry{Title: strPtr("foo.docx"), FileExst: strPtr(".docx")} + b := &FileEntry{Title: strPtr("foo.md"), FileExst: strPtr(".md")} + files := []*FileEntry{a, b} + got := FindFilesByDedupKey(files, "foo", ".docx") + if len(got) != 1 || got[0] != a { + t.Fatalf("got %+v", got) + } +} + +func TestFindWithinFolderDuplicates(t *testing.T) { + t1 := time.Date(2026, 8, 27, 16, 0, 0, 0, time.UTC) + t2 := t1.Add(time.Hour) + old := &FileEntry{ID: jsonNum("1"), Title: strPtr("idx.docx"), FileExst: strPtr(".docx"), Updated: &t1} + new := &FileEntry{ID: jsonNum("2"), Title: strPtr("idx.docx"), FileExst: strPtr(".docx"), Updated: &t2} + indexed := []ProjectFolderFile{ + {FolderID: "490", FolderTitle: "00-Index", File: old}, + {FolderID: "490", FolderTitle: "00-Index", File: new}, + } + groups := findWithinFolderDuplicates(indexed) + if len(groups) != 1 { + t.Fatalf("groups=%d", len(groups)) + } + if FileEntryNumericID(groups[0].Keep) != 2 { + t.Fatalf("keep id=%d", FileEntryNumericID(groups[0].Keep)) + } + if len(groups[0].Remove) != 1 || FileEntryNumericID(groups[0].Remove[0]) != 1 { + t.Fatalf("remove=%v", groups[0].Remove) + } +} + +func TestCrossFolderPrefersNonTrash(t *testing.T) { + t1 := time.Date(2026, 8, 27, 18, 0, 0, 0, time.UTC) + t2 := t1.Add(-time.Hour) + trash := &FileEntry{ID: jsonNum("10"), Title: strPtr("INDEX.md"), FileExst: strPtr(".md"), Updated: &t1} + good := &FileEntry{ID: jsonNum("20"), Title: strPtr("INDEX.md"), FileExst: strPtr(".md"), Updated: &t2} + indexed := []ProjectFolderFile{ + {FolderID: "493", FolderTitle: "_trash-md", File: trash}, + {FolderID: "492", FolderTitle: "OCR", File: good}, + } + groups := findCrossFolderDuplicates(indexed) + if len(groups) != 1 { + t.Fatalf("groups=%d", len(groups)) + } + if FileEntryNumericID(groups[0].Keep) != 20 { + t.Fatalf("keep id=%d", FileEntryNumericID(groups[0].Keep)) + } +} + +func TestIsTrashFolderTitle(t *testing.T) { + if !IsTrashFolderTitle("_trash-md") { + t.Fatal("expected trash") + } + if IsTrashFolderTitle("00-Index") { + t.Fatal("expected not trash") + } +} + +func strPtr(s string) *string { return &s } + +func jsonNum(s string) *json.Number { + n := json.Number(s) + return &n +} diff --git a/files_stem.go b/files_stem.go index 18eac89..b9552a6 100644 --- a/files_stem.go +++ b/files_stem.go @@ -89,7 +89,8 @@ func FindFilesByStem(files []*FileEntry, stem string) []*FileEntry { return out } -// DeleteFilesByStem removes all files in folderID matching stem (for put-md upsert). +// DeleteFilesByStem removes all files in folderID matching stem (any extension). +// Prefer DeleteFilesByDedupKey when the upload extension is known. func (c *Client) DeleteFilesByStem(ctx context.Context, folderID, stem string) ([]int, error) { files, err := c.FolderFiles(ctx, folderID) if err != nil { @@ -115,10 +116,11 @@ func (c *Client) DeleteFilesByStem(ctx context.Context, folderID, stem string) ( return ids, nil } -// UploadToFolderReplacing deletes same-stem files then uploads localPath. +// UploadToFolderReplacing deletes same stem+ext files then uploads localPath. func (c *Client) UploadToFolderReplacing(ctx context.Context, folderID, localPath string) (*FileEntry, []int, error) { stem := UploadStemFromLocal(localPath) - deleted, err := c.DeleteFilesByStem(ctx, folderID, stem) + ext := UploadExtFromLocal(localPath) + deleted, err := c.DeleteFilesByDedupKey(ctx, folderID, stem, ext) if err != nil { return nil, deleted, err }