diff --git a/AGENTS.md b/AGENTS.md index 497fb4b..46c0aa4 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -28,6 +28,7 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the - Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them. - Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason. - CLI commands follow **subject → verb** structure (`oo `), never `oo -`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/.go` for a genuinely new domain. +- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`. Use `oo docs put-md` (md→docx upload) and `oo docs as-md` (download→OCR if needed→markdown). Local converters live in `internal/docpipe` (pandoc / ocrmypdf / pdftotext). - Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command. - No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only. - Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`. diff --git a/cmd/oo/docs.go b/cmd/oo/docs.go new file mode 100644 index 0000000..426cb7a --- /dev/null +++ b/cmd/oo/docs.go @@ -0,0 +1,314 @@ +package main + +import ( + "fmt" + "os" + "path/filepath" + "strings" + + onlyoffice "github.com/eslider/go-onlyoffice" + "github.com/eslider/go-onlyoffice/internal/docpipe" + "github.com/spf13/cobra" +) + +func init() { + rootCmd.AddCommand(docsCmd()) +} + +func docsCmd() *cobra.Command { + cmd := &cobra.Command{ + Use: "docs", + Short: "Local document pipeline: md↔docx, OCR→PDF, extract Markdown", + Long: `Agent-friendly conversions (requires pandoc / ocrmypdf / pdftotext on PATH). + +OnlyOffice Documents UI is poor for .md — keep Markdown in git, store .docx in OO. +Upload Markdown as DOCX: oo docs put-md PROJECT_ID file.md +Read an OO file as MD: oo docs as-md FILE_ID +OCR a scan locally: oo docs ocr scan.pdf --md out.md`, + } + cmd.AddCommand(docsConvertCmd()) + cmd.AddCommand(docsOCRCmd()) + cmd.AddCommand(docsAsMDCmd()) + cmd.AddCommand(docsPutMDCmd()) + cmd.AddCommand(docsToolsCmd()) + return cmd +} + +func docsToolsCmd() *cobra.Command { + return &cobra.Command{ + Use: "tools", + Short: "Show which converter binaries are on PATH", + RunE: func(cmd *cobra.Command, args []string) error { + t := docpipe.LookPath() + printObject(map[string]any{ + "pandoc": strOrNil(t.Pandoc), + "ocrmypdf": strOrNil(t.OCRMyPDF), + "pdftotext": strOrNil(t.PDFToText), + "tesseract": strOrNil(t.Tesseract), + }) + return nil + }, + } +} + +func strOrNil(s string) any { + if s == "" { + return nil + } + return s +} + +func docsConvertCmd() *cobra.Command { + var to string + cmd := &cobra.Command{ + Use: "convert PATH", + Short: "Convert a local file with pandoc (md↔docx by default)", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + in := args[0] + t := docpipe.LookPath() + out := to + if out == "" { + switch docpipe.Ext(in) { + case ".md", ".markdown": + out = docpipe.SiblingDOCX(in) + case ".docx": + out = strings.TrimSuffix(in, docpipe.Ext(in)) + ".md" + default: + return fmt.Errorf("--to required for input type %s", docpipe.Ext(in)) + } + } + if err := docpipe.EnsureDir(out); err != nil { + return err + } + if err := t.ConvertFile(in, out); err != nil { + return err + } + printObject(map[string]any{"in": in, "out": out}) + return nil + }, + } + cmd.Flags().StringVar(&to, "to", "", "output path (default: sibling .docx or .md)") + return cmd +} + +func docsOCRCmd() *cobra.Command { + var out, mdOut, lang string + var force bool + var writeMD bool + cmd := &cobra.Command{ + Use: "ocr PATH", + Short: "OCR image/PDF → searchable PDF (and optional Markdown)", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + in := args[0] + t := docpipe.LookPath() + if out == "" { + base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in)) + out = filepath.Join(filepath.Dir(in), base+".ocr.pdf") + } + if err := docpipe.EnsureDir(out); err != nil { + return err + } + if err := t.OCRToPDF(in, out, force, lang); err != nil { + return err + } + res := map[string]any{"in": in, "pdf": out} + if writeMD || mdOut != "" { + if mdOut == "" { + mdOut = strings.TrimSuffix(out, filepath.Ext(out)) + ".md" + } + text, err := t.ExtractPDFText(out) + if err != nil { + return err + } + body := "# " + filepath.Base(in) + "\n\n" + strings.TrimSpace(text) + "\n" + if err := os.WriteFile(mdOut, []byte(body), 0o644); err != nil { + return err + } + res["md"] = mdOut + } + printObject(res) + return nil + }, + } + cmd.Flags().StringVar(&out, "out", "", "output searchable PDF (default: .ocr.pdf)") + cmd.Flags().StringVar(&mdOut, "md", "", "write Markdown extraction to this path") + cmd.Flags().BoolVar(&writeMD, "markdown", false, "also write sibling .md next to OCR PDF") + cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language(s) for tesseract/ocrmypdf") + cmd.Flags().BoolVar(&force, "force", true, "force OCR even if a text layer exists") + return cmd +} + +func docsAsMDCmd() *cobra.Command { + var to, lang string + var minChars int + var uploadOCR bool + cmd := &cobra.Command{ + Use: "as-md FILE_ID", + Short: "Download an OO Documents file and emit Markdown (OCR PDF/image if needed)", + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + c, err := newOO(cmd) + if err != nil { + return err + } + ctx := cmd.Context() + meta, err := c.GetFile(ctx, args[0]) + if err != nil { + return err + } + title := onlyoffice.FileEntryTitle(meta) + dir, err := os.MkdirTemp("", "oo-docs-as-md-*") + if err != nil { + return err + } + defer os.RemoveAll(dir) + + local := filepath.Join(dir, onlyoffice.SafeLocalFileName(title)) + f, err := os.Create(local) + if err != nil { + return err + } + if _, err := c.DownloadFile(ctx, args[0], f); err != nil { + _ = f.Close() + return err + } + _ = f.Close() + + tools := docpipe.LookPath() + res, err := tools.ToMarkdown(local, dir, lang, minChars) + if err != nil { + return err + } + outPath := to + if outPath == "" { + base := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title))) + if base == "" || base == "download" { + base = "file-" + args[0] + } + outPath = base + ".md" + } + if err := docpipe.EnsureDir(outPath); err != nil { + return err + } + if err := os.WriteFile(outPath, []byte(res.Markdown), 0o644); err != nil { + return err + } + + obj := map[string]any{ + "file_id": args[0], + "title": title, + "md": outPath, + "did_ocr": res.DidOCR, + } + if res.OCRPDFPath != "" && uploadOCR { + folderID := onlyoffice.FileFolderID(meta) + upName := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title))) + ".ocr.pdf" + tmpUp := filepath.Join(dir, upName) + data, err := os.ReadFile(res.OCRPDFPath) + if err != nil { + return err + } + if err := os.WriteFile(tmpUp, data, 0o644); err != nil { + return err + } + if folderID == "" { + obj["ocr_pdf_local"] = res.OCRPDFPath + obj["note"] = "file has no folderId; OCR PDF left local — pass after moving into a folder" + } else { + ent, err := c.UploadToFolder(ctx, folderID, tmpUp) + if err != nil { + return err + } + obj["ocr_pdf_file_id"] = fileIDStr(ent) + obj["ocr_pdf_title"] = onlyoffice.FileEntryTitle(ent) + } + } else if res.OCRPDFPath != "" { + // Keep OCR PDF outside temp by copying beside md if requested via env-less default: + kept := strings.TrimSuffix(outPath, filepath.Ext(outPath)) + ".ocr.pdf" + if b, err := os.ReadFile(res.OCRPDFPath); err == nil { + _ = os.WriteFile(kept, b, 0o644) + obj["ocr_pdf_local"] = kept + } else { + obj["ocr_pdf_local"] = res.OCRPDFPath + } + } + printObject(obj) + return nil + }, + } + cmd.Flags().StringVar(&to, "to", "", "write Markdown to this path (default: ./.md)") + cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language") + cmd.Flags().IntVar(&minChars, "min-chars", docpipe.DefaultMinTextChars, "OCR PDF if text layer shorter than this") + cmd.Flags().BoolVar(&uploadOCR, "upload-ocr", false, "upload searchable OCR PDF back into the same OO folder") + return cmd +} + +func docsPutMDCmd() *cobra.Command { + var folderID string + var keepLocalDOCX string + cmd := &cobra.Command{ + Use: "put-md PROJECT_ID MARKDOWN_PATH", + Short: "Convert Markdown→DOCX and upload DOCX into a project (OO-friendly)", + Long: `Agents edit .md locally; this uploads .docx so OnlyOffice can open/version it.`, + Args: cobra.ExactArgs(2), + RunE: func(cmd *cobra.Command, args []string) error { + pid, mdPath := args[0], args[1] + c, err := newOO(cmd) + if err != nil { + return err + } + tools := docpipe.LookPath() + dir, err := os.MkdirTemp("", "oo-docs-put-md-*") + if err != nil { + return err + } + defer os.RemoveAll(dir) + docxName := strings.TrimSuffix(filepath.Base(mdPath), filepath.Ext(mdPath)) + ".docx" + docxPath := filepath.Join(dir, docxName) + if err := tools.MDToDOCX(mdPath, docxPath); err != nil { + return err + } + if keepLocalDOCX != "" { + if err := docpipe.EnsureDir(keepLocalDOCX); err != nil { + return err + } + b, err := os.ReadFile(docxPath) + if err != nil { + return err + } + if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil { + return err + } + } + ctx := cmd.Context() + if folderID != "" { + ent, err := c.UploadToFolder(ctx, folderID, docxPath) + if err != nil { + return err + } + printObject(map[string]any{ + "project_id": pid, + "folder_id": folderID, + "md": mdPath, + "uploaded": fileEntryToMap(ent), + }) + return nil + } + ent, err := c.UploadProjectFile(ctx, pid, docxPath) + if err != nil { + return err + } + printObject(map[string]any{ + "project_id": pid, + "md": mdPath, + "uploaded": fileEntryToMap(ent), + }) + return nil + }, + } + cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)") + cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path") + return cmd +} diff --git a/cmd/oo/main.go b/cmd/oo/main.go index 30ba5b6..63c2f83 100644 --- a/cmd/oo/main.go +++ b/cmd/oo/main.go @@ -3,7 +3,7 @@ // Command tree is subject-based (mirrors the library split and the `tea` CLI): // // oo calendar list | events | add | delete -// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete) +// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete|as-md|put-md) // oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach) // oo users list | self (alias: oo whoami) // oo contacts list | get | delete | info-add | merge | dedupe-info @@ -15,6 +15,7 @@ // oo crm cleanup // oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | delete // oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items … +// oo docs tools | convert | ocr | as-md | put-md // // CRM association rules: docs/crm-associations.md // diff --git a/cmd/oo/projects_files.go b/cmd/oo/projects_files.go index 2acef18..f92a105 100644 --- a/cmd/oo/projects_files.go +++ b/cmd/oo/projects_files.go @@ -24,9 +24,26 @@ func projectFilesCmd() *cobra.Command { cmd.AddCommand(prjFilesDownloadCmd()) cmd.AddCommand(prjFilesRenameCmd()) cmd.AddCommand(prjFilesDeleteCmd()) + // Convenience aliases into oo docs (md↔docx / OCR pipeline). + cmd.AddCommand(aliasDocsAsMD()) + cmd.AddCommand(aliasDocsPutMD()) return cmd } +func aliasDocsAsMD() *cobra.Command { + c := docsAsMDCmd() + c.Use = "as-md FILE_ID" + c.Short = "Alias of `oo docs as-md` — download OO file as Markdown (OCR if needed)" + return c +} + +func aliasDocsPutMD() *cobra.Command { + c := docsPutMDCmd() + c.Use = "put-md PROJECT_ID MARKDOWN_PATH" + c.Short = "Alias of `oo docs put-md` — Markdown→DOCX upload into project" + return c +} + func prjFilesListCmd() *cobra.Command { var showFolders bool cmd := &cobra.Command{ diff --git a/files.go b/files.go index c0d0cdd..0f0e33b 100644 --- a/files.go +++ b/files.go @@ -281,6 +281,74 @@ func (c *Client) DeleteFiles(ctx context.Context, fileIDs []int) error { return err } +// ListFolder returns the Documents module listing for a folder id +// (GET /api/2.0/files/{folderId}). +func (c *Client) ListFolder(ctx context.Context, folderID string) (map[string]any, error) { + if folderID == "" { + return nil, fmt.Errorf("folder id is required") + } + out, err := c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+".json") + if err != nil { + out, err = c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID)) + } + return out, err +} + +// CreateFolder creates a subfolder under parentFolderID. +func (c *Client) CreateFolder(ctx context.Context, parentFolderID, title string) (map[string]any, error) { + if parentFolderID == "" || title == "" { + return nil, fmt.Errorf("parent folder id and title are required") + } + body := map[string]any{"title": title} + out, err := c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID)+".json", body) + if err != nil { + out, err = c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID), body) + } + return out, err +} + +// MoveFiles moves file ids into destFolderID (Documents fileops/move). +func (c *Client) MoveFiles(ctx context.Context, destFolderID int, fileIDs []int) (map[string]any, error) { + if destFolderID == 0 || len(fileIDs) == 0 { + return nil, fmt.Errorf("dest folder and file ids are required") + } + body := map[string]any{ + "folderIds": []int{}, + "fileIds": fileIDs, + "destFolderId": destFolderID, + } + out, err := c.putJSONObject(ctx, "/api/2.0/files/fileops/move.json", body) + if err != nil { + out, err = c.putJSONObject(ctx, "/api/2.0/files/fileops/move", body) + } + return out, err +} + +// UploadToFolder uploads a local file into an arbitrary Documents folder id. +func (c *Client) UploadToFolder(ctx context.Context, folderID, localPath string) (*FileEntry, error) { + if folderID == "" || localPath == "" { + return nil, fmt.Errorf("folder id and local path are required") + } + uploadPath := fmt.Sprintf("/api/2.0/files/%s/upload.json", url.PathEscape(folderID)) + raw, err := c.uploadMultipart(ctx, uploadPath, "file", localPath) + if err != nil { + uploadPath = fmt.Sprintf("/api/2.0/files/%s/upload", url.PathEscape(folderID)) + raw, err = c.uploadMultipart(ctx, uploadPath, "file", localPath) + if err != nil { + return nil, err + } + } + return decodeResponseFileEntry(raw) +} + +// FileFolderID returns the parent folder id string for a file entry, if known. +func FileFolderID(f *FileEntry) string { + if f == nil || f.FolderID == nil { + return "" + } + return f.FolderID.String() +} + // DownloadFile streams file bytes from the file's viewUrl using the same auth // as API calls. Writes into dst. func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) { diff --git a/internal/docpipe/docpipe.go b/internal/docpipe/docpipe.go new file mode 100644 index 0000000..16d1211 --- /dev/null +++ b/internal/docpipe/docpipe.go @@ -0,0 +1,311 @@ +// Package docpipe converts documents for OnlyOffice agent workflows: +// Markdown ↔ DOCX (pandoc) and image/PDF OCR → searchable PDF + Markdown text. +// +// External tools (optional at runtime; helpers skip/error clearly when missing): +// - pandoc — md↔docx +// - ocrmypdf — OCR into a searchable PDF +// - pdftotext — extract text layer +// - tesseract — OCR single images when ocrmypdf is unsuitable +package docpipe + +import ( + "bytes" + "fmt" + "os" + "os/exec" + "path/filepath" + "strings" +) + +// DefaultMinTextChars: below this, a PDF is treated as needing OCR. +const DefaultMinTextChars = 200 + +// Tools reports which converters are available on PATH. +type Tools struct { + Pandoc string + OCRMyPDF string + PDFToText string + Tesseract string +} + +// LookPath resolves converter binaries (empty string if missing). +func LookPath() Tools { + find := func(names ...string) string { + for _, n := range names { + if p, err := exec.LookPath(n); err == nil { + return p + } + } + return "" + } + return Tools{ + Pandoc: find("pandoc"), + OCRMyPDF: find("ocrmypdf"), + PDFToText: find("pdftotext"), + Tesseract: find("tesseract"), + } +} + +func (t Tools) requirePandoc() error { + if t.Pandoc == "" { + return fmt.Errorf("pandoc not found on PATH (needed for md↔docx)") + } + return nil +} + +// Ext returns lower-case extension including dot (".pdf"). +func Ext(path string) string { + return strings.ToLower(filepath.Ext(path)) +} + +// ConvertFile converts between md and docx (and other pandoc formats) via pandoc. +// outExt may be ".md", ".docx", or a full output path. +func (t Tools) ConvertFile(inPath, outPath string) error { + if err := t.requirePandoc(); err != nil { + return err + } + if strings.TrimSpace(outPath) == "" { + return fmt.Errorf("output path required") + } + cmd := exec.Command(t.Pandoc, inPath, "-o", outPath) + var stderr bytes.Buffer + cmd.Stderr = &stderr + if err := cmd.Run(); err != nil { + return fmt.Errorf("pandoc %s → %s: %w (%s)", inPath, outPath, err, strings.TrimSpace(stderr.String())) + } + return nil +} + +// MDToDOCX writes a DOCX next to or at outPath from a Markdown file. +func (t Tools) MDToDOCX(mdPath, docxPath string) error { + if Ext(mdPath) != ".md" && Ext(mdPath) != ".markdown" { + return fmt.Errorf("expected markdown input, got %q", mdPath) + } + if docxPath == "" { + docxPath = strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx" + } + return t.ConvertFile(mdPath, docxPath) +} + +// DOCXToMD writes Markdown from a DOCX file. +func (t Tools) DOCXToMD(docxPath, mdPath string) error { + if Ext(docxPath) != ".docx" { + return fmt.Errorf("expected .docx input, got %q", docxPath) + } + if mdPath == "" { + mdPath = strings.TrimSuffix(docxPath, Ext(docxPath)) + ".md" + } + return t.ConvertFile(docxPath, mdPath) +} + +// PDFTextLayerChars returns approximate extracted character count (0 if unavailable). +func (t Tools) PDFTextLayerChars(pdfPath string) (int, error) { + if t.PDFToText == "" { + return 0, fmt.Errorf("pdftotext not found on PATH") + } + cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-") + out, err := cmd.Output() + if err != nil { + return 0, err + } + return len(bytes.TrimSpace(out)), nil +} + +// NeedsOCR reports whether path likely needs OCR before text extraction. +func (t Tools) NeedsOCR(path string, minChars int) (bool, error) { + if minChars <= 0 { + minChars = DefaultMinTextChars + } + switch Ext(path) { + case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp": + return true, nil + case ".pdf": + n, err := t.PDFTextLayerChars(path) + if err != nil { + // If we cannot measure, prefer OCR. + return true, nil + } + return n < minChars, nil + default: + return false, nil + } +} + +// OCRToPDF runs ocrmypdf into outPDF (searchable). Forces OCR when force is true. +func (t Tools) OCRToPDF(inPath, outPDF string, force bool, lang string) error { + if t.OCRMyPDF == "" { + return fmt.Errorf("ocrmypdf not found on PATH") + } + if outPDF == "" { + return fmt.Errorf("output PDF path required") + } + if lang == "" { + lang = "eng" + } + args := []string{"-l", lang, "--skip-big", "100"} + if force { + args = append(args, "--force-ocr") + } else { + args = append(args, "--skip-text") + } + args = append(args, inPath, outPDF) + cmd := exec.Command(t.OCRMyPDF, args...) + var stderr bytes.Buffer + cmd.Stderr = &stderr + if err := cmd.Run(); err != nil { + // Retry with force if skip-text refused. + if !force && strings.Contains(stderr.String(), "PriorOcrFoundError") == false { + args2 := []string{"-l", lang, "--force-ocr", inPath, outPDF} + cmd2 := exec.Command(t.OCRMyPDF, args2...) + var stderr2 bytes.Buffer + cmd2.Stderr = &stderr2 + if err2 := cmd2.Run(); err2 == nil { + return nil + } + } + return fmt.Errorf("ocrmypdf: %w (%s)", err, strings.TrimSpace(stderr.String())) + } + return nil +} + +// ImageToText OCRs a raster image with tesseract (stdout text). +func (t Tools) ImageToText(imgPath, lang string) (string, error) { + if t.Tesseract == "" { + return "", fmt.Errorf("tesseract not found on PATH") + } + if lang == "" { + lang = "eng" + } + cmd := exec.Command(t.Tesseract, imgPath, "stdout", "-l", lang) + var stderr bytes.Buffer + cmd.Stderr = &stderr + out, err := cmd.Output() + if err != nil { + return "", fmt.Errorf("tesseract: %w (%s)", err, strings.TrimSpace(stderr.String())) + } + return string(out), nil +} + +// ExtractPDFText returns layout text from a PDF via pdftotext. +func (t Tools) ExtractPDFText(pdfPath string) (string, error) { + if t.PDFToText == "" { + return "", fmt.Errorf("pdftotext not found on PATH") + } + cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-") + out, err := cmd.Output() + if err != nil { + return "", err + } + return string(out), nil +} + +// Result of ToMarkdown. +type Result struct { + Markdown string + OCRPDFPath string // set when a searchable PDF was produced + DidOCR bool + Source string +} + +// ToMarkdown turns a local file into Markdown text. +// PDFs/images with weak/no text layer are OCR'd to a searchable PDF first (when tools exist). +func (t Tools) ToMarkdown(path string, workDir string, lang string, minChars int) (Result, error) { + res := Result{Source: path} + ext := Ext(path) + switch ext { + case ".md", ".markdown", ".txt": + b, err := os.ReadFile(path) + if err != nil { + return res, err + } + res.Markdown = string(b) + return res, nil + case ".docx", ".odt", ".rtf", ".html", ".htm": + if err := t.requirePandoc(); err != nil { + return res, err + } + tmp := filepath.Join(workDir, "out.md") + if err := t.ConvertFile(path, tmp); err != nil { + return res, err + } + b, err := os.ReadFile(tmp) + if err != nil { + return res, err + } + res.Markdown = string(b) + return res, nil + case ".pdf": + need, _ := t.NeedsOCR(path, minChars) + pdf := path + if need { + if workDir == "" { + workDir = os.TempDir() + } + outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf") + if err := t.OCRToPDF(path, outPDF, true, lang); err != nil { + return res, err + } + res.DidOCR = true + res.OCRPDFPath = outPDF + pdf = outPDF + } + text, err := t.ExtractPDFText(pdf) + if err != nil { + return res, err + } + res.Markdown = wrapMD(filepath.Base(path), text) + return res, nil + case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp": + if workDir == "" { + workDir = os.TempDir() + } + outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf") + if t.OCRMyPDF != "" { + if err := t.OCRToPDF(path, outPDF, true, lang); err == nil { + res.DidOCR = true + res.OCRPDFPath = outPDF + text, err := t.ExtractPDFText(outPDF) + if err != nil { + return res, err + } + res.Markdown = wrapMD(filepath.Base(path), text) + return res, nil + } + } + text, err := t.ImageToText(path, lang) + if err != nil { + return res, err + } + res.DidOCR = true + res.Markdown = wrapMD(filepath.Base(path), text) + return res, nil + default: + return res, fmt.Errorf("unsupported type %q for markdown extraction", ext) + } +} + +func wrapMD(title, body string) string { + body = strings.TrimSpace(body) + if body == "" { + return "# " + title + "\n\n_(empty text layer)_\n" + } + return "# " + title + "\n\n" + body + "\n" +} + +func trimExt(name string) string { + return strings.TrimSuffix(name, filepath.Ext(name)) +} + +// SiblingDOCX returns path with .docx extension replacing the original ext. +func SiblingDOCX(mdPath string) string { + return strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx" +} + +// EnsureDir creates parent directories for path. +func EnsureDir(path string) error { + dir := filepath.Dir(path) + if dir == "" || dir == "." { + return nil + } + return os.MkdirAll(dir, 0o755) +} diff --git a/internal/docpipe/docpipe_test.go b/internal/docpipe/docpipe_test.go new file mode 100644 index 0000000..fdbfc55 --- /dev/null +++ b/internal/docpipe/docpipe_test.go @@ -0,0 +1,76 @@ +package docpipe + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestExt(t *testing.T) { + if Ext("Foo.PDF") != ".pdf" { + t.Fatalf("Ext: %q", Ext("Foo.PDF")) + } +} + +func TestSiblingDOCX(t *testing.T) { + if got := SiblingDOCX("notes.md"); got != "notes.docx" { + t.Fatalf("got %q", got) + } +} + +func TestWrapMD(t *testing.T) { + s := wrapMD("a.pdf", " hello ") + if !strings.HasPrefix(s, "# a.pdf\n") || !strings.Contains(s, "hello") { + t.Fatalf("wrap: %q", s) + } +} + +func TestNeedsOCR_Image(t *testing.T) { + tools := LookPath() + need, err := tools.NeedsOCR("x.jpg", 0) + if err != nil || !need { + t.Fatalf("jpg should need OCR: need=%v err=%v", need, err) + } +} + +func TestMDDocxRoundTrip(t *testing.T) { + tools := LookPath() + if tools.Pandoc == "" { + t.Skip("pandoc not installed") + } + dir := t.TempDir() + md := filepath.Join(dir, "n.md") + docx := filepath.Join(dir, "n.docx") + md2 := filepath.Join(dir, "n2.md") + if err := os.WriteFile(md, []byte("# Title\n\nHello **world**.\n"), 0o644); err != nil { + t.Fatal(err) + } + if err := tools.MDToDOCX(md, docx); err != nil { + t.Fatal(err) + } + if _, err := os.Stat(docx); err != nil { + t.Fatal(err) + } + if err := tools.DOCXToMD(docx, md2); err != nil { + t.Fatal(err) + } + b, err := os.ReadFile(md2) + if err != nil { + t.Fatal(err) + } + if !strings.Contains(string(b), "Hello") { + t.Fatalf("round-trip missing Hello: %s", b) + } +} + +func TestToMarkdown_PlainMD(t *testing.T) { + tools := LookPath() + dir := t.TempDir() + p := filepath.Join(dir, "a.md") + _ = os.WriteFile(p, []byte("hi"), 0o644) + res, err := tools.ToMarkdown(p, dir, "eng", 0) + if err != nil || res.Markdown != "hi" { + t.Fatalf("got %+v err=%v", res, err) + } +}