feat(docs): md↔docx convert, OCR→PDF, as-md/put-md for agents
Add oo docs pipeline (pandoc/ocrmypdf) so agents keep Markdown locally while OnlyOffice stores versioned DOCX; OCR weak PDFs/images before returning MD. Also expose ListFolder/CreateFolder/MoveFiles/UploadToFolder. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -28,6 +28,7 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
|
||||
- Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them.
|
||||
- Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason.
|
||||
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain.
|
||||
- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`. Use `oo docs put-md` (md→docx upload) and `oo docs as-md` (download→OCR if needed→markdown). Local converters live in `internal/docpipe` (pandoc / ocrmypdf / pdftotext).
|
||||
- Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command.
|
||||
- No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only.
|
||||
- Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`.
|
||||
|
||||
+314
@@ -0,0 +1,314 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
onlyoffice "github.com/eslider/go-onlyoffice"
|
||||
"github.com/eslider/go-onlyoffice/internal/docpipe"
|
||||
"github.com/spf13/cobra"
|
||||
)
|
||||
|
||||
func init() {
|
||||
rootCmd.AddCommand(docsCmd())
|
||||
}
|
||||
|
||||
func docsCmd() *cobra.Command {
|
||||
cmd := &cobra.Command{
|
||||
Use: "docs",
|
||||
Short: "Local document pipeline: md↔docx, OCR→PDF, extract Markdown",
|
||||
Long: `Agent-friendly conversions (requires pandoc / ocrmypdf / pdftotext on PATH).
|
||||
|
||||
OnlyOffice Documents UI is poor for .md — keep Markdown in git, store .docx in OO.
|
||||
Upload Markdown as DOCX: oo docs put-md PROJECT_ID file.md
|
||||
Read an OO file as MD: oo docs as-md FILE_ID
|
||||
OCR a scan locally: oo docs ocr scan.pdf --md out.md`,
|
||||
}
|
||||
cmd.AddCommand(docsConvertCmd())
|
||||
cmd.AddCommand(docsOCRCmd())
|
||||
cmd.AddCommand(docsAsMDCmd())
|
||||
cmd.AddCommand(docsPutMDCmd())
|
||||
cmd.AddCommand(docsToolsCmd())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsToolsCmd() *cobra.Command {
|
||||
return &cobra.Command{
|
||||
Use: "tools",
|
||||
Short: "Show which converter binaries are on PATH",
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
t := docpipe.LookPath()
|
||||
printObject(map[string]any{
|
||||
"pandoc": strOrNil(t.Pandoc),
|
||||
"ocrmypdf": strOrNil(t.OCRMyPDF),
|
||||
"pdftotext": strOrNil(t.PDFToText),
|
||||
"tesseract": strOrNil(t.Tesseract),
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func strOrNil(s string) any {
|
||||
if s == "" {
|
||||
return nil
|
||||
}
|
||||
return s
|
||||
}
|
||||
|
||||
func docsConvertCmd() *cobra.Command {
|
||||
var to string
|
||||
cmd := &cobra.Command{
|
||||
Use: "convert PATH",
|
||||
Short: "Convert a local file with pandoc (md↔docx by default)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
out := to
|
||||
if out == "" {
|
||||
switch docpipe.Ext(in) {
|
||||
case ".md", ".markdown":
|
||||
out = docpipe.SiblingDOCX(in)
|
||||
case ".docx":
|
||||
out = strings.TrimSuffix(in, docpipe.Ext(in)) + ".md"
|
||||
default:
|
||||
return fmt.Errorf("--to required for input type %s", docpipe.Ext(in))
|
||||
}
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := t.ConvertFile(in, out); err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{"in": in, "out": out})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "output path (default: sibling .docx or .md)")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsOCRCmd() *cobra.Command {
|
||||
var out, mdOut, lang string
|
||||
var force bool
|
||||
var writeMD bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "ocr PATH",
|
||||
Short: "OCR image/PDF → searchable PDF (and optional Markdown)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
in := args[0]
|
||||
t := docpipe.LookPath()
|
||||
if out == "" {
|
||||
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
|
||||
out = filepath.Join(filepath.Dir(in), base+".ocr.pdf")
|
||||
}
|
||||
if err := docpipe.EnsureDir(out); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := t.OCRToPDF(in, out, force, lang); err != nil {
|
||||
return err
|
||||
}
|
||||
res := map[string]any{"in": in, "pdf": out}
|
||||
if writeMD || mdOut != "" {
|
||||
if mdOut == "" {
|
||||
mdOut = strings.TrimSuffix(out, filepath.Ext(out)) + ".md"
|
||||
}
|
||||
text, err := t.ExtractPDFText(out)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
body := "# " + filepath.Base(in) + "\n\n" + strings.TrimSpace(text) + "\n"
|
||||
if err := os.WriteFile(mdOut, []byte(body), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
res["md"] = mdOut
|
||||
}
|
||||
printObject(res)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&out, "out", "", "output searchable PDF (default: <name>.ocr.pdf)")
|
||||
cmd.Flags().StringVar(&mdOut, "md", "", "write Markdown extraction to this path")
|
||||
cmd.Flags().BoolVar(&writeMD, "markdown", false, "also write sibling .md next to OCR PDF")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language(s) for tesseract/ocrmypdf")
|
||||
cmd.Flags().BoolVar(&force, "force", true, "force OCR even if a text layer exists")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsAsMDCmd() *cobra.Command {
|
||||
var to, lang string
|
||||
var minChars int
|
||||
var uploadOCR bool
|
||||
cmd := &cobra.Command{
|
||||
Use: "as-md FILE_ID",
|
||||
Short: "Download an OO Documents file and emit Markdown (OCR PDF/image if needed)",
|
||||
Args: cobra.ExactArgs(1),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
meta, err := c.GetFile(ctx, args[0])
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
title := onlyoffice.FileEntryTitle(meta)
|
||||
dir, err := os.MkdirTemp("", "oo-docs-as-md-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
|
||||
local := filepath.Join(dir, onlyoffice.SafeLocalFileName(title))
|
||||
f, err := os.Create(local)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if _, err := c.DownloadFile(ctx, args[0], f); err != nil {
|
||||
_ = f.Close()
|
||||
return err
|
||||
}
|
||||
_ = f.Close()
|
||||
|
||||
tools := docpipe.LookPath()
|
||||
res, err := tools.ToMarkdown(local, dir, lang, minChars)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
outPath := to
|
||||
if outPath == "" {
|
||||
base := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title)))
|
||||
if base == "" || base == "download" {
|
||||
base = "file-" + args[0]
|
||||
}
|
||||
outPath = base + ".md"
|
||||
}
|
||||
if err := docpipe.EnsureDir(outPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(outPath, []byte(res.Markdown), 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
obj := map[string]any{
|
||||
"file_id": args[0],
|
||||
"title": title,
|
||||
"md": outPath,
|
||||
"did_ocr": res.DidOCR,
|
||||
}
|
||||
if res.OCRPDFPath != "" && uploadOCR {
|
||||
folderID := onlyoffice.FileFolderID(meta)
|
||||
upName := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title))) + ".ocr.pdf"
|
||||
tmpUp := filepath.Join(dir, upName)
|
||||
data, err := os.ReadFile(res.OCRPDFPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(tmpUp, data, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
if folderID == "" {
|
||||
obj["ocr_pdf_local"] = res.OCRPDFPath
|
||||
obj["note"] = "file has no folderId; OCR PDF left local — pass after moving into a folder"
|
||||
} else {
|
||||
ent, err := c.UploadToFolder(ctx, folderID, tmpUp)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
obj["ocr_pdf_file_id"] = fileIDStr(ent)
|
||||
obj["ocr_pdf_title"] = onlyoffice.FileEntryTitle(ent)
|
||||
}
|
||||
} else if res.OCRPDFPath != "" {
|
||||
// Keep OCR PDF outside temp by copying beside md if requested via env-less default:
|
||||
kept := strings.TrimSuffix(outPath, filepath.Ext(outPath)) + ".ocr.pdf"
|
||||
if b, err := os.ReadFile(res.OCRPDFPath); err == nil {
|
||||
_ = os.WriteFile(kept, b, 0o644)
|
||||
obj["ocr_pdf_local"] = kept
|
||||
} else {
|
||||
obj["ocr_pdf_local"] = res.OCRPDFPath
|
||||
}
|
||||
}
|
||||
printObject(obj)
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&to, "to", "", "write Markdown to this path (default: ./<title>.md)")
|
||||
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language")
|
||||
cmd.Flags().IntVar(&minChars, "min-chars", docpipe.DefaultMinTextChars, "OCR PDF if text layer shorter than this")
|
||||
cmd.Flags().BoolVar(&uploadOCR, "upload-ocr", false, "upload searchable OCR PDF back into the same OO folder")
|
||||
return cmd
|
||||
}
|
||||
|
||||
func docsPutMDCmd() *cobra.Command {
|
||||
var folderID string
|
||||
var keepLocalDOCX string
|
||||
cmd := &cobra.Command{
|
||||
Use: "put-md PROJECT_ID MARKDOWN_PATH",
|
||||
Short: "Convert Markdown→DOCX and upload DOCX into a project (OO-friendly)",
|
||||
Long: `Agents edit .md locally; this uploads .docx so OnlyOffice can open/version it.`,
|
||||
Args: cobra.ExactArgs(2),
|
||||
RunE: func(cmd *cobra.Command, args []string) error {
|
||||
pid, mdPath := args[0], args[1]
|
||||
c, err := newOO(cmd)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tools := docpipe.LookPath()
|
||||
dir, err := os.MkdirTemp("", "oo-docs-put-md-*")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer os.RemoveAll(dir)
|
||||
docxName := strings.TrimSuffix(filepath.Base(mdPath), filepath.Ext(mdPath)) + ".docx"
|
||||
docxPath := filepath.Join(dir, docxName)
|
||||
if err := tools.MDToDOCX(mdPath, docxPath); err != nil {
|
||||
return err
|
||||
}
|
||||
if keepLocalDOCX != "" {
|
||||
if err := docpipe.EnsureDir(keepLocalDOCX); err != nil {
|
||||
return err
|
||||
}
|
||||
b, err := os.ReadFile(docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
ctx := cmd.Context()
|
||||
if folderID != "" {
|
||||
ent, err := c.UploadToFolder(ctx, folderID, docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{
|
||||
"project_id": pid,
|
||||
"folder_id": folderID,
|
||||
"md": mdPath,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
})
|
||||
return nil
|
||||
}
|
||||
ent, err := c.UploadProjectFile(ctx, pid, docxPath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
printObject(map[string]any{
|
||||
"project_id": pid,
|
||||
"md": mdPath,
|
||||
"uploaded": fileEntryToMap(ent),
|
||||
})
|
||||
return nil
|
||||
},
|
||||
}
|
||||
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
|
||||
cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path")
|
||||
return cmd
|
||||
}
|
||||
+2
-1
@@ -3,7 +3,7 @@
|
||||
// Command tree is subject-based (mirrors the library split and the `tea` CLI):
|
||||
//
|
||||
// oo calendar list | events | add | delete
|
||||
// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete)
|
||||
// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete|as-md|put-md)
|
||||
// oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach)
|
||||
// oo users list | self (alias: oo whoami)
|
||||
// oo contacts list | get | delete | info-add | merge | dedupe-info
|
||||
@@ -15,6 +15,7 @@
|
||||
// oo crm cleanup
|
||||
// oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | delete
|
||||
// oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items …
|
||||
// oo docs tools | convert | ocr | as-md | put-md
|
||||
//
|
||||
// CRM association rules: docs/crm-associations.md
|
||||
//
|
||||
|
||||
@@ -24,9 +24,26 @@ func projectFilesCmd() *cobra.Command {
|
||||
cmd.AddCommand(prjFilesDownloadCmd())
|
||||
cmd.AddCommand(prjFilesRenameCmd())
|
||||
cmd.AddCommand(prjFilesDeleteCmd())
|
||||
// Convenience aliases into oo docs (md↔docx / OCR pipeline).
|
||||
cmd.AddCommand(aliasDocsAsMD())
|
||||
cmd.AddCommand(aliasDocsPutMD())
|
||||
return cmd
|
||||
}
|
||||
|
||||
func aliasDocsAsMD() *cobra.Command {
|
||||
c := docsAsMDCmd()
|
||||
c.Use = "as-md FILE_ID"
|
||||
c.Short = "Alias of `oo docs as-md` — download OO file as Markdown (OCR if needed)"
|
||||
return c
|
||||
}
|
||||
|
||||
func aliasDocsPutMD() *cobra.Command {
|
||||
c := docsPutMDCmd()
|
||||
c.Use = "put-md PROJECT_ID MARKDOWN_PATH"
|
||||
c.Short = "Alias of `oo docs put-md` — Markdown→DOCX upload into project"
|
||||
return c
|
||||
}
|
||||
|
||||
func prjFilesListCmd() *cobra.Command {
|
||||
var showFolders bool
|
||||
cmd := &cobra.Command{
|
||||
|
||||
@@ -281,6 +281,74 @@ func (c *Client) DeleteFiles(ctx context.Context, fileIDs []int) error {
|
||||
return err
|
||||
}
|
||||
|
||||
// ListFolder returns the Documents module listing for a folder id
|
||||
// (GET /api/2.0/files/{folderId}).
|
||||
func (c *Client) ListFolder(ctx context.Context, folderID string) (map[string]any, error) {
|
||||
if folderID == "" {
|
||||
return nil, fmt.Errorf("folder id is required")
|
||||
}
|
||||
out, err := c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+".json")
|
||||
if err != nil {
|
||||
out, err = c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID))
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// CreateFolder creates a subfolder under parentFolderID.
|
||||
func (c *Client) CreateFolder(ctx context.Context, parentFolderID, title string) (map[string]any, error) {
|
||||
if parentFolderID == "" || title == "" {
|
||||
return nil, fmt.Errorf("parent folder id and title are required")
|
||||
}
|
||||
body := map[string]any{"title": title}
|
||||
out, err := c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID)+".json", body)
|
||||
if err != nil {
|
||||
out, err = c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID), body)
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// MoveFiles moves file ids into destFolderID (Documents fileops/move).
|
||||
func (c *Client) MoveFiles(ctx context.Context, destFolderID int, fileIDs []int) (map[string]any, error) {
|
||||
if destFolderID == 0 || len(fileIDs) == 0 {
|
||||
return nil, fmt.Errorf("dest folder and file ids are required")
|
||||
}
|
||||
body := map[string]any{
|
||||
"folderIds": []int{},
|
||||
"fileIds": fileIDs,
|
||||
"destFolderId": destFolderID,
|
||||
}
|
||||
out, err := c.putJSONObject(ctx, "/api/2.0/files/fileops/move.json", body)
|
||||
if err != nil {
|
||||
out, err = c.putJSONObject(ctx, "/api/2.0/files/fileops/move", body)
|
||||
}
|
||||
return out, err
|
||||
}
|
||||
|
||||
// UploadToFolder uploads a local file into an arbitrary Documents folder id.
|
||||
func (c *Client) UploadToFolder(ctx context.Context, folderID, localPath string) (*FileEntry, error) {
|
||||
if folderID == "" || localPath == "" {
|
||||
return nil, fmt.Errorf("folder id and local path are required")
|
||||
}
|
||||
uploadPath := fmt.Sprintf("/api/2.0/files/%s/upload.json", url.PathEscape(folderID))
|
||||
raw, err := c.uploadMultipart(ctx, uploadPath, "file", localPath)
|
||||
if err != nil {
|
||||
uploadPath = fmt.Sprintf("/api/2.0/files/%s/upload", url.PathEscape(folderID))
|
||||
raw, err = c.uploadMultipart(ctx, uploadPath, "file", localPath)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return decodeResponseFileEntry(raw)
|
||||
}
|
||||
|
||||
// FileFolderID returns the parent folder id string for a file entry, if known.
|
||||
func FileFolderID(f *FileEntry) string {
|
||||
if f == nil || f.FolderID == nil {
|
||||
return ""
|
||||
}
|
||||
return f.FolderID.String()
|
||||
}
|
||||
|
||||
// DownloadFile streams file bytes from the file's viewUrl using the same auth
|
||||
// as API calls. Writes into dst.
|
||||
func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) {
|
||||
|
||||
@@ -0,0 +1,311 @@
|
||||
// Package docpipe converts documents for OnlyOffice agent workflows:
|
||||
// Markdown ↔ DOCX (pandoc) and image/PDF OCR → searchable PDF + Markdown text.
|
||||
//
|
||||
// External tools (optional at runtime; helpers skip/error clearly when missing):
|
||||
// - pandoc — md↔docx
|
||||
// - ocrmypdf — OCR into a searchable PDF
|
||||
// - pdftotext — extract text layer
|
||||
// - tesseract — OCR single images when ocrmypdf is unsuitable
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// DefaultMinTextChars: below this, a PDF is treated as needing OCR.
|
||||
const DefaultMinTextChars = 200
|
||||
|
||||
// Tools reports which converters are available on PATH.
|
||||
type Tools struct {
|
||||
Pandoc string
|
||||
OCRMyPDF string
|
||||
PDFToText string
|
||||
Tesseract string
|
||||
}
|
||||
|
||||
// LookPath resolves converter binaries (empty string if missing).
|
||||
func LookPath() Tools {
|
||||
find := func(names ...string) string {
|
||||
for _, n := range names {
|
||||
if p, err := exec.LookPath(n); err == nil {
|
||||
return p
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
return Tools{
|
||||
Pandoc: find("pandoc"),
|
||||
OCRMyPDF: find("ocrmypdf"),
|
||||
PDFToText: find("pdftotext"),
|
||||
Tesseract: find("tesseract"),
|
||||
}
|
||||
}
|
||||
|
||||
func (t Tools) requirePandoc() error {
|
||||
if t.Pandoc == "" {
|
||||
return fmt.Errorf("pandoc not found on PATH (needed for md↔docx)")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Ext returns lower-case extension including dot (".pdf").
|
||||
func Ext(path string) string {
|
||||
return strings.ToLower(filepath.Ext(path))
|
||||
}
|
||||
|
||||
// ConvertFile converts between md and docx (and other pandoc formats) via pandoc.
|
||||
// outExt may be ".md", ".docx", or a full output path.
|
||||
func (t Tools) ConvertFile(inPath, outPath string) error {
|
||||
if err := t.requirePandoc(); err != nil {
|
||||
return err
|
||||
}
|
||||
if strings.TrimSpace(outPath) == "" {
|
||||
return fmt.Errorf("output path required")
|
||||
}
|
||||
cmd := exec.Command(t.Pandoc, inPath, "-o", outPath)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
return fmt.Errorf("pandoc %s → %s: %w (%s)", inPath, outPath, err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// MDToDOCX writes a DOCX next to or at outPath from a Markdown file.
|
||||
func (t Tools) MDToDOCX(mdPath, docxPath string) error {
|
||||
if Ext(mdPath) != ".md" && Ext(mdPath) != ".markdown" {
|
||||
return fmt.Errorf("expected markdown input, got %q", mdPath)
|
||||
}
|
||||
if docxPath == "" {
|
||||
docxPath = strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
|
||||
}
|
||||
return t.ConvertFile(mdPath, docxPath)
|
||||
}
|
||||
|
||||
// DOCXToMD writes Markdown from a DOCX file.
|
||||
func (t Tools) DOCXToMD(docxPath, mdPath string) error {
|
||||
if Ext(docxPath) != ".docx" {
|
||||
return fmt.Errorf("expected .docx input, got %q", docxPath)
|
||||
}
|
||||
if mdPath == "" {
|
||||
mdPath = strings.TrimSuffix(docxPath, Ext(docxPath)) + ".md"
|
||||
}
|
||||
return t.ConvertFile(docxPath, mdPath)
|
||||
}
|
||||
|
||||
// PDFTextLayerChars returns approximate extracted character count (0 if unavailable).
|
||||
func (t Tools) PDFTextLayerChars(pdfPath string) (int, error) {
|
||||
if t.PDFToText == "" {
|
||||
return 0, fmt.Errorf("pdftotext not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
return len(bytes.TrimSpace(out)), nil
|
||||
}
|
||||
|
||||
// NeedsOCR reports whether path likely needs OCR before text extraction.
|
||||
func (t Tools) NeedsOCR(path string, minChars int) (bool, error) {
|
||||
if minChars <= 0 {
|
||||
minChars = DefaultMinTextChars
|
||||
}
|
||||
switch Ext(path) {
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
return true, nil
|
||||
case ".pdf":
|
||||
n, err := t.PDFTextLayerChars(path)
|
||||
if err != nil {
|
||||
// If we cannot measure, prefer OCR.
|
||||
return true, nil
|
||||
}
|
||||
return n < minChars, nil
|
||||
default:
|
||||
return false, nil
|
||||
}
|
||||
}
|
||||
|
||||
// OCRToPDF runs ocrmypdf into outPDF (searchable). Forces OCR when force is true.
|
||||
func (t Tools) OCRToPDF(inPath, outPDF string, force bool, lang string) error {
|
||||
if t.OCRMyPDF == "" {
|
||||
return fmt.Errorf("ocrmypdf not found on PATH")
|
||||
}
|
||||
if outPDF == "" {
|
||||
return fmt.Errorf("output PDF path required")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
args := []string{"-l", lang, "--skip-big", "100"}
|
||||
if force {
|
||||
args = append(args, "--force-ocr")
|
||||
} else {
|
||||
args = append(args, "--skip-text")
|
||||
}
|
||||
args = append(args, inPath, outPDF)
|
||||
cmd := exec.Command(t.OCRMyPDF, args...)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
// Retry with force if skip-text refused.
|
||||
if !force && strings.Contains(stderr.String(), "PriorOcrFoundError") == false {
|
||||
args2 := []string{"-l", lang, "--force-ocr", inPath, outPDF}
|
||||
cmd2 := exec.Command(t.OCRMyPDF, args2...)
|
||||
var stderr2 bytes.Buffer
|
||||
cmd2.Stderr = &stderr2
|
||||
if err2 := cmd2.Run(); err2 == nil {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("ocrmypdf: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// ImageToText OCRs a raster image with tesseract (stdout text).
|
||||
func (t Tools) ImageToText(imgPath, lang string) (string, error) {
|
||||
if t.Tesseract == "" {
|
||||
return "", fmt.Errorf("tesseract not found on PATH")
|
||||
}
|
||||
if lang == "" {
|
||||
lang = "eng"
|
||||
}
|
||||
cmd := exec.Command(t.Tesseract, imgPath, "stdout", "-l", lang)
|
||||
var stderr bytes.Buffer
|
||||
cmd.Stderr = &stderr
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("tesseract: %w (%s)", err, strings.TrimSpace(stderr.String()))
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
// ExtractPDFText returns layout text from a PDF via pdftotext.
|
||||
func (t Tools) ExtractPDFText(pdfPath string) (string, error) {
|
||||
if t.PDFToText == "" {
|
||||
return "", fmt.Errorf("pdftotext not found on PATH")
|
||||
}
|
||||
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(out), nil
|
||||
}
|
||||
|
||||
// Result of ToMarkdown.
|
||||
type Result struct {
|
||||
Markdown string
|
||||
OCRPDFPath string // set when a searchable PDF was produced
|
||||
DidOCR bool
|
||||
Source string
|
||||
}
|
||||
|
||||
// ToMarkdown turns a local file into Markdown text.
|
||||
// PDFs/images with weak/no text layer are OCR'd to a searchable PDF first (when tools exist).
|
||||
func (t Tools) ToMarkdown(path string, workDir string, lang string, minChars int) (Result, error) {
|
||||
res := Result{Source: path}
|
||||
ext := Ext(path)
|
||||
switch ext {
|
||||
case ".md", ".markdown", ".txt":
|
||||
b, err := os.ReadFile(path)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = string(b)
|
||||
return res, nil
|
||||
case ".docx", ".odt", ".rtf", ".html", ".htm":
|
||||
if err := t.requirePandoc(); err != nil {
|
||||
return res, err
|
||||
}
|
||||
tmp := filepath.Join(workDir, "out.md")
|
||||
if err := t.ConvertFile(path, tmp); err != nil {
|
||||
return res, err
|
||||
}
|
||||
b, err := os.ReadFile(tmp)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = string(b)
|
||||
return res, nil
|
||||
case ".pdf":
|
||||
need, _ := t.NeedsOCR(path, minChars)
|
||||
pdf := path
|
||||
if need {
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
|
||||
if err := t.OCRToPDF(path, outPDF, true, lang); err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.OCRPDFPath = outPDF
|
||||
pdf = outPDF
|
||||
}
|
||||
text, err := t.ExtractPDFText(pdf)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
|
||||
if workDir == "" {
|
||||
workDir = os.TempDir()
|
||||
}
|
||||
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
|
||||
if t.OCRMyPDF != "" {
|
||||
if err := t.OCRToPDF(path, outPDF, true, lang); err == nil {
|
||||
res.DidOCR = true
|
||||
res.OCRPDFPath = outPDF
|
||||
text, err := t.ExtractPDFText(outPDF)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
}
|
||||
}
|
||||
text, err := t.ImageToText(path, lang)
|
||||
if err != nil {
|
||||
return res, err
|
||||
}
|
||||
res.DidOCR = true
|
||||
res.Markdown = wrapMD(filepath.Base(path), text)
|
||||
return res, nil
|
||||
default:
|
||||
return res, fmt.Errorf("unsupported type %q for markdown extraction", ext)
|
||||
}
|
||||
}
|
||||
|
||||
func wrapMD(title, body string) string {
|
||||
body = strings.TrimSpace(body)
|
||||
if body == "" {
|
||||
return "# " + title + "\n\n_(empty text layer)_\n"
|
||||
}
|
||||
return "# " + title + "\n\n" + body + "\n"
|
||||
}
|
||||
|
||||
func trimExt(name string) string {
|
||||
return strings.TrimSuffix(name, filepath.Ext(name))
|
||||
}
|
||||
|
||||
// SiblingDOCX returns path with .docx extension replacing the original ext.
|
||||
func SiblingDOCX(mdPath string) string {
|
||||
return strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
|
||||
}
|
||||
|
||||
// EnsureDir creates parent directories for path.
|
||||
func EnsureDir(path string) error {
|
||||
dir := filepath.Dir(path)
|
||||
if dir == "" || dir == "." {
|
||||
return nil
|
||||
}
|
||||
return os.MkdirAll(dir, 0o755)
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
package docpipe
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestExt(t *testing.T) {
|
||||
if Ext("Foo.PDF") != ".pdf" {
|
||||
t.Fatalf("Ext: %q", Ext("Foo.PDF"))
|
||||
}
|
||||
}
|
||||
|
||||
func TestSiblingDOCX(t *testing.T) {
|
||||
if got := SiblingDOCX("notes.md"); got != "notes.docx" {
|
||||
t.Fatalf("got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestWrapMD(t *testing.T) {
|
||||
s := wrapMD("a.pdf", " hello ")
|
||||
if !strings.HasPrefix(s, "# a.pdf\n") || !strings.Contains(s, "hello") {
|
||||
t.Fatalf("wrap: %q", s)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNeedsOCR_Image(t *testing.T) {
|
||||
tools := LookPath()
|
||||
need, err := tools.NeedsOCR("x.jpg", 0)
|
||||
if err != nil || !need {
|
||||
t.Fatalf("jpg should need OCR: need=%v err=%v", need, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestMDDocxRoundTrip(t *testing.T) {
|
||||
tools := LookPath()
|
||||
if tools.Pandoc == "" {
|
||||
t.Skip("pandoc not installed")
|
||||
}
|
||||
dir := t.TempDir()
|
||||
md := filepath.Join(dir, "n.md")
|
||||
docx := filepath.Join(dir, "n.docx")
|
||||
md2 := filepath.Join(dir, "n2.md")
|
||||
if err := os.WriteFile(md, []byte("# Title\n\nHello **world**.\n"), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.MDToDOCX(md, docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := os.Stat(docx); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := tools.DOCXToMD(docx, md2); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
b, err := os.ReadFile(md2)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !strings.Contains(string(b), "Hello") {
|
||||
t.Fatalf("round-trip missing Hello: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToMarkdown_PlainMD(t *testing.T) {
|
||||
tools := LookPath()
|
||||
dir := t.TempDir()
|
||||
p := filepath.Join(dir, "a.md")
|
||||
_ = os.WriteFile(p, []byte("hi"), 0o644)
|
||||
res, err := tools.ToMarkdown(p, dir, "eng", 0)
|
||||
if err != nil || res.Markdown != "hi" {
|
||||
t.Fatalf("got %+v err=%v", res, err)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user