feat(docs): md↔docx convert, OCR→PDF, as-md/put-md for agents

Add oo docs pipeline (pandoc/ocrmypdf) so agents keep Markdown locally while OnlyOffice stores versioned DOCX; OCR weak PDFs/images before returning MD. Also expose ListFolder/CreateFolder/MoveFiles/UploadToFolder.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-27 16:49:06 +01:00
co-authored by Cursor
parent 50d8974154
commit 6ea2fbadf7
7 changed files with 789 additions and 1 deletions
+1
View File
@@ -28,6 +28,7 @@ Canonical Go client for OnlyOffice Workspace (Projects + Calendar + CRM) and the
- Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them. - Prefer `ResponseObject` / `postFormObject` / `putFormObject` / `deleteObject` over hand-rolled `json.Unmarshal(responseField(...))` blocks — they exist for DRY, use them.
- Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason. - Domain split is by file, **not** by subpackage. Don't introduce `internal/` or `pkg/*` subpackages inside the library — it flattens the `*Client` call surface for a reason.
- CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain. - CLI commands follow **subject → verb** structure (`oo <subject> <verb>`), never `oo <verb>-<subject>`. Add new commands to the existing subject file if one fits; create a new `cmd/oo/<subject>.go` for a genuinely new domain.
- **Documents for agents:** prefer Markdown in git; OnlyOffice UI is weak for `.md`. Use `oo docs put-md` (md→docx upload) and `oo docs as-md` (download→OCR if needed→markdown). Local converters live in `internal/docpipe` (pandoc / ocrmypdf / pdftotext).
- Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command. - Every table output goes through `printTable(headers, rows)`; every single-object through `printObject(v)`. Do not `fmt.Println` rows ad-hoc or the `--output json` flag breaks for that command.
- No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only. - No secrets in the repo; use `.env` (gitignored). Commit `.env.example` only.
- Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`. - Follow SemVer on tags; this repo is tagged at GitHub under `git@github.com:eSlider/go-onlyoffice.git`.
+314
View File
@@ -0,0 +1,314 @@
package main
import (
"fmt"
"os"
"path/filepath"
"strings"
onlyoffice "github.com/eslider/go-onlyoffice"
"github.com/eslider/go-onlyoffice/internal/docpipe"
"github.com/spf13/cobra"
)
func init() {
rootCmd.AddCommand(docsCmd())
}
func docsCmd() *cobra.Command {
cmd := &cobra.Command{
Use: "docs",
Short: "Local document pipeline: md↔docx, OCR→PDF, extract Markdown",
Long: `Agent-friendly conversions (requires pandoc / ocrmypdf / pdftotext on PATH).
OnlyOffice Documents UI is poor for .md — keep Markdown in git, store .docx in OO.
Upload Markdown as DOCX: oo docs put-md PROJECT_ID file.md
Read an OO file as MD: oo docs as-md FILE_ID
OCR a scan locally: oo docs ocr scan.pdf --md out.md`,
}
cmd.AddCommand(docsConvertCmd())
cmd.AddCommand(docsOCRCmd())
cmd.AddCommand(docsAsMDCmd())
cmd.AddCommand(docsPutMDCmd())
cmd.AddCommand(docsToolsCmd())
return cmd
}
func docsToolsCmd() *cobra.Command {
return &cobra.Command{
Use: "tools",
Short: "Show which converter binaries are on PATH",
RunE: func(cmd *cobra.Command, args []string) error {
t := docpipe.LookPath()
printObject(map[string]any{
"pandoc": strOrNil(t.Pandoc),
"ocrmypdf": strOrNil(t.OCRMyPDF),
"pdftotext": strOrNil(t.PDFToText),
"tesseract": strOrNil(t.Tesseract),
})
return nil
},
}
}
func strOrNil(s string) any {
if s == "" {
return nil
}
return s
}
func docsConvertCmd() *cobra.Command {
var to string
cmd := &cobra.Command{
Use: "convert PATH",
Short: "Convert a local file with pandoc (md↔docx by default)",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
in := args[0]
t := docpipe.LookPath()
out := to
if out == "" {
switch docpipe.Ext(in) {
case ".md", ".markdown":
out = docpipe.SiblingDOCX(in)
case ".docx":
out = strings.TrimSuffix(in, docpipe.Ext(in)) + ".md"
default:
return fmt.Errorf("--to required for input type %s", docpipe.Ext(in))
}
}
if err := docpipe.EnsureDir(out); err != nil {
return err
}
if err := t.ConvertFile(in, out); err != nil {
return err
}
printObject(map[string]any{"in": in, "out": out})
return nil
},
}
cmd.Flags().StringVar(&to, "to", "", "output path (default: sibling .docx or .md)")
return cmd
}
func docsOCRCmd() *cobra.Command {
var out, mdOut, lang string
var force bool
var writeMD bool
cmd := &cobra.Command{
Use: "ocr PATH",
Short: "OCR image/PDF → searchable PDF (and optional Markdown)",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
in := args[0]
t := docpipe.LookPath()
if out == "" {
base := strings.TrimSuffix(filepath.Base(in), filepath.Ext(in))
out = filepath.Join(filepath.Dir(in), base+".ocr.pdf")
}
if err := docpipe.EnsureDir(out); err != nil {
return err
}
if err := t.OCRToPDF(in, out, force, lang); err != nil {
return err
}
res := map[string]any{"in": in, "pdf": out}
if writeMD || mdOut != "" {
if mdOut == "" {
mdOut = strings.TrimSuffix(out, filepath.Ext(out)) + ".md"
}
text, err := t.ExtractPDFText(out)
if err != nil {
return err
}
body := "# " + filepath.Base(in) + "\n\n" + strings.TrimSpace(text) + "\n"
if err := os.WriteFile(mdOut, []byte(body), 0o644); err != nil {
return err
}
res["md"] = mdOut
}
printObject(res)
return nil
},
}
cmd.Flags().StringVar(&out, "out", "", "output searchable PDF (default: <name>.ocr.pdf)")
cmd.Flags().StringVar(&mdOut, "md", "", "write Markdown extraction to this path")
cmd.Flags().BoolVar(&writeMD, "markdown", false, "also write sibling .md next to OCR PDF")
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language(s) for tesseract/ocrmypdf")
cmd.Flags().BoolVar(&force, "force", true, "force OCR even if a text layer exists")
return cmd
}
func docsAsMDCmd() *cobra.Command {
var to, lang string
var minChars int
var uploadOCR bool
cmd := &cobra.Command{
Use: "as-md FILE_ID",
Short: "Download an OO Documents file and emit Markdown (OCR PDF/image if needed)",
Args: cobra.ExactArgs(1),
RunE: func(cmd *cobra.Command, args []string) error {
c, err := newOO(cmd)
if err != nil {
return err
}
ctx := cmd.Context()
meta, err := c.GetFile(ctx, args[0])
if err != nil {
return err
}
title := onlyoffice.FileEntryTitle(meta)
dir, err := os.MkdirTemp("", "oo-docs-as-md-*")
if err != nil {
return err
}
defer os.RemoveAll(dir)
local := filepath.Join(dir, onlyoffice.SafeLocalFileName(title))
f, err := os.Create(local)
if err != nil {
return err
}
if _, err := c.DownloadFile(ctx, args[0], f); err != nil {
_ = f.Close()
return err
}
_ = f.Close()
tools := docpipe.LookPath()
res, err := tools.ToMarkdown(local, dir, lang, minChars)
if err != nil {
return err
}
outPath := to
if outPath == "" {
base := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title)))
if base == "" || base == "download" {
base = "file-" + args[0]
}
outPath = base + ".md"
}
if err := docpipe.EnsureDir(outPath); err != nil {
return err
}
if err := os.WriteFile(outPath, []byte(res.Markdown), 0o644); err != nil {
return err
}
obj := map[string]any{
"file_id": args[0],
"title": title,
"md": outPath,
"did_ocr": res.DidOCR,
}
if res.OCRPDFPath != "" && uploadOCR {
folderID := onlyoffice.FileFolderID(meta)
upName := strings.TrimSuffix(onlyoffice.SafeLocalFileName(title), filepath.Ext(onlyoffice.SafeLocalFileName(title))) + ".ocr.pdf"
tmpUp := filepath.Join(dir, upName)
data, err := os.ReadFile(res.OCRPDFPath)
if err != nil {
return err
}
if err := os.WriteFile(tmpUp, data, 0o644); err != nil {
return err
}
if folderID == "" {
obj["ocr_pdf_local"] = res.OCRPDFPath
obj["note"] = "file has no folderId; OCR PDF left local — pass after moving into a folder"
} else {
ent, err := c.UploadToFolder(ctx, folderID, tmpUp)
if err != nil {
return err
}
obj["ocr_pdf_file_id"] = fileIDStr(ent)
obj["ocr_pdf_title"] = onlyoffice.FileEntryTitle(ent)
}
} else if res.OCRPDFPath != "" {
// Keep OCR PDF outside temp by copying beside md if requested via env-less default:
kept := strings.TrimSuffix(outPath, filepath.Ext(outPath)) + ".ocr.pdf"
if b, err := os.ReadFile(res.OCRPDFPath); err == nil {
_ = os.WriteFile(kept, b, 0o644)
obj["ocr_pdf_local"] = kept
} else {
obj["ocr_pdf_local"] = res.OCRPDFPath
}
}
printObject(obj)
return nil
},
}
cmd.Flags().StringVar(&to, "to", "", "write Markdown to this path (default: ./<title>.md)")
cmd.Flags().StringVar(&lang, "lang", "eng", "OCR language")
cmd.Flags().IntVar(&minChars, "min-chars", docpipe.DefaultMinTextChars, "OCR PDF if text layer shorter than this")
cmd.Flags().BoolVar(&uploadOCR, "upload-ocr", false, "upload searchable OCR PDF back into the same OO folder")
return cmd
}
func docsPutMDCmd() *cobra.Command {
var folderID string
var keepLocalDOCX string
cmd := &cobra.Command{
Use: "put-md PROJECT_ID MARKDOWN_PATH",
Short: "Convert Markdown→DOCX and upload DOCX into a project (OO-friendly)",
Long: `Agents edit .md locally; this uploads .docx so OnlyOffice can open/version it.`,
Args: cobra.ExactArgs(2),
RunE: func(cmd *cobra.Command, args []string) error {
pid, mdPath := args[0], args[1]
c, err := newOO(cmd)
if err != nil {
return err
}
tools := docpipe.LookPath()
dir, err := os.MkdirTemp("", "oo-docs-put-md-*")
if err != nil {
return err
}
defer os.RemoveAll(dir)
docxName := strings.TrimSuffix(filepath.Base(mdPath), filepath.Ext(mdPath)) + ".docx"
docxPath := filepath.Join(dir, docxName)
if err := tools.MDToDOCX(mdPath, docxPath); err != nil {
return err
}
if keepLocalDOCX != "" {
if err := docpipe.EnsureDir(keepLocalDOCX); err != nil {
return err
}
b, err := os.ReadFile(docxPath)
if err != nil {
return err
}
if err := os.WriteFile(keepLocalDOCX, b, 0o644); err != nil {
return err
}
}
ctx := cmd.Context()
if folderID != "" {
ent, err := c.UploadToFolder(ctx, folderID, docxPath)
if err != nil {
return err
}
printObject(map[string]any{
"project_id": pid,
"folder_id": folderID,
"md": mdPath,
"uploaded": fileEntryToMap(ent),
})
return nil
}
ent, err := c.UploadProjectFile(ctx, pid, docxPath)
if err != nil {
return err
}
printObject(map[string]any{
"project_id": pid,
"md": mdPath,
"uploaded": fileEntryToMap(ent),
})
return nil
},
}
cmd.Flags().StringVar(&folderID, "folder", "", "Documents folder id (default: project root)")
cmd.Flags().StringVar(&keepLocalDOCX, "keep-docx", "", "also write the generated DOCX to this local path")
return cmd
}
+2 -1
View File
@@ -3,7 +3,7 @@
// Command tree is subject-based (mirrors the library split and the `tea` CLI): // Command tree is subject-based (mirrors the library split and the `tea` CLI):
// //
// oo calendar list | events | add | delete // oo calendar list | events | add | delete
// oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete) // oo projects list | get | milestones | create | update | delete | files (list|upload|download|rename|delete|as-md|put-md)
// oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach) // oo tasks list | get | create | update | delete | subtask add | files (list|upload|detach)
// oo users list | self (alias: oo whoami) // oo users list | self (alias: oo whoami)
// oo contacts list | get | delete | info-add | merge | dedupe-info // oo contacts list | get | delete | info-add | merge | dedupe-info
@@ -15,6 +15,7 @@
// oo crm cleanup // oo crm cleanup
// oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | delete // oo mails accounts | folders | list | get | download-attachment | draft | attach | draft-invoice | delete
// oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items … // oo invoices list | get | create | update | pdf | pdf-cleanup | status | delete | items …
// oo docs tools | convert | ocr | as-md | put-md
// //
// CRM association rules: docs/crm-associations.md // CRM association rules: docs/crm-associations.md
// //
+17
View File
@@ -24,9 +24,26 @@ func projectFilesCmd() *cobra.Command {
cmd.AddCommand(prjFilesDownloadCmd()) cmd.AddCommand(prjFilesDownloadCmd())
cmd.AddCommand(prjFilesRenameCmd()) cmd.AddCommand(prjFilesRenameCmd())
cmd.AddCommand(prjFilesDeleteCmd()) cmd.AddCommand(prjFilesDeleteCmd())
// Convenience aliases into oo docs (md↔docx / OCR pipeline).
cmd.AddCommand(aliasDocsAsMD())
cmd.AddCommand(aliasDocsPutMD())
return cmd return cmd
} }
func aliasDocsAsMD() *cobra.Command {
c := docsAsMDCmd()
c.Use = "as-md FILE_ID"
c.Short = "Alias of `oo docs as-md` — download OO file as Markdown (OCR if needed)"
return c
}
func aliasDocsPutMD() *cobra.Command {
c := docsPutMDCmd()
c.Use = "put-md PROJECT_ID MARKDOWN_PATH"
c.Short = "Alias of `oo docs put-md` — Markdown→DOCX upload into project"
return c
}
func prjFilesListCmd() *cobra.Command { func prjFilesListCmd() *cobra.Command {
var showFolders bool var showFolders bool
cmd := &cobra.Command{ cmd := &cobra.Command{
+68
View File
@@ -281,6 +281,74 @@ func (c *Client) DeleteFiles(ctx context.Context, fileIDs []int) error {
return err return err
} }
// ListFolder returns the Documents module listing for a folder id
// (GET /api/2.0/files/{folderId}).
func (c *Client) ListFolder(ctx context.Context, folderID string) (map[string]any, error) {
if folderID == "" {
return nil, fmt.Errorf("folder id is required")
}
out, err := c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID)+".json")
if err != nil {
out, err = c.ResponseObject(ctx, "/api/2.0/files/"+url.PathEscape(folderID))
}
return out, err
}
// CreateFolder creates a subfolder under parentFolderID.
func (c *Client) CreateFolder(ctx context.Context, parentFolderID, title string) (map[string]any, error) {
if parentFolderID == "" || title == "" {
return nil, fmt.Errorf("parent folder id and title are required")
}
body := map[string]any{"title": title}
out, err := c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID)+".json", body)
if err != nil {
out, err = c.postJSONObject(ctx, "/api/2.0/files/folder/"+url.PathEscape(parentFolderID), body)
}
return out, err
}
// MoveFiles moves file ids into destFolderID (Documents fileops/move).
func (c *Client) MoveFiles(ctx context.Context, destFolderID int, fileIDs []int) (map[string]any, error) {
if destFolderID == 0 || len(fileIDs) == 0 {
return nil, fmt.Errorf("dest folder and file ids are required")
}
body := map[string]any{
"folderIds": []int{},
"fileIds": fileIDs,
"destFolderId": destFolderID,
}
out, err := c.putJSONObject(ctx, "/api/2.0/files/fileops/move.json", body)
if err != nil {
out, err = c.putJSONObject(ctx, "/api/2.0/files/fileops/move", body)
}
return out, err
}
// UploadToFolder uploads a local file into an arbitrary Documents folder id.
func (c *Client) UploadToFolder(ctx context.Context, folderID, localPath string) (*FileEntry, error) {
if folderID == "" || localPath == "" {
return nil, fmt.Errorf("folder id and local path are required")
}
uploadPath := fmt.Sprintf("/api/2.0/files/%s/upload.json", url.PathEscape(folderID))
raw, err := c.uploadMultipart(ctx, uploadPath, "file", localPath)
if err != nil {
uploadPath = fmt.Sprintf("/api/2.0/files/%s/upload", url.PathEscape(folderID))
raw, err = c.uploadMultipart(ctx, uploadPath, "file", localPath)
if err != nil {
return nil, err
}
}
return decodeResponseFileEntry(raw)
}
// FileFolderID returns the parent folder id string for a file entry, if known.
func FileFolderID(f *FileEntry) string {
if f == nil || f.FolderID == nil {
return ""
}
return f.FolderID.String()
}
// DownloadFile streams file bytes from the file's viewUrl using the same auth // DownloadFile streams file bytes from the file's viewUrl using the same auth
// as API calls. Writes into dst. // as API calls. Writes into dst.
func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) { func (c *Client) DownloadFile(ctx context.Context, fileID string, dst io.Writer) (int64, error) {
+311
View File
@@ -0,0 +1,311 @@
// Package docpipe converts documents for OnlyOffice agent workflows:
// Markdown ↔ DOCX (pandoc) and image/PDF OCR → searchable PDF + Markdown text.
//
// External tools (optional at runtime; helpers skip/error clearly when missing):
// - pandoc — md↔docx
// - ocrmypdf — OCR into a searchable PDF
// - pdftotext — extract text layer
// - tesseract — OCR single images when ocrmypdf is unsuitable
package docpipe
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
)
// DefaultMinTextChars: below this, a PDF is treated as needing OCR.
const DefaultMinTextChars = 200
// Tools reports which converters are available on PATH.
type Tools struct {
Pandoc string
OCRMyPDF string
PDFToText string
Tesseract string
}
// LookPath resolves converter binaries (empty string if missing).
func LookPath() Tools {
find := func(names ...string) string {
for _, n := range names {
if p, err := exec.LookPath(n); err == nil {
return p
}
}
return ""
}
return Tools{
Pandoc: find("pandoc"),
OCRMyPDF: find("ocrmypdf"),
PDFToText: find("pdftotext"),
Tesseract: find("tesseract"),
}
}
func (t Tools) requirePandoc() error {
if t.Pandoc == "" {
return fmt.Errorf("pandoc not found on PATH (needed for md↔docx)")
}
return nil
}
// Ext returns lower-case extension including dot (".pdf").
func Ext(path string) string {
return strings.ToLower(filepath.Ext(path))
}
// ConvertFile converts between md and docx (and other pandoc formats) via pandoc.
// outExt may be ".md", ".docx", or a full output path.
func (t Tools) ConvertFile(inPath, outPath string) error {
if err := t.requirePandoc(); err != nil {
return err
}
if strings.TrimSpace(outPath) == "" {
return fmt.Errorf("output path required")
}
cmd := exec.Command(t.Pandoc, inPath, "-o", outPath)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
return fmt.Errorf("pandoc %s → %s: %w (%s)", inPath, outPath, err, strings.TrimSpace(stderr.String()))
}
return nil
}
// MDToDOCX writes a DOCX next to or at outPath from a Markdown file.
func (t Tools) MDToDOCX(mdPath, docxPath string) error {
if Ext(mdPath) != ".md" && Ext(mdPath) != ".markdown" {
return fmt.Errorf("expected markdown input, got %q", mdPath)
}
if docxPath == "" {
docxPath = strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
}
return t.ConvertFile(mdPath, docxPath)
}
// DOCXToMD writes Markdown from a DOCX file.
func (t Tools) DOCXToMD(docxPath, mdPath string) error {
if Ext(docxPath) != ".docx" {
return fmt.Errorf("expected .docx input, got %q", docxPath)
}
if mdPath == "" {
mdPath = strings.TrimSuffix(docxPath, Ext(docxPath)) + ".md"
}
return t.ConvertFile(docxPath, mdPath)
}
// PDFTextLayerChars returns approximate extracted character count (0 if unavailable).
func (t Tools) PDFTextLayerChars(pdfPath string) (int, error) {
if t.PDFToText == "" {
return 0, fmt.Errorf("pdftotext not found on PATH")
}
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
out, err := cmd.Output()
if err != nil {
return 0, err
}
return len(bytes.TrimSpace(out)), nil
}
// NeedsOCR reports whether path likely needs OCR before text extraction.
func (t Tools) NeedsOCR(path string, minChars int) (bool, error) {
if minChars <= 0 {
minChars = DefaultMinTextChars
}
switch Ext(path) {
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
return true, nil
case ".pdf":
n, err := t.PDFTextLayerChars(path)
if err != nil {
// If we cannot measure, prefer OCR.
return true, nil
}
return n < minChars, nil
default:
return false, nil
}
}
// OCRToPDF runs ocrmypdf into outPDF (searchable). Forces OCR when force is true.
func (t Tools) OCRToPDF(inPath, outPDF string, force bool, lang string) error {
if t.OCRMyPDF == "" {
return fmt.Errorf("ocrmypdf not found on PATH")
}
if outPDF == "" {
return fmt.Errorf("output PDF path required")
}
if lang == "" {
lang = "eng"
}
args := []string{"-l", lang, "--skip-big", "100"}
if force {
args = append(args, "--force-ocr")
} else {
args = append(args, "--skip-text")
}
args = append(args, inPath, outPDF)
cmd := exec.Command(t.OCRMyPDF, args...)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
// Retry with force if skip-text refused.
if !force && strings.Contains(stderr.String(), "PriorOcrFoundError") == false {
args2 := []string{"-l", lang, "--force-ocr", inPath, outPDF}
cmd2 := exec.Command(t.OCRMyPDF, args2...)
var stderr2 bytes.Buffer
cmd2.Stderr = &stderr2
if err2 := cmd2.Run(); err2 == nil {
return nil
}
}
return fmt.Errorf("ocrmypdf: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
return nil
}
// ImageToText OCRs a raster image with tesseract (stdout text).
func (t Tools) ImageToText(imgPath, lang string) (string, error) {
if t.Tesseract == "" {
return "", fmt.Errorf("tesseract not found on PATH")
}
if lang == "" {
lang = "eng"
}
cmd := exec.Command(t.Tesseract, imgPath, "stdout", "-l", lang)
var stderr bytes.Buffer
cmd.Stderr = &stderr
out, err := cmd.Output()
if err != nil {
return "", fmt.Errorf("tesseract: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
return string(out), nil
}
// ExtractPDFText returns layout text from a PDF via pdftotext.
func (t Tools) ExtractPDFText(pdfPath string) (string, error) {
if t.PDFToText == "" {
return "", fmt.Errorf("pdftotext not found on PATH")
}
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
out, err := cmd.Output()
if err != nil {
return "", err
}
return string(out), nil
}
// Result of ToMarkdown.
type Result struct {
Markdown string
OCRPDFPath string // set when a searchable PDF was produced
DidOCR bool
Source string
}
// ToMarkdown turns a local file into Markdown text.
// PDFs/images with weak/no text layer are OCR'd to a searchable PDF first (when tools exist).
func (t Tools) ToMarkdown(path string, workDir string, lang string, minChars int) (Result, error) {
res := Result{Source: path}
ext := Ext(path)
switch ext {
case ".md", ".markdown", ".txt":
b, err := os.ReadFile(path)
if err != nil {
return res, err
}
res.Markdown = string(b)
return res, nil
case ".docx", ".odt", ".rtf", ".html", ".htm":
if err := t.requirePandoc(); err != nil {
return res, err
}
tmp := filepath.Join(workDir, "out.md")
if err := t.ConvertFile(path, tmp); err != nil {
return res, err
}
b, err := os.ReadFile(tmp)
if err != nil {
return res, err
}
res.Markdown = string(b)
return res, nil
case ".pdf":
need, _ := t.NeedsOCR(path, minChars)
pdf := path
if need {
if workDir == "" {
workDir = os.TempDir()
}
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
if err := t.OCRToPDF(path, outPDF, true, lang); err != nil {
return res, err
}
res.DidOCR = true
res.OCRPDFPath = outPDF
pdf = outPDF
}
text, err := t.ExtractPDFText(pdf)
if err != nil {
return res, err
}
res.Markdown = wrapMD(filepath.Base(path), text)
return res, nil
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
if workDir == "" {
workDir = os.TempDir()
}
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
if t.OCRMyPDF != "" {
if err := t.OCRToPDF(path, outPDF, true, lang); err == nil {
res.DidOCR = true
res.OCRPDFPath = outPDF
text, err := t.ExtractPDFText(outPDF)
if err != nil {
return res, err
}
res.Markdown = wrapMD(filepath.Base(path), text)
return res, nil
}
}
text, err := t.ImageToText(path, lang)
if err != nil {
return res, err
}
res.DidOCR = true
res.Markdown = wrapMD(filepath.Base(path), text)
return res, nil
default:
return res, fmt.Errorf("unsupported type %q for markdown extraction", ext)
}
}
func wrapMD(title, body string) string {
body = strings.TrimSpace(body)
if body == "" {
return "# " + title + "\n\n_(empty text layer)_\n"
}
return "# " + title + "\n\n" + body + "\n"
}
func trimExt(name string) string {
return strings.TrimSuffix(name, filepath.Ext(name))
}
// SiblingDOCX returns path with .docx extension replacing the original ext.
func SiblingDOCX(mdPath string) string {
return strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
}
// EnsureDir creates parent directories for path.
func EnsureDir(path string) error {
dir := filepath.Dir(path)
if dir == "" || dir == "." {
return nil
}
return os.MkdirAll(dir, 0o755)
}
+76
View File
@@ -0,0 +1,76 @@
package docpipe
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestExt(t *testing.T) {
if Ext("Foo.PDF") != ".pdf" {
t.Fatalf("Ext: %q", Ext("Foo.PDF"))
}
}
func TestSiblingDOCX(t *testing.T) {
if got := SiblingDOCX("notes.md"); got != "notes.docx" {
t.Fatalf("got %q", got)
}
}
func TestWrapMD(t *testing.T) {
s := wrapMD("a.pdf", " hello ")
if !strings.HasPrefix(s, "# a.pdf\n") || !strings.Contains(s, "hello") {
t.Fatalf("wrap: %q", s)
}
}
func TestNeedsOCR_Image(t *testing.T) {
tools := LookPath()
need, err := tools.NeedsOCR("x.jpg", 0)
if err != nil || !need {
t.Fatalf("jpg should need OCR: need=%v err=%v", need, err)
}
}
func TestMDDocxRoundTrip(t *testing.T) {
tools := LookPath()
if tools.Pandoc == "" {
t.Skip("pandoc not installed")
}
dir := t.TempDir()
md := filepath.Join(dir, "n.md")
docx := filepath.Join(dir, "n.docx")
md2 := filepath.Join(dir, "n2.md")
if err := os.WriteFile(md, []byte("# Title\n\nHello **world**.\n"), 0o644); err != nil {
t.Fatal(err)
}
if err := tools.MDToDOCX(md, docx); err != nil {
t.Fatal(err)
}
if _, err := os.Stat(docx); err != nil {
t.Fatal(err)
}
if err := tools.DOCXToMD(docx, md2); err != nil {
t.Fatal(err)
}
b, err := os.ReadFile(md2)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(b), "Hello") {
t.Fatalf("round-trip missing Hello: %s", b)
}
}
func TestToMarkdown_PlainMD(t *testing.T) {
tools := LookPath()
dir := t.TempDir()
p := filepath.Join(dir, "a.md")
_ = os.WriteFile(p, []byte("hi"), 0o644)
res, err := tools.ToMarkdown(p, dir, "eng", 0)
if err != nil || res.Markdown != "hi" {
t.Fatalf("got %+v err=%v", res, err)
}
}