feat(docs): md↔docx convert, OCR→PDF, as-md/put-md for agents

Add oo docs pipeline (pandoc/ocrmypdf) so agents keep Markdown locally while OnlyOffice stores versioned DOCX; OCR weak PDFs/images before returning MD. Also expose ListFolder/CreateFolder/MoveFiles/UploadToFolder.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-08-27 16:49:06 +01:00
co-authored by Cursor
parent 50d8974154
commit 6ea2fbadf7
7 changed files with 789 additions and 1 deletions
+311
View File
@@ -0,0 +1,311 @@
// Package docpipe converts documents for OnlyOffice agent workflows:
// Markdown ↔ DOCX (pandoc) and image/PDF OCR → searchable PDF + Markdown text.
//
// External tools (optional at runtime; helpers skip/error clearly when missing):
// - pandoc — md↔docx
// - ocrmypdf — OCR into a searchable PDF
// - pdftotext — extract text layer
// - tesseract — OCR single images when ocrmypdf is unsuitable
package docpipe
import (
"bytes"
"fmt"
"os"
"os/exec"
"path/filepath"
"strings"
)
// DefaultMinTextChars: below this, a PDF is treated as needing OCR.
const DefaultMinTextChars = 200
// Tools reports which converters are available on PATH.
type Tools struct {
Pandoc string
OCRMyPDF string
PDFToText string
Tesseract string
}
// LookPath resolves converter binaries (empty string if missing).
func LookPath() Tools {
find := func(names ...string) string {
for _, n := range names {
if p, err := exec.LookPath(n); err == nil {
return p
}
}
return ""
}
return Tools{
Pandoc: find("pandoc"),
OCRMyPDF: find("ocrmypdf"),
PDFToText: find("pdftotext"),
Tesseract: find("tesseract"),
}
}
func (t Tools) requirePandoc() error {
if t.Pandoc == "" {
return fmt.Errorf("pandoc not found on PATH (needed for md↔docx)")
}
return nil
}
// Ext returns lower-case extension including dot (".pdf").
func Ext(path string) string {
return strings.ToLower(filepath.Ext(path))
}
// ConvertFile converts between md and docx (and other pandoc formats) via pandoc.
// outExt may be ".md", ".docx", or a full output path.
func (t Tools) ConvertFile(inPath, outPath string) error {
if err := t.requirePandoc(); err != nil {
return err
}
if strings.TrimSpace(outPath) == "" {
return fmt.Errorf("output path required")
}
cmd := exec.Command(t.Pandoc, inPath, "-o", outPath)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
return fmt.Errorf("pandoc %s → %s: %w (%s)", inPath, outPath, err, strings.TrimSpace(stderr.String()))
}
return nil
}
// MDToDOCX writes a DOCX next to or at outPath from a Markdown file.
func (t Tools) MDToDOCX(mdPath, docxPath string) error {
if Ext(mdPath) != ".md" && Ext(mdPath) != ".markdown" {
return fmt.Errorf("expected markdown input, got %q", mdPath)
}
if docxPath == "" {
docxPath = strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
}
return t.ConvertFile(mdPath, docxPath)
}
// DOCXToMD writes Markdown from a DOCX file.
func (t Tools) DOCXToMD(docxPath, mdPath string) error {
if Ext(docxPath) != ".docx" {
return fmt.Errorf("expected .docx input, got %q", docxPath)
}
if mdPath == "" {
mdPath = strings.TrimSuffix(docxPath, Ext(docxPath)) + ".md"
}
return t.ConvertFile(docxPath, mdPath)
}
// PDFTextLayerChars returns approximate extracted character count (0 if unavailable).
func (t Tools) PDFTextLayerChars(pdfPath string) (int, error) {
if t.PDFToText == "" {
return 0, fmt.Errorf("pdftotext not found on PATH")
}
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
out, err := cmd.Output()
if err != nil {
return 0, err
}
return len(bytes.TrimSpace(out)), nil
}
// NeedsOCR reports whether path likely needs OCR before text extraction.
func (t Tools) NeedsOCR(path string, minChars int) (bool, error) {
if minChars <= 0 {
minChars = DefaultMinTextChars
}
switch Ext(path) {
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
return true, nil
case ".pdf":
n, err := t.PDFTextLayerChars(path)
if err != nil {
// If we cannot measure, prefer OCR.
return true, nil
}
return n < minChars, nil
default:
return false, nil
}
}
// OCRToPDF runs ocrmypdf into outPDF (searchable). Forces OCR when force is true.
func (t Tools) OCRToPDF(inPath, outPDF string, force bool, lang string) error {
if t.OCRMyPDF == "" {
return fmt.Errorf("ocrmypdf not found on PATH")
}
if outPDF == "" {
return fmt.Errorf("output PDF path required")
}
if lang == "" {
lang = "eng"
}
args := []string{"-l", lang, "--skip-big", "100"}
if force {
args = append(args, "--force-ocr")
} else {
args = append(args, "--skip-text")
}
args = append(args, inPath, outPDF)
cmd := exec.Command(t.OCRMyPDF, args...)
var stderr bytes.Buffer
cmd.Stderr = &stderr
if err := cmd.Run(); err != nil {
// Retry with force if skip-text refused.
if !force && strings.Contains(stderr.String(), "PriorOcrFoundError") == false {
args2 := []string{"-l", lang, "--force-ocr", inPath, outPDF}
cmd2 := exec.Command(t.OCRMyPDF, args2...)
var stderr2 bytes.Buffer
cmd2.Stderr = &stderr2
if err2 := cmd2.Run(); err2 == nil {
return nil
}
}
return fmt.Errorf("ocrmypdf: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
return nil
}
// ImageToText OCRs a raster image with tesseract (stdout text).
func (t Tools) ImageToText(imgPath, lang string) (string, error) {
if t.Tesseract == "" {
return "", fmt.Errorf("tesseract not found on PATH")
}
if lang == "" {
lang = "eng"
}
cmd := exec.Command(t.Tesseract, imgPath, "stdout", "-l", lang)
var stderr bytes.Buffer
cmd.Stderr = &stderr
out, err := cmd.Output()
if err != nil {
return "", fmt.Errorf("tesseract: %w (%s)", err, strings.TrimSpace(stderr.String()))
}
return string(out), nil
}
// ExtractPDFText returns layout text from a PDF via pdftotext.
func (t Tools) ExtractPDFText(pdfPath string) (string, error) {
if t.PDFToText == "" {
return "", fmt.Errorf("pdftotext not found on PATH")
}
cmd := exec.Command(t.PDFToText, "-layout", pdfPath, "-")
out, err := cmd.Output()
if err != nil {
return "", err
}
return string(out), nil
}
// Result of ToMarkdown.
type Result struct {
Markdown string
OCRPDFPath string // set when a searchable PDF was produced
DidOCR bool
Source string
}
// ToMarkdown turns a local file into Markdown text.
// PDFs/images with weak/no text layer are OCR'd to a searchable PDF first (when tools exist).
func (t Tools) ToMarkdown(path string, workDir string, lang string, minChars int) (Result, error) {
res := Result{Source: path}
ext := Ext(path)
switch ext {
case ".md", ".markdown", ".txt":
b, err := os.ReadFile(path)
if err != nil {
return res, err
}
res.Markdown = string(b)
return res, nil
case ".docx", ".odt", ".rtf", ".html", ".htm":
if err := t.requirePandoc(); err != nil {
return res, err
}
tmp := filepath.Join(workDir, "out.md")
if err := t.ConvertFile(path, tmp); err != nil {
return res, err
}
b, err := os.ReadFile(tmp)
if err != nil {
return res, err
}
res.Markdown = string(b)
return res, nil
case ".pdf":
need, _ := t.NeedsOCR(path, minChars)
pdf := path
if need {
if workDir == "" {
workDir = os.TempDir()
}
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
if err := t.OCRToPDF(path, outPDF, true, lang); err != nil {
return res, err
}
res.DidOCR = true
res.OCRPDFPath = outPDF
pdf = outPDF
}
text, err := t.ExtractPDFText(pdf)
if err != nil {
return res, err
}
res.Markdown = wrapMD(filepath.Base(path), text)
return res, nil
case ".jpg", ".jpeg", ".png", ".tif", ".tiff", ".webp", ".gif", ".bmp":
if workDir == "" {
workDir = os.TempDir()
}
outPDF := filepath.Join(workDir, trimExt(filepath.Base(path))+".ocr.pdf")
if t.OCRMyPDF != "" {
if err := t.OCRToPDF(path, outPDF, true, lang); err == nil {
res.DidOCR = true
res.OCRPDFPath = outPDF
text, err := t.ExtractPDFText(outPDF)
if err != nil {
return res, err
}
res.Markdown = wrapMD(filepath.Base(path), text)
return res, nil
}
}
text, err := t.ImageToText(path, lang)
if err != nil {
return res, err
}
res.DidOCR = true
res.Markdown = wrapMD(filepath.Base(path), text)
return res, nil
default:
return res, fmt.Errorf("unsupported type %q for markdown extraction", ext)
}
}
func wrapMD(title, body string) string {
body = strings.TrimSpace(body)
if body == "" {
return "# " + title + "\n\n_(empty text layer)_\n"
}
return "# " + title + "\n\n" + body + "\n"
}
func trimExt(name string) string {
return strings.TrimSuffix(name, filepath.Ext(name))
}
// SiblingDOCX returns path with .docx extension replacing the original ext.
func SiblingDOCX(mdPath string) string {
return strings.TrimSuffix(mdPath, Ext(mdPath)) + ".docx"
}
// EnsureDir creates parent directories for path.
func EnsureDir(path string) error {
dir := filepath.Dir(path)
if dir == "" || dir == "." {
return nil
}
return os.MkdirAll(dir, 0o755)
}
+76
View File
@@ -0,0 +1,76 @@
package docpipe
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestExt(t *testing.T) {
if Ext("Foo.PDF") != ".pdf" {
t.Fatalf("Ext: %q", Ext("Foo.PDF"))
}
}
func TestSiblingDOCX(t *testing.T) {
if got := SiblingDOCX("notes.md"); got != "notes.docx" {
t.Fatalf("got %q", got)
}
}
func TestWrapMD(t *testing.T) {
s := wrapMD("a.pdf", " hello ")
if !strings.HasPrefix(s, "# a.pdf\n") || !strings.Contains(s, "hello") {
t.Fatalf("wrap: %q", s)
}
}
func TestNeedsOCR_Image(t *testing.T) {
tools := LookPath()
need, err := tools.NeedsOCR("x.jpg", 0)
if err != nil || !need {
t.Fatalf("jpg should need OCR: need=%v err=%v", need, err)
}
}
func TestMDDocxRoundTrip(t *testing.T) {
tools := LookPath()
if tools.Pandoc == "" {
t.Skip("pandoc not installed")
}
dir := t.TempDir()
md := filepath.Join(dir, "n.md")
docx := filepath.Join(dir, "n.docx")
md2 := filepath.Join(dir, "n2.md")
if err := os.WriteFile(md, []byte("# Title\n\nHello **world**.\n"), 0o644); err != nil {
t.Fatal(err)
}
if err := tools.MDToDOCX(md, docx); err != nil {
t.Fatal(err)
}
if _, err := os.Stat(docx); err != nil {
t.Fatal(err)
}
if err := tools.DOCXToMD(docx, md2); err != nil {
t.Fatal(err)
}
b, err := os.ReadFile(md2)
if err != nil {
t.Fatal(err)
}
if !strings.Contains(string(b), "Hello") {
t.Fatalf("round-trip missing Hello: %s", b)
}
}
func TestToMarkdown_PlainMD(t *testing.T) {
tools := LookPath()
dir := t.TempDir()
p := filepath.Join(dir, "a.md")
_ = os.WriteFile(p, []byte("hi"), 0o644)
res, err := tools.ToMarkdown(p, dir, "eng", 0)
if err != nil || res.Markdown != "hi" {
t.Fatalf("got %+v err=%v", res, err)
}
}