feat: OCR scans with tesseract, drop docling from the default path.
pdftotext still wins on born-digital PDFs. Empty text layers go through pdftoppm + tesseract eng+deu. Optional OCR_ENGINE=paddle. Gitea #6.
This commit is contained in:
+11
-77
@@ -7,7 +7,7 @@
|
||||
bin/mail/import --since 2026-01-01 only messages after a date
|
||||
bin/mail/import --limit 50 cap messages per run
|
||||
bin/mail/import --no-attachments body only, skip attachment conversion
|
||||
bin/mail/import --ocr OCR scanned PDFs/images via docling
|
||||
bin/mail/import --ocr OCR images (PDFs OCR when textless)
|
||||
bin/mail/import --dry-run list messages without writing anything
|
||||
|
||||
Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||
@@ -16,10 +16,11 @@ Writes one directory per message: var/mail/{folder}/{message_id}/
|
||||
attachments/*.md converted attachment content
|
||||
|
||||
Indexing is a separate step (`bin/brain/index.go --rebuild`): conversion can
|
||||
crash in native docling and must not leave the brain DB mid-transaction.
|
||||
crash and must not leave the brain DB mid-transaction.
|
||||
|
||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env). Idempotent: a message
|
||||
already present (message.md exists) is skipped unless --force.
|
||||
Requires ONLYOFFICE_URL/USER/PASS in .env (or env) except `--from-raw`.
|
||||
Idempotent: a message already present (message.md exists) is skipped unless
|
||||
--force.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -41,10 +42,10 @@ from mailconv import ( # noqa: E402
|
||||
IMAGE_SUFFIXES,
|
||||
LEGACY_OFFICE_SUFFIXES,
|
||||
TEXT_SUFFIXES,
|
||||
convert_pdf,
|
||||
html_to_markdown,
|
||||
is_convertible,
|
||||
normalize_markdown,
|
||||
subject_to_filename,
|
||||
ocr_image,
|
||||
zip_extract_safe,
|
||||
)
|
||||
|
||||
@@ -147,9 +148,9 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
||||
except Exception as e:
|
||||
return f"\n<!-- conversion failed: {e} -->\n"
|
||||
if suffix == ".pdf":
|
||||
return _convert_pdf(path, ocr)
|
||||
return convert_pdf(path, ocr)
|
||||
if suffix in IMAGE_SUFFIXES and ocr:
|
||||
return _convert_pdf(path, ocr)
|
||||
return ocr_image(path) or "\n<!-- ocr unavailable -->\n"
|
||||
if suffix in LEGACY_OFFICE_SUFFIXES:
|
||||
return _convert_legacy(path)
|
||||
if suffix in ARCHIVE_SUFFIXES:
|
||||
@@ -157,67 +158,6 @@ def convert_file_to_md(path: Path, ocr: bool) -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
def _convert_pdf(path: Path, ocr: bool) -> str:
|
||||
"""Convert one PDF to markdown.
|
||||
|
||||
Fast path: poppler's pdftotext (-layout) extracts exact text from
|
||||
born-digital PDFs in ~15ms vs docling's 1-3s. Only textless PDFs (scanned
|
||||
pages, layout-heavy) fall back to docling, which runs isolated in a
|
||||
subprocess because its native onnx/RT-DETR has segfaulted the main process.
|
||||
"""
|
||||
text = _pdf_fast_text(path)
|
||||
if ocr or text is None or not text.strip():
|
||||
return _convert_pdf_docling(path, ocr)
|
||||
return normalize_markdown(text)
|
||||
|
||||
|
||||
def _pdf_fast_text(path: Path) -> str | None:
|
||||
"""pdftotext -layout; None when poppler is unavailable (or the PDF has no text layer)."""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["pdftotext", "-layout", str(path), "-"],
|
||||
capture_output=True, timeout=60)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if proc.returncode != 0:
|
||||
return None
|
||||
return proc.stdout.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
def _convert_pdf_docling(path: Path, ocr: bool) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[sys.executable, os.path.abspath(__file__), "--pdf-worker", str(path),
|
||||
"--ocr" if ocr else "--no-ocr"],
|
||||
capture_output=True, text=True, timeout=600)
|
||||
except subprocess.TimeoutExpired:
|
||||
return "\n<!-- pdf conversion timed out -->\n"
|
||||
if proc.returncode != 0:
|
||||
tail = proc.stderr.strip().splitlines()[-3:]
|
||||
return f"\n<!-- pdf conversion failed: {proc.returncode}: {' | '.join(tail)} -->\n"
|
||||
return proc.stdout
|
||||
|
||||
|
||||
def _pdf_worker(path: Path, ocr: bool) -> None:
|
||||
"""docling worker entry: prints converted markdown on stdout, exits non-zero on error."""
|
||||
try:
|
||||
from docling.document_converter import DocumentConverter, PdfFormatOption
|
||||
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
||||
opts = PdfPipelineOptions()
|
||||
opts.do_ocr = bool(ocr)
|
||||
opts.do_table_structure = True
|
||||
conv = DocumentConverter(format_options={"pdf": PdfFormatOption(pipeline_options=opts)})
|
||||
res = conv.convert(str(path))
|
||||
sys.stdout.write(normalize_markdown(res.document.export_to_markdown()))
|
||||
sys.exit(0)
|
||||
except Exception as e:
|
||||
# errors/stacktraces to stderr; the caller only reports a one-liner
|
||||
print(f"pdf-worker: {e}", file=sys.stderr)
|
||||
import traceback
|
||||
traceback.print_exc(file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def _convert_legacy(path: Path) -> str:
|
||||
"""Legacy .doc/.xls/.ppt -> md via pandoc (installed) or a stub."""
|
||||
try:
|
||||
@@ -356,19 +296,12 @@ def main(argv: list[str]) -> int:
|
||||
p.add_argument("--from-raw", default="",
|
||||
help="convert Go-synced dirs (var/mail/<folder>/<id>/message.json) to markdown")
|
||||
p.add_argument("--no-attachments", action="store_true", help="skip attachment download+convert")
|
||||
p.add_argument("--ocr", action="store_true", help="OCR scanned PDFs/images via docling")
|
||||
p.add_argument("--ocr", action="store_true", help="OCR images (PDFs OCR when textless)")
|
||||
p.add_argument("--force", action="store_true", help="re-import even if message.md exists")
|
||||
p.add_argument("--dry-run", action="store_true", help="list messages, write nothing")
|
||||
p.add_argument("--json", action="store_true")
|
||||
p.add_argument("--pdf-worker", default="", help=argparse.SUPPRESS)
|
||||
p.add_argument("--no-ocr", action="store_true", help=argparse.SUPPRESS)
|
||||
a = p.parse_args(argv)
|
||||
|
||||
if a.pdf_worker:
|
||||
_pdf_worker(Path(a.pdf_worker), ocr=not a.no_ocr)
|
||||
return 0
|
||||
|
||||
conf = load_env()
|
||||
fid = folder_id(a.folder)
|
||||
out_root = ROOT / "var" / "mail"
|
||||
summary: list[dict] = []
|
||||
@@ -394,6 +327,7 @@ def main(argv: list[str]) -> int:
|
||||
target_dir=msg_dir.parent))
|
||||
summary.append(entry)
|
||||
else:
|
||||
conf = load_env()
|
||||
OOCLIENT = OOClient(conf)
|
||||
if a.id:
|
||||
messages = [{"id": i} for i in a.id]
|
||||
|
||||
Executable
+48
@@ -0,0 +1,48 @@
|
||||
//usr/bin/env go run -tags=mail_ocr "$0" "$@"; exit
|
||||
//go:build mail_ocr
|
||||
//
|
||||
// bin/mail/ocr.go - OCR an image or scanned PDF (tesseract eng+deu).
|
||||
//
|
||||
// ./bin/mail/ocr.go scan.png
|
||||
// ./bin/mail/ocr.go scan.pdf
|
||||
// OCR_ENGINE=paddle ./bin/mail/ocr.go scan.png
|
||||
//
|
||||
// PDFs try pdftotext -layout first; empty text layer uses pdftoppm + tesseract.
|
||||
// No gocv. Tesseract CGO bindings are not used (D21 Zig owns Ladybug CGO).
|
||||
// NOTE: never run `gofmt -w` on this file — it breaks the shebang.
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"strings"
|
||||
|
||||
"github.com/eSlider/2dph/internal/ocr"
|
||||
)
|
||||
|
||||
func main() {
|
||||
os.Exit(run(os.Args[1:]))
|
||||
}
|
||||
|
||||
func run(args []string) int {
|
||||
if len(args) != 1 || strings.HasPrefix(args[0], "-") {
|
||||
fmt.Fprintln(os.Stderr, `usage: bin/mail/ocr.go <image|pdf>`)
|
||||
return 2
|
||||
}
|
||||
path := args[0]
|
||||
var (
|
||||
text string
|
||||
err error
|
||||
)
|
||||
if strings.HasSuffix(strings.ToLower(path), ".pdf") {
|
||||
text, err = ocr.PDFFile(path)
|
||||
} else {
|
||||
text, err = ocr.ImageFile(path)
|
||||
}
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "mail/ocr: %v\n", err)
|
||||
return 1
|
||||
}
|
||||
fmt.Println(text)
|
||||
return 0
|
||||
}
|
||||
+84
-1
@@ -7,7 +7,10 @@ offline against fixtures.
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
@@ -18,9 +21,10 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
|
||||
PDF_SUFFIXES = {".pdf"}
|
||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
||||
ARCHIVE_SUFFIXES = {".zip"}
|
||||
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
|
||||
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
|
||||
# pandoc first, else leave a stub.
|
||||
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
||||
TESS_LANG = "eng+deu"
|
||||
|
||||
CONVERTIBLE_SUFFIXES = (
|
||||
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
||||
@@ -146,3 +150,82 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
|
||||
|
||||
def is_convertible(suffix: str) -> bool:
|
||||
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
||||
|
||||
|
||||
def convert_pdf(path: Path, ocr: bool = False) -> str:
|
||||
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
|
||||
|
||||
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
|
||||
automatically. This path never execs an ONNX document converter.
|
||||
"""
|
||||
del ocr # scans OCR when the text layer is empty; flag is for images
|
||||
text = pdf_fast_text(path)
|
||||
if text and text.strip():
|
||||
return normalize_markdown(text)
|
||||
scanned = ocr_pdf(path)
|
||||
if scanned and scanned.strip():
|
||||
return normalize_markdown(scanned)
|
||||
if text:
|
||||
return normalize_markdown(text)
|
||||
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
|
||||
|
||||
|
||||
def pdf_fast_text(path: Path) -> str | None:
|
||||
"""pdftotext -layout; None when poppler is missing or the command fails."""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["pdftotext", "-layout", str(path), "-"],
|
||||
capture_output=True, timeout=60)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if proc.returncode != 0:
|
||||
return None
|
||||
return proc.stdout.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
def ocr_pdf(path: Path) -> str:
|
||||
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
|
||||
try:
|
||||
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
|
||||
prefix = str(Path(tmp) / "page")
|
||||
proc = subprocess.run(
|
||||
["pdftoppm", "-png", "-r", "200", str(path), prefix],
|
||||
capture_output=True, timeout=120)
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
pages = sorted(Path(tmp).glob("page*.png"))
|
||||
parts = [ocr_image(p) for p in pages]
|
||||
return "\n\n".join(p for p in parts if p and p.strip())
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
|
||||
|
||||
def ocr_image(path: Path) -> str:
|
||||
engine = os.environ.get("OCR_ENGINE", "tesseract")
|
||||
if engine == "paddle":
|
||||
return _ocr_paddle(path)
|
||||
return _ocr_tesseract(path)
|
||||
|
||||
|
||||
def _ocr_tesseract(path: Path) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
|
||||
capture_output=True, timeout=120)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||
|
||||
|
||||
def _ocr_paddle(path: Path) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["paddleocr", "ocr", "-i", str(path)],
|
||||
capture_output=True, timeout=180)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||
|
||||
@@ -136,6 +136,33 @@ class BinLayoutTest(unittest.TestCase):
|
||||
"index_mail must point at bin/brain/index.go",
|
||||
)
|
||||
|
||||
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
|
||||
self._assert_shebang("bin/mail/ocr.go")
|
||||
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
|
||||
self.assertIn("internal/ocr", ocr)
|
||||
self.assertIn("mail_ocr", ocr)
|
||||
self.assertNotIn("github.com/otiai10/gosseract", ocr)
|
||||
py = (ROOT / "bin" / "mail" / "import").read_text()
|
||||
self.assertNotIn("from docling", py)
|
||||
self.assertNotIn("import docling", py)
|
||||
self.assertIn("convert_pdf", py)
|
||||
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
|
||||
self.assertIn("pdftotext", conv)
|
||||
self.assertIn("pdftoppm", conv)
|
||||
self.assertIn("tesseract", conv)
|
||||
self.assertIn("eng+deu", conv)
|
||||
self.assertNotIn("from docling", conv)
|
||||
self.assertNotIn("import docling", conv)
|
||||
self.assertNotIn("gocv", conv.lower())
|
||||
proj = (ROOT / "pyproject.toml").read_text()
|
||||
self.assertNotIn("docling", proj)
|
||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
||||
self.assertIn("tesseract-ocr", ci)
|
||||
self.assertIn("./internal/ocr", ci)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn("ocr-paddle", compose)
|
||||
self.assertIn("OCR_ENGINE", compose)
|
||||
|
||||
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
||||
self._assert_shebang("bin/markdown/import.go")
|
||||
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
||||
|
||||
@@ -8,10 +8,13 @@ from pathlib import Path
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
|
||||
from mailconv import ( # noqa: E402
|
||||
TESS_LANG,
|
||||
clean_email_address,
|
||||
convert_pdf,
|
||||
html_to_markdown,
|
||||
is_convertible,
|
||||
normalize_markdown,
|
||||
ocr_image,
|
||||
split_zip_members,
|
||||
subject_to_filename,
|
||||
zip_extract_safe,
|
||||
@@ -100,6 +103,92 @@ class TestMailConv(unittest.TestCase):
|
||||
self.assertFalse(is_convertible(".exe"))
|
||||
self.assertFalse(is_convertible(".unknown"))
|
||||
|
||||
def test_convert_pdf_prefers_pdftotext(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b"Invoice BM25 layout"
|
||||
stderr = b""
|
||||
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
out = convert_pdf(Path(self._tmp("born.pdf")))
|
||||
self.assertIn("BM25", out)
|
||||
self.assertEqual(calls[0][:2], ["pdftotext", "-layout"])
|
||||
self.assertFalse(any(c[0] == "tesseract" for c in calls))
|
||||
self.assertFalse(any(c[0] == "pdftoppm" for c in calls))
|
||||
|
||||
def test_convert_pdf_empty_layer_uses_pdftoppm_tesseract(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b""
|
||||
stderr = b""
|
||||
|
||||
if cmd[0] == "pdftotext":
|
||||
P.stdout = b" \n"
|
||||
return P()
|
||||
if cmd[0] == "pdftoppm":
|
||||
prefix = Path(cmd[-1])
|
||||
(prefix.parent / "page-1.png").write_bytes(b"fake")
|
||||
return P()
|
||||
if cmd[0] == "tesseract":
|
||||
P.stdout = b"scanned HELLO"
|
||||
return P()
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
out = convert_pdf(Path(self._tmp("scan.pdf")))
|
||||
self.assertIn("HELLO", out)
|
||||
bins = [c[0] for c in calls]
|
||||
self.assertIn("pdftotext", bins)
|
||||
self.assertIn("pdftoppm", bins)
|
||||
self.assertIn("tesseract", bins)
|
||||
tess = next(c for c in calls if c[0] == "tesseract")
|
||||
self.assertIn(TESS_LANG, tess)
|
||||
self.assertNotIn("docling", " ".join(bins))
|
||||
|
||||
def test_ocr_image_paddle_engine(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b"paddle text"
|
||||
stderr = b""
|
||||
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
os.environ["OCR_ENGINE"] = "paddle"
|
||||
try:
|
||||
out = ocr_image(Path(self._tmp("x.png")))
|
||||
finally:
|
||||
os.environ.pop("OCR_ENGINE", None)
|
||||
self.assertEqual(out, "paddle text")
|
||||
self.assertEqual(calls[0][:2], ["paddleocr", "ocr"])
|
||||
|
||||
def _patch_run(self, mod, fn) -> None:
|
||||
self.addCleanup(setattr, mod.subprocess, "run", mod.subprocess.run)
|
||||
mod.subprocess.run = fn
|
||||
|
||||
def _mk_zip(self, members):
|
||||
zpath = Path(self._tmp("arc.zip"))
|
||||
with zipfile.ZipFile(zpath, "w") as zf:
|
||||
|
||||
Reference in New Issue
Block a user