feat: OCR scans with tesseract, drop docling from the default path. (#33)
pdftotext still wins on born-digital PDFs. Empty text layers go through pdftoppm + tesseract eng+deu. Optional OCR_ENGINE=paddle. Gitea #6.
This commit is contained in:
+84
-1
@@ -7,7 +7,10 @@ offline against fixtures.
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
@@ -18,9 +21,10 @@ OFFICE_SUFFIXES = {".docx", ".pptx", ".xlsx", ".html", ".htm", ".epub", ".eml",
|
||||
PDF_SUFFIXES = {".pdf"}
|
||||
IMAGE_SUFFIXES = {".png", ".jpg", ".jpeg", ".gif", ".bmp", ".tiff", ".tif", ".webp"}
|
||||
ARCHIVE_SUFFIXES = {".zip"}
|
||||
# Legacy binary Office (doc/xls/ppt) — markitdown/docling skip them; we try
|
||||
# Legacy binary Office (doc/xls/ppt) — markitdown skip them; we try
|
||||
# pandoc first, else leave a stub.
|
||||
LEGACY_OFFICE_SUFFIXES = {".doc", ".xls", ".ppt"}
|
||||
TESS_LANG = "eng+deu"
|
||||
|
||||
CONVERTIBLE_SUFFIXES = (
|
||||
TEXT_SUFFIXES | OFFICE_SUFFIXES | PDF_SUFFIXES | IMAGE_SUFFIXES | ARCHIVE_SUFFIXES | LEGACY_OFFICE_SUFFIXES
|
||||
@@ -146,3 +150,82 @@ def zip_extract_safe(zip_path: Path, dest: Path) -> list[Path]:
|
||||
|
||||
def is_convertible(suffix: str) -> bool:
|
||||
return suffix.lower() in CONVERTIBLE_SUFFIXES
|
||||
|
||||
|
||||
def convert_pdf(path: Path, ocr: bool = False) -> str:
|
||||
"""pdftotext -layout first; empty text layer → pdftoppm + tesseract.
|
||||
|
||||
`ocr` is unused for born-digital PDFs (text layer wins). Scans OCR
|
||||
automatically. This path never execs an ONNX document converter.
|
||||
"""
|
||||
del ocr # scans OCR when the text layer is empty; flag is for images
|
||||
text = pdf_fast_text(path)
|
||||
if text and text.strip():
|
||||
return normalize_markdown(text)
|
||||
scanned = ocr_pdf(path)
|
||||
if scanned and scanned.strip():
|
||||
return normalize_markdown(scanned)
|
||||
if text:
|
||||
return normalize_markdown(text)
|
||||
return "\n<!-- pdf has no text layer (ocr unavailable) -->\n"
|
||||
|
||||
|
||||
def pdf_fast_text(path: Path) -> str | None:
|
||||
"""pdftotext -layout; None when poppler is missing or the command fails."""
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["pdftotext", "-layout", str(path), "-"],
|
||||
capture_output=True, timeout=60)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return None
|
||||
if proc.returncode != 0:
|
||||
return None
|
||||
return proc.stdout.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
def ocr_pdf(path: Path) -> str:
|
||||
"""Rasterize with pdftoppm and OCR each page (tesseract or paddle)."""
|
||||
try:
|
||||
with tempfile.TemporaryDirectory(prefix="2dph-ocr-") as tmp:
|
||||
prefix = str(Path(tmp) / "page")
|
||||
proc = subprocess.run(
|
||||
["pdftoppm", "-png", "-r", "200", str(path), prefix],
|
||||
capture_output=True, timeout=120)
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
pages = sorted(Path(tmp).glob("page*.png"))
|
||||
parts = [ocr_image(p) for p in pages]
|
||||
return "\n\n".join(p for p in parts if p and p.strip())
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
|
||||
|
||||
def ocr_image(path: Path) -> str:
|
||||
engine = os.environ.get("OCR_ENGINE", "tesseract")
|
||||
if engine == "paddle":
|
||||
return _ocr_paddle(path)
|
||||
return _ocr_tesseract(path)
|
||||
|
||||
|
||||
def _ocr_tesseract(path: Path) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["tesseract", str(path), "stdout", "-l", TESS_LANG, "--psm", "6"],
|
||||
capture_output=True, timeout=120)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||
|
||||
|
||||
def _ocr_paddle(path: Path) -> str:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
["paddleocr", "ocr", "-i", str(path)],
|
||||
capture_output=True, timeout=180)
|
||||
except (OSError, subprocess.TimeoutExpired):
|
||||
return ""
|
||||
if proc.returncode != 0:
|
||||
return ""
|
||||
return proc.stdout.decode("utf-8", errors="replace").strip()
|
||||
|
||||
@@ -136,6 +136,33 @@ class BinLayoutTest(unittest.TestCase):
|
||||
"index_mail must point at bin/brain/index.go",
|
||||
)
|
||||
|
||||
def test_mail_ocr_is_tesseract_not_docling(self) -> None:
|
||||
self._assert_shebang("bin/mail/ocr.go")
|
||||
ocr = (ROOT / "bin" / "mail" / "ocr.go").read_text()
|
||||
self.assertIn("internal/ocr", ocr)
|
||||
self.assertIn("mail_ocr", ocr)
|
||||
self.assertNotIn("github.com/otiai10/gosseract", ocr)
|
||||
py = (ROOT / "bin" / "mail" / "import").read_text()
|
||||
self.assertNotIn("from docling", py)
|
||||
self.assertNotIn("import docling", py)
|
||||
self.assertIn("convert_pdf", py)
|
||||
conv = (ROOT / "bin" / "tools" / "mailconv.py").read_text()
|
||||
self.assertIn("pdftotext", conv)
|
||||
self.assertIn("pdftoppm", conv)
|
||||
self.assertIn("tesseract", conv)
|
||||
self.assertIn("eng+deu", conv)
|
||||
self.assertNotIn("from docling", conv)
|
||||
self.assertNotIn("import docling", conv)
|
||||
self.assertNotIn("gocv", conv.lower())
|
||||
proj = (ROOT / "pyproject.toml").read_text()
|
||||
self.assertNotIn("docling", proj)
|
||||
ci = (ROOT / ".github" / "workflows" / "ci.yml").read_text()
|
||||
self.assertIn("tesseract-ocr", ci)
|
||||
self.assertIn("./internal/ocr", ci)
|
||||
compose = (ROOT / "compose.yaml").read_text()
|
||||
self.assertIn("ocr-paddle", compose)
|
||||
self.assertIn("OCR_ENGINE", compose)
|
||||
|
||||
def test_markdown_import_is_go_not_python_exec(self) -> None:
|
||||
self._assert_shebang("bin/markdown/import.go")
|
||||
text = (ROOT / "bin" / "markdown" / "import.go").read_text()
|
||||
|
||||
@@ -8,10 +8,13 @@ from pathlib import Path
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
|
||||
from mailconv import ( # noqa: E402
|
||||
TESS_LANG,
|
||||
clean_email_address,
|
||||
convert_pdf,
|
||||
html_to_markdown,
|
||||
is_convertible,
|
||||
normalize_markdown,
|
||||
ocr_image,
|
||||
split_zip_members,
|
||||
subject_to_filename,
|
||||
zip_extract_safe,
|
||||
@@ -100,6 +103,92 @@ class TestMailConv(unittest.TestCase):
|
||||
self.assertFalse(is_convertible(".exe"))
|
||||
self.assertFalse(is_convertible(".unknown"))
|
||||
|
||||
def test_convert_pdf_prefers_pdftotext(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b"Invoice BM25 layout"
|
||||
stderr = b""
|
||||
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
out = convert_pdf(Path(self._tmp("born.pdf")))
|
||||
self.assertIn("BM25", out)
|
||||
self.assertEqual(calls[0][:2], ["pdftotext", "-layout"])
|
||||
self.assertFalse(any(c[0] == "tesseract" for c in calls))
|
||||
self.assertFalse(any(c[0] == "pdftoppm" for c in calls))
|
||||
|
||||
def test_convert_pdf_empty_layer_uses_pdftoppm_tesseract(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b""
|
||||
stderr = b""
|
||||
|
||||
if cmd[0] == "pdftotext":
|
||||
P.stdout = b" \n"
|
||||
return P()
|
||||
if cmd[0] == "pdftoppm":
|
||||
prefix = Path(cmd[-1])
|
||||
(prefix.parent / "page-1.png").write_bytes(b"fake")
|
||||
return P()
|
||||
if cmd[0] == "tesseract":
|
||||
P.stdout = b"scanned HELLO"
|
||||
return P()
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
out = convert_pdf(Path(self._tmp("scan.pdf")))
|
||||
self.assertIn("HELLO", out)
|
||||
bins = [c[0] for c in calls]
|
||||
self.assertIn("pdftotext", bins)
|
||||
self.assertIn("pdftoppm", bins)
|
||||
self.assertIn("tesseract", bins)
|
||||
tess = next(c for c in calls if c[0] == "tesseract")
|
||||
self.assertIn(TESS_LANG, tess)
|
||||
self.assertNotIn("docling", " ".join(bins))
|
||||
|
||||
def test_ocr_image_paddle_engine(self):
|
||||
import mailconv as mc
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def fake_run(cmd, **kwargs):
|
||||
calls.append(list(cmd))
|
||||
|
||||
class P:
|
||||
returncode = 0
|
||||
stdout = b"paddle text"
|
||||
stderr = b""
|
||||
|
||||
return P()
|
||||
|
||||
self._patch_run(mc, fake_run)
|
||||
os.environ["OCR_ENGINE"] = "paddle"
|
||||
try:
|
||||
out = ocr_image(Path(self._tmp("x.png")))
|
||||
finally:
|
||||
os.environ.pop("OCR_ENGINE", None)
|
||||
self.assertEqual(out, "paddle text")
|
||||
self.assertEqual(calls[0][:2], ["paddleocr", "ocr"])
|
||||
|
||||
def _patch_run(self, mod, fn) -> None:
|
||||
self.addCleanup(setattr, mod.subprocess, "run", mod.subprocess.run)
|
||||
mod.subprocess.run = fn
|
||||
|
||||
def _mk_zip(self, members):
|
||||
zpath = Path(self._tmp("arc.zip"))
|
||||
with zipfile.ZipFile(zpath, "w") as zf:
|
||||
|
||||
Reference in New Issue
Block a user