feat: OCR scans with tesseract, drop docling from the default path.
pdftotext still wins on born-digital PDFs. Empty text layers go through pdftoppm + tesseract eng+deu. Optional OCR_ENGINE=paddle. Gitea #6.
This commit is contained in:
@@ -6,7 +6,6 @@ readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
license = { text = "MIT" }
|
||||
dependencies = [
|
||||
"docling>=2.119.0",
|
||||
"ladybug==0.19.1",
|
||||
"markitdown[docx,epub,html,image-exif,pdf,pptx,xlsx,zip]>=0.1.7",
|
||||
"mistune==3.3.4",
|
||||
|
||||
Reference in New Issue
Block a user