mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-05 02:13:24 +00:00
ocr update
This commit is contained in:
1 parent
d7a7d4d084
commit
d4e92be0ab
27 files changed
+1656
-134
No files matched your search
@@ -66,6 +66,16 @@ RUN apt-get update && \
|
||||
ln -s /usr/bin/python3.12 /usr/bin/python && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# The recommended OCR engine (OCR_ENGINE=tesseract) shells out to the system
|
||||
# tesseract binary, so ship it alongside the optional docling engine. Extra
|
||||
# language packs are a deployment concern (apt: tesseract-ocr-<lang>).
|
||||
ARG INSTALL_DOCLING=false
|
||||
RUN if [ "$INSTALL_DOCLING" = "true" ]; then \
|
||||
apt-get update && \
|
||||
apt-get install -y --no-install-recommends tesseract-ocr tesseract-ocr-eng && \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
fi
|
||||
|
||||
# Set working directory
|
||||
WORKDIR /app
|
||||
|
||||
|
||||
@@ -272,7 +272,21 @@ class UploadFile(Resource):
|
||||
# Office/e-book containers are ZIP-based formats but must
|
||||
# be parsed as documents, not expanded as user archives.
|
||||
is_office_format = safe_file.lower().endswith(
|
||||
(".docx", ".xlsx", ".pptx", ".odt", ".ods", ".odp", ".epub")
|
||||
(
|
||||
".docx",
|
||||
".docm",
|
||||
".xlsx",
|
||||
".xlsm",
|
||||
".xlsb",
|
||||
".pptx",
|
||||
".pptm",
|
||||
".ppsx",
|
||||
".ppsm",
|
||||
".odt",
|
||||
".ods",
|
||||
".odp",
|
||||
".epub",
|
||||
)
|
||||
)
|
||||
if zipfile.is_zipfile(temp_file_path) and not is_office_format:
|
||||
extract_dir = os.path.join(upload_dir, "extracted")
|
||||
|
||||
@@ -129,6 +129,40 @@ class Settings(BaseSettings):
|
||||
# and the upload cap is 100 MB, so the gate is what keeps one upload from
|
||||
# taking the ingest worker down. 0 disables it.
|
||||
MARKUP_MAX_BYTES: int = 8_000_000
|
||||
# Trust-check anydoc's PDF output (application/parser/file/pdf_trust.py):
|
||||
# flag composite (Type0) fonts without a ToUnicode map, and CJK-declaring
|
||||
# PDFs whose extracted text has almost no CJK — the two classes where
|
||||
# anydoc drops text silently. A flagged file re-parses on the docling
|
||||
# fallback when docling is installed; otherwise the anydoc output is kept
|
||||
# and the document gets extra_info["parse_warnings"]. ~30 ms per scanned MB.
|
||||
PDF_TRUST_CHECK: bool = True
|
||||
# Rewrite dot-leader / whitespace-aligned table runs in anydoc's PDF
|
||||
# markdown into GFM tables (application/parser/file/tableize.py). Off by
|
||||
# default: it rewrites content on a heuristic (>=3 uniform label+numbers
|
||||
# lines) validated only on a small corpus so far.
|
||||
ANYDOC_TABLEIZE: bool = False
|
||||
# OCR engine used by the docling parsers when OCR is on
|
||||
# (DOCLING_OCR_ENABLED / DOCLING_OCR_ATTACHMENTS_ENABLED). Benched 2026-08
|
||||
# on EN/ZH/table/degraded scans (docs/Guides/ocr has the menu):
|
||||
# tesseract — recommended: best classic-engine accuracy (perfect EN word
|
||||
# recall, 0.000 bilingual CER, 100% table cells), ~35 MB, CPU-only.
|
||||
# Needs the system binary + language packs (the Docker image installs
|
||||
# them when built with INSTALL_DOCLING=true).
|
||||
# auto — docling's pick: ocrmac on macOS (excellent), rapidocr on Linux
|
||||
# (silently shreds some long text lines — avoid as a server default).
|
||||
# ocrmac | rapidocr — force one of those.
|
||||
# deepseek — DeepSeek-OCR through docling's VLM pipeline against an
|
||||
# Ollama/vLLM endpoint (OCR_DEEPSEEK_*). Best table/CJK quality; the
|
||||
# worker stays light (~370 MB, no layout models) but each page costs
|
||||
# seconds on the model server.
|
||||
# An engine that is not installed degrades to "auto" with a warning
|
||||
# instead of failing the parse.
|
||||
OCR_ENGINE: str = "tesseract"
|
||||
# Tesseract language packs, "+"-separated (e.g. "eng+chi_sim+deu"). Other
|
||||
# engines keep their own defaults — their language codes differ.
|
||||
OCR_LANGS: str = "eng"
|
||||
OCR_DEEPSEEK_URL: str = "http://localhost:11434/v1/chat/completions"
|
||||
OCR_DEEPSEEK_MODEL: str = "deepseek-ocr:3b"
|
||||
# Chars-per-page floor below which an OCR'd PDF/image parse is treated as a docling
|
||||
# pipeline dropout (long-running workers were observed returning zero characters for
|
||||
# every scanned page after a long scanned PDF, with no error) rather than as content.
|
||||
|
||||
@@ -18,6 +18,7 @@ import logging
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Optional, Tuple, Union
|
||||
|
||||
from application.core.settings import settings
|
||||
from application.parser.file.base_parser import (
|
||||
BaseParser,
|
||||
DocumentParseError,
|
||||
@@ -27,6 +28,11 @@ from application.parser.file.base_parser import (
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# A fallback parse (scanned-PDF delegation or trust-check re-parse) yielding
|
||||
# fewer stripped characters than this is an empty document in the making,
|
||||
# not content.
|
||||
_MIN_SCAN_FALLBACK_CHARS = 50
|
||||
|
||||
# Suffixes anydoc 0.2.3 converts, verified with ``anydoc.format_from_extension``.
|
||||
# ``.epub`` is deliberately absent: it stays on ``EpubParser``.
|
||||
ANYDOC_SUFFIXES: Tuple[str, ...] = (
|
||||
@@ -56,11 +62,39 @@ ANYDOC_SUFFIXES: Tuple[str, ...] = (
|
||||
)
|
||||
|
||||
|
||||
# The five formats every engine has its own parser for (docling / legacy);
|
||||
# under the anydoc engine they get that parser as the fallback. Everything
|
||||
# else in ``ANYDOC_SUFFIXES`` is anydoc-only ("gained") and is mapped to a
|
||||
# fallback-less ``AnydocParser`` in every parser map.
|
||||
_CORE_SUFFIXES = frozenset({".pdf", ".docx", ".pptx", ".xlsx", ".csv"})
|
||||
ANYDOC_GAINED_SUFFIXES: Tuple[str, ...] = tuple(
|
||||
suffix for suffix in ANYDOC_SUFFIXES if suffix not in _CORE_SUFFIXES
|
||||
)
|
||||
|
||||
|
||||
def anydoc_available() -> bool:
|
||||
"""Whether the ``anydoc`` extension can be imported."""
|
||||
return module_available("anydoc")
|
||||
|
||||
|
||||
def _result_chars(result: Union[str, List[str]]) -> int:
|
||||
"""Stripped character count of a parser result (str or list of rows)."""
|
||||
if isinstance(result, list):
|
||||
return sum(len(str(part).strip()) for part in result)
|
||||
return len(str(result).strip())
|
||||
|
||||
|
||||
def _is_docling_backed(parser: Optional[BaseParser]) -> bool:
|
||||
"""True when ``parser`` is a docling parser (worth a trust-check re-parse)."""
|
||||
if parser is None:
|
||||
return False
|
||||
try:
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
except ImportError:
|
||||
return False
|
||||
return isinstance(parser, DoclingParser)
|
||||
|
||||
|
||||
class AnydocParser(BaseParser):
|
||||
"""Markdown conversion via ``anydoc.to_markdown`` with a typed fallback.
|
||||
|
||||
@@ -82,6 +116,10 @@ class AnydocParser(BaseParser):
|
||||
super().__init__(parser_config)
|
||||
self.fallback_parser = fallback_parser
|
||||
self.last_engine: Optional[str] = None
|
||||
# (path, problems) from the most recent parse whose output the trust
|
||||
# check flagged but kept — surfaced via ``get_file_metadata`` as
|
||||
# ``parse_warnings`` so the document carries its own caveat.
|
||||
self._last_warnings: Optional[Tuple[Path, List[str]]] = None
|
||||
|
||||
def _init_parser(self) -> Dict:
|
||||
# Import for real rather than trusting ``find_spec``: a wheel whose
|
||||
@@ -117,6 +155,7 @@ class AnydocParser(BaseParser):
|
||||
import anydoc
|
||||
|
||||
path = Path(file)
|
||||
self._last_warnings = None
|
||||
try:
|
||||
content = anydoc.to_markdown(str(path))
|
||||
except anydoc.ResourceLimitError as exc:
|
||||
@@ -132,7 +171,10 @@ class AnydocParser(BaseParser):
|
||||
# Typed, per-document failures another engine may still handle:
|
||||
# UnsupportedError (scanned PDF / unknown format), MalformedError,
|
||||
# EncryptedError, MissingPartError.
|
||||
return self._delegate(path, errors, f"{type(exc).__name__}: {exc}")
|
||||
ocr_needed = isinstance(exc, anydoc.UnsupportedError) and "OCR" in str(exc)
|
||||
return self._delegate(
|
||||
path, errors, f"{type(exc).__name__}: {exc}", ocr_needed=ocr_needed
|
||||
)
|
||||
except OSError as exc:
|
||||
raise DocumentParseError(
|
||||
f"Failed to parse {path.name}: the file could not be read."
|
||||
@@ -147,10 +189,117 @@ class AnydocParser(BaseParser):
|
||||
return self._delegate(path, errors, "anydoc produced no text")
|
||||
|
||||
self.last_engine = "anydoc"
|
||||
if path.suffix.lower() == ".pdf":
|
||||
content = self._finish_pdf(path, content, errors)
|
||||
return content
|
||||
|
||||
def _delegate(self, path: Path, errors: str, reason: str) -> Union[str, List[str]]:
|
||||
"""Hand ``path`` to the fallback parser, or fail loudly without one."""
|
||||
def _finish_pdf(self, path: Path, content: str, errors: str) -> Union[str, List[str]]:
|
||||
"""Post-process a successful anydoc PDF conversion.
|
||||
|
||||
Two PDF-only steps:
|
||||
|
||||
1. Trust check (``PDF_TRUST_CHECK``): the two classes where anydoc
|
||||
drops text *silently* — Type0 fonts without ToUnicode, and
|
||||
CJK-declaring PDFs with CJK-less output. A flagged file re-parses
|
||||
on the docling fallback when one is wired; otherwise the anydoc
|
||||
output is kept and the problems ride along as ``parse_warnings``
|
||||
metadata.
|
||||
2. ``ANYDOC_TABLEIZE`` (off by default): rewrite dot-leader /
|
||||
whitespace table runs as GFM tables. Never applied to a docling
|
||||
re-parse — docling emits real tables already.
|
||||
"""
|
||||
problems = self._trust_problems(path, content)
|
||||
if problems:
|
||||
rerouted = self._reroute_flagged(path, errors, problems)
|
||||
if rerouted is not None:
|
||||
return rerouted
|
||||
self._last_warnings = (path, problems)
|
||||
logger.warning(
|
||||
f"anydoc output for {path.name} failed the PDF trust check "
|
||||
f"({'; '.join(problems)}); keeping it with parse_warnings"
|
||||
)
|
||||
if settings.ANYDOC_TABLEIZE:
|
||||
from application.parser.file.tableize import tableize
|
||||
|
||||
content = tableize(content)
|
||||
return content
|
||||
|
||||
def _trust_problems(self, path: Path, content: str) -> List[str]:
|
||||
"""Trust-check findings for ``content``; [] when disabled, clean, or the check errors."""
|
||||
if not settings.PDF_TRUST_CHECK:
|
||||
return []
|
||||
try:
|
||||
from application.parser.file.pdf_trust import verify_pdf_file
|
||||
|
||||
return verify_pdf_file(path, content)
|
||||
except Exception:
|
||||
# The check exists to catch silent loss; it must never turn a
|
||||
# successful parse into a failure.
|
||||
logger.warning(
|
||||
f"PDF trust check errored on {path.name}; trusting the output",
|
||||
exc_info=True,
|
||||
)
|
||||
return []
|
||||
|
||||
def _reroute_flagged(
|
||||
self, path: Path, errors: str, problems: List[str]
|
||||
) -> Optional[Union[str, List[str]]]:
|
||||
"""Re-parse a trust-flagged PDF with a docling-backed fallback.
|
||||
|
||||
Returns the fallback's output, or None when there is no docling
|
||||
fallback, it fails, or it comes back near-empty — the caller then
|
||||
keeps the anydoc output.
|
||||
Only docling is worth the re-parse: it ships Adobe's predefined
|
||||
CMaps, which is exactly what the flagged font class needs; the
|
||||
legacy pypdf parser does not.
|
||||
"""
|
||||
fallback = self.fallback_parser
|
||||
if not _is_docling_backed(fallback):
|
||||
return None
|
||||
logger.warning(
|
||||
f"anydoc output for {path.name} failed the PDF trust check "
|
||||
f"({'; '.join(problems)}); re-parsing with {type(fallback).__name__}"
|
||||
)
|
||||
try:
|
||||
result = delegate_parse(fallback, path, errors)
|
||||
except DocumentParseError:
|
||||
logger.warning(
|
||||
f"docling re-parse of trust-flagged {path.name} failed; "
|
||||
"keeping the anydoc output",
|
||||
exc_info=True,
|
||||
)
|
||||
return None
|
||||
if _result_chars(result) < _MIN_SCAN_FALLBACK_CHARS:
|
||||
# A docling pipeline dropout ('' / '<!-- image -->') returns
|
||||
# without raising; adopting it would swap anydoc's real text for
|
||||
# an empty document.
|
||||
logger.warning(
|
||||
f"docling re-parse of trust-flagged {path.name} returned almost "
|
||||
"no text; keeping the anydoc output"
|
||||
)
|
||||
return None
|
||||
self.last_engine = getattr(fallback, "last_engine", None) or type(fallback).__name__
|
||||
return result
|
||||
|
||||
def get_file_metadata(self, file: Path) -> Dict:
|
||||
"""Surface trust-check findings for the file just parsed, if any."""
|
||||
last = self._last_warnings
|
||||
if last is not None and last[0] == Path(file):
|
||||
return {"parse_warnings": list(last[1])}
|
||||
return {}
|
||||
|
||||
def _delegate(
|
||||
self, path: Path, errors: str, reason: str, ocr_needed: bool = False
|
||||
) -> Union[str, List[str]]:
|
||||
"""Hand ``path`` to the fallback parser, or fail loudly without one.
|
||||
|
||||
``ocr_needed`` marks anydoc's scanned-PDF refusal ("OCR is required").
|
||||
The fallback still runs first — docling extracts text layers anydoc
|
||||
refuses (seen on degenerate CJK text layers) even with OCR off — but
|
||||
when it comes back near-empty the parse fails loudly, telling the
|
||||
user how to get OCR, instead of silently storing an empty document
|
||||
for a scan.
|
||||
"""
|
||||
fallback = self.fallback_parser
|
||||
if fallback is None:
|
||||
self.last_engine = None
|
||||
@@ -160,5 +309,19 @@ class AnydocParser(BaseParser):
|
||||
f"falling back to {type(fallback).__name__}"
|
||||
)
|
||||
result = delegate_parse(fallback, path, errors)
|
||||
if ocr_needed and _result_chars(result) < _MIN_SCAN_FALLBACK_CHARS:
|
||||
self.last_engine = None
|
||||
if getattr(fallback, "ocr_enabled", False):
|
||||
hint = " even with OCR enabled."
|
||||
else:
|
||||
hint = (
|
||||
". Enable OCR to ingest scans: set DOCLING_OCR_ENABLED=true "
|
||||
"(and DOCLING_OCR_ATTACHMENTS_ENABLED for attachments) with "
|
||||
"the docling extra installed."
|
||||
)
|
||||
raise DocumentParseError(
|
||||
f"{path.name} appears to be a scanned PDF (no text layer), and "
|
||||
f"{type(fallback).__name__} extracted almost nothing{hint}"
|
||||
)
|
||||
self.last_engine = getattr(fallback, "last_engine", None) or type(fallback).__name__
|
||||
return result
|
||||
@@ -49,6 +49,25 @@ def _wrap_pdf_fast_path(pdf_parser: BaseParser) -> BaseParser:
|
||||
)
|
||||
|
||||
|
||||
def _gained_format_entries() -> Dict[str, BaseParser]:
|
||||
"""Anydoc-only formats (legacy/macro Office, OpenDocument, RTF) for every map.
|
||||
|
||||
anydoc is a core dependency, so these suffixes are parseable under both
|
||||
engines. When anydoc is somehow missing the entries are omitted, and such
|
||||
files fall to ``SimpleDirectoryReader``'s plain-text read — the
|
||||
pre-existing behaviour for unmapped suffixes.
|
||||
"""
|
||||
from application.parser.file.anydoc_parser import (
|
||||
ANYDOC_GAINED_SUFFIXES,
|
||||
AnydocParser,
|
||||
anydoc_available,
|
||||
)
|
||||
|
||||
if not anydoc_available():
|
||||
return {}
|
||||
return {suffix: AnydocParser() for suffix in ANYDOC_GAINED_SUFFIXES}
|
||||
|
||||
|
||||
def _legacy_file_extractor(pdf_text_fast_path: bool = False) -> Dict[str, BaseParser]:
|
||||
"""Parser map that needs neither docling nor anydoc.
|
||||
|
||||
@@ -79,6 +98,7 @@ def _legacy_file_extractor(pdf_text_fast_path: bool = False) -> Dict[str, BasePa
|
||||
".jpg": ImageParser(),
|
||||
".jpeg": ImageParser(),
|
||||
**_build_audio_parser_mapping(),
|
||||
**_gained_format_entries(),
|
||||
}
|
||||
|
||||
|
||||
@@ -165,6 +185,8 @@ def _docling_file_extractor(
|
||||
".xml": DoclingXMLParser(),
|
||||
# Formats docling doesn't support - use standard parsers
|
||||
".epub": EpubParser(),
|
||||
# Formats only anydoc reads (legacy/macro Office, OpenDocument, RTF)
|
||||
**_gained_format_entries(),
|
||||
}
|
||||
|
||||
|
||||
@@ -211,7 +233,13 @@ def _anydoc_file_extractor(ocr_enabled: bool, pdf_text_fast_path: bool = False)
|
||||
)
|
||||
extractor = dict(base)
|
||||
for suffix in ANYDOC_SUFFIXES:
|
||||
extractor[suffix] = AnydocParser(fallback_parser=base.get(suffix))
|
||||
fallback = base.get(suffix)
|
||||
if isinstance(fallback, AnydocParser):
|
||||
# A gained-format entry from the base map is already a
|
||||
# fallback-less AnydocParser — anydoc delegating to anydoc would
|
||||
# just repeat the same failure.
|
||||
continue
|
||||
extractor[suffix] = AnydocParser(fallback_parser=fallback)
|
||||
extractor[".html"] = HTMLMarkdownParser()
|
||||
extractor[".xhtml"] = HTMLMarkdownParser()
|
||||
return extractor
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
"""Shared file-extension constants for parsing and ingestion flows."""
|
||||
|
||||
from application.parser.file.anydoc_parser import ANYDOC_GAINED_SUFFIXES
|
||||
from application.stt.constants import SUPPORTED_AUDIO_EXTENSIONS
|
||||
|
||||
|
||||
@@ -16,6 +17,12 @@ SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS = (
|
||||
".json",
|
||||
".xlsx",
|
||||
".pptx",
|
||||
# Read by the HTML parsers on every engine.
|
||||
".xhtml",
|
||||
# Read by the anydoc engine (legacy/macro Office, OpenDocument, RTF).
|
||||
# Parseable regardless of DOC_PARSER_ENGINE: anydoc is a core dependency,
|
||||
# and both parser maps route these suffixes to it.
|
||||
*ANYDOC_GAINED_SUFFIXES,
|
||||
)
|
||||
|
||||
SUPPORTED_SOURCE_IMAGE_EXTENSIONS = (".png", ".jpg", ".jpeg")
|
||||
|
||||
@@ -10,6 +10,8 @@ import importlib.util
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
@@ -71,6 +73,112 @@ def _apply_inference_settings() -> None:
|
||||
inference.compile_torch_models = settings.DOCLING_COMPILE_TORCH_MODELS
|
||||
|
||||
|
||||
_VALID_OCR_ENGINES = ("tesseract", "auto", "ocrmac", "rapidocr", "deepseek")
|
||||
|
||||
|
||||
def _resolve_ocr_engine(requested: Optional[str]) -> str:
|
||||
"""Resolve the OCR engine to build, degrading to ``auto`` when unavailable.
|
||||
|
||||
``auto`` is docling's own selection (ocrmac on macOS, rapidocr on a
|
||||
typical Linux server). The degradation is deliberate: OCR is switched on
|
||||
by deployments that expect scans to work, so a missing engine must warn
|
||||
and OCR with what exists rather than fail every parse.
|
||||
|
||||
Args:
|
||||
requested: Engine name, or None to read ``settings.OCR_ENGINE``.
|
||||
|
||||
Returns:
|
||||
One of ``_VALID_OCR_ENGINES``, guaranteed buildable here.
|
||||
"""
|
||||
from application.core.settings import settings
|
||||
from application.parser.file.base_parser import module_available
|
||||
|
||||
engine = str(requested or settings.OCR_ENGINE or "auto").strip().lower()
|
||||
if engine not in _VALID_OCR_ENGINES:
|
||||
logger.warning(
|
||||
f"Unknown OCR_ENGINE {engine!r}; using docling auto-selection"
|
||||
)
|
||||
return "auto"
|
||||
if engine == "tesseract" and shutil.which("tesseract") is None:
|
||||
logger.warning(
|
||||
"OCR_ENGINE=tesseract but no tesseract binary is on PATH (install "
|
||||
"tesseract-ocr plus language packs, or build the Docker image with "
|
||||
"INSTALL_DOCLING=true); using docling auto-selection"
|
||||
)
|
||||
return "auto"
|
||||
if engine == "ocrmac" and (
|
||||
sys.platform != "darwin" or not module_available("ocrmac")
|
||||
):
|
||||
logger.warning(
|
||||
"OCR_ENGINE=ocrmac needs macOS with the ocrmac package; "
|
||||
"using docling auto-selection"
|
||||
)
|
||||
return "auto"
|
||||
if engine == "rapidocr" and not module_available("rapidocr"):
|
||||
logger.warning(
|
||||
"OCR_ENGINE=rapidocr but rapidocr is not installed; "
|
||||
"using docling auto-selection"
|
||||
)
|
||||
return "auto"
|
||||
return engine
|
||||
|
||||
|
||||
def _build_ocr_options(
|
||||
engine: str, languages: Optional[List[str]], force_full_page_ocr: bool
|
||||
):
|
||||
"""docling OCR options for a resolved classic engine.
|
||||
|
||||
Returns None for ``auto`` (docling's default pipeline options already run
|
||||
auto-selection) and on any build failure — the parse then proceeds on the
|
||||
default engine rather than failing; the caller re-applies
|
||||
``force_full_page_ocr`` onto whatever options end up active.
|
||||
|
||||
Args:
|
||||
engine: A ``_resolve_ocr_engine`` result other than ``deepseek``.
|
||||
languages: Engine-specific language list; None uses the engine's
|
||||
default (tesseract reads ``settings.OCR_LANGS``).
|
||||
force_full_page_ocr: OCR whole pages instead of only bitmap regions.
|
||||
"""
|
||||
if engine == "auto":
|
||||
return None
|
||||
from application.core.settings import settings
|
||||
|
||||
try:
|
||||
if engine == "tesseract":
|
||||
from docling.datamodel.pipeline_options import TesseractCliOcrOptions
|
||||
|
||||
langs = (
|
||||
languages
|
||||
or [lang.strip() for lang in settings.OCR_LANGS.split("+") if lang.strip()]
|
||||
or ["eng"]
|
||||
)
|
||||
return TesseractCliOcrOptions(
|
||||
lang=langs, force_full_page_ocr=force_full_page_ocr
|
||||
)
|
||||
if engine == "rapidocr":
|
||||
from docling.datamodel.pipeline_options import RapidOcrOptions
|
||||
|
||||
return RapidOcrOptions(
|
||||
lang=languages or ["english"],
|
||||
force_full_page_ocr=force_full_page_ocr,
|
||||
)
|
||||
if engine == "ocrmac":
|
||||
from docling.datamodel.pipeline_options import OcrMacOptions
|
||||
|
||||
if languages:
|
||||
return OcrMacOptions(
|
||||
lang=languages, force_full_page_ocr=force_full_page_ocr
|
||||
)
|
||||
return OcrMacOptions(force_full_page_ocr=force_full_page_ocr)
|
||||
except ImportError as e:
|
||||
logger.warning(f"Failed to build {engine} OCR options: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Error building {engine} OCR options: {e}")
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def _tabular_content_size(file: Path) -> int:
|
||||
"""Effective content size of a tabular file, in bytes.
|
||||
|
||||
@@ -303,7 +411,7 @@ class DoclingParser(BaseParser):
|
||||
- Advanced PDF layout analysis
|
||||
- Table structure recognition
|
||||
- Reading order detection
|
||||
- OCR for scanned documents (supports RapidOCR)
|
||||
- OCR for scanned documents (engine chosen by ``OCR_ENGINE``)
|
||||
- Unified DoclingDocument format
|
||||
- Export to Markdown
|
||||
|
||||
@@ -317,7 +425,7 @@ class DoclingParser(BaseParser):
|
||||
ocr_enabled: bool = True,
|
||||
table_structure: bool = True,
|
||||
export_format: str = "markdown",
|
||||
use_rapidocr: bool = True,
|
||||
ocr_engine: Optional[str] = None,
|
||||
ocr_languages: Optional[List[str]] = None,
|
||||
force_full_page_ocr: bool = False,
|
||||
):
|
||||
@@ -327,29 +435,33 @@ class DoclingParser(BaseParser):
|
||||
ocr_enabled: Enable OCR for bitmap/image regions in documents
|
||||
table_structure: Enable table structure recognition
|
||||
export_format: Output format ('markdown', 'text', 'html')
|
||||
use_rapidocr: Use RapidOCR engine (default True, works well in Docker)
|
||||
ocr_languages: List of OCR languages (default: ['english'])
|
||||
ocr_engine: OCR engine when OCR is enabled — one of
|
||||
``tesseract | auto | ocrmac | rapidocr | deepseek``. None
|
||||
reads ``settings.OCR_ENGINE`` at converter build time; an
|
||||
unavailable engine degrades to docling's auto-selection with
|
||||
a warning.
|
||||
ocr_languages: Engine-specific language list; None keeps the
|
||||
engine's own default (tesseract reads ``settings.OCR_LANGS``).
|
||||
force_full_page_ocr: Force OCR on entire page (False = smart hybrid OCR)
|
||||
"""
|
||||
super().__init__()
|
||||
self.ocr_enabled = ocr_enabled
|
||||
self.table_structure = table_structure
|
||||
self.export_format = export_format
|
||||
self.use_rapidocr = use_rapidocr
|
||||
self.ocr_languages = ocr_languages or ["english"]
|
||||
self.ocr_engine = ocr_engine
|
||||
self.ocr_languages = ocr_languages
|
||||
self.force_full_page_ocr = force_full_page_ocr
|
||||
self._converter = None
|
||||
|
||||
def _create_converter(self):
|
||||
"""Create a docling converter with hybrid OCR configuration.
|
||||
"""Create a docling converter for the configured OCR engine.
|
||||
|
||||
Uses smart OCR approach:
|
||||
- When ocr_enabled=True and force_full_page_ocr=False (default):
|
||||
Layout model detects text vs bitmap regions, OCR only runs on bitmaps
|
||||
- When ocr_enabled=True and force_full_page_ocr=True:
|
||||
OCR runs on entire page (for scanned documents/images)
|
||||
- When ocr_enabled=False:
|
||||
No OCR, only native text extraction
|
||||
- ``ocr_enabled=False``: no OCR, native text extraction only.
|
||||
- Classic engines (tesseract / auto / ocrmac / rapidocr): the standard
|
||||
PDF pipeline (layout + TableFormer) with that engine's OCR options.
|
||||
``force_full_page_ocr=False`` (default) OCRs only the bitmap regions
|
||||
the layout model finds; True routes whole pages through OCR.
|
||||
- ``deepseek``: the VLM pipeline instead (``_create_vlm_converter``).
|
||||
|
||||
Returns:
|
||||
DocumentConverter instance
|
||||
@@ -364,6 +476,10 @@ class DoclingParser(BaseParser):
|
||||
|
||||
_apply_inference_settings()
|
||||
|
||||
engine = _resolve_ocr_engine(self.ocr_engine) if self.ocr_enabled else None
|
||||
if engine == "deepseek":
|
||||
return self._create_vlm_converter()
|
||||
|
||||
pipeline_options = PdfPipelineOptions(
|
||||
do_ocr=self.ocr_enabled,
|
||||
do_table_structure=self.table_structure,
|
||||
@@ -371,13 +487,15 @@ class DoclingParser(BaseParser):
|
||||
_apply_pipeline_caps(pipeline_options)
|
||||
|
||||
if self.ocr_enabled:
|
||||
ocr_options = self._get_ocr_options()
|
||||
ocr_options = _build_ocr_options(
|
||||
engine, self.ocr_languages, self.force_full_page_ocr
|
||||
)
|
||||
if ocr_options is not None:
|
||||
pipeline_options.ocr_options = ocr_options
|
||||
# Docling's *default* OCR options carry their own flag, so without
|
||||
# this the setting was silently dropped whenever `_get_ocr_options`
|
||||
# returned None (use_rapidocr=False) — including the dropout retry,
|
||||
# whose whole point is forcing full-page OCR.
|
||||
# this the setting was silently dropped whenever no explicit
|
||||
# options were built (engine=auto, or a build failure) — including
|
||||
# the dropout retry, whose whole point is forcing full-page OCR.
|
||||
active_ocr_options = getattr(pipeline_options, "ocr_options", None)
|
||||
if hasattr(active_ocr_options, "force_full_page_ocr"):
|
||||
active_ocr_options.force_full_page_ocr = self.force_full_page_ocr
|
||||
@@ -393,12 +511,55 @@ class DoclingParser(BaseParser):
|
||||
}
|
||||
)
|
||||
|
||||
def _create_vlm_converter(self):
|
||||
"""DeepSeek-OCR converter via docling's VLM pipeline.
|
||||
|
||||
Each page goes to an OpenAI-compatible endpoint (Ollama or vLLM;
|
||||
``OCR_DEEPSEEK_URL`` / ``OCR_DEEPSEEK_MODEL``) and the grounded output
|
||||
is parsed back into a DoclingDocument. This replaces the *entire*
|
||||
classic pipeline — no layout/TableFormer/OCR models load in the worker
|
||||
(~370 MB RSS vs 1.0-1.6 GB measured), the compute lives in the model
|
||||
server. Bench trade-offs (2026-08): best table/CJK/degraded-scan
|
||||
quality of every engine tried, ~10-20 s/page on modest hardware, and
|
||||
occasional silent drops of page-level elements (titles).
|
||||
"""
|
||||
from docling.datamodel import vlm_model_specs
|
||||
from docling.datamodel.pipeline_options import VlmPipelineOptions
|
||||
from docling.document_converter import (
|
||||
DocumentConverter,
|
||||
ImageFormatOption,
|
||||
InputFormat,
|
||||
PdfFormatOption,
|
||||
)
|
||||
from docling.pipeline.vlm_pipeline import VlmPipeline
|
||||
|
||||
from application.core.settings import settings
|
||||
|
||||
vlm_options = vlm_model_specs.DEEPSEEKOCR_OLLAMA.model_copy(deep=True)
|
||||
vlm_options.url = settings.OCR_DEEPSEEK_URL
|
||||
vlm_options.params["model"] = settings.OCR_DEEPSEEK_MODEL
|
||||
pipeline_options = VlmPipelineOptions(
|
||||
vlm_options=vlm_options, enable_remote_services=True
|
||||
)
|
||||
return DocumentConverter(
|
||||
format_options={
|
||||
InputFormat.PDF: PdfFormatOption(
|
||||
pipeline_cls=VlmPipeline, pipeline_options=pipeline_options
|
||||
),
|
||||
InputFormat.IMAGE: ImageFormatOption(
|
||||
pipeline_cls=VlmPipeline, pipeline_options=pipeline_options
|
||||
),
|
||||
}
|
||||
)
|
||||
|
||||
def _init_parser(self) -> Dict:
|
||||
"""Initialize the docling converter with hybrid OCR."""
|
||||
from application.core.settings import settings
|
||||
|
||||
logger.info("Initializing DoclingParser...")
|
||||
logger.info(f" ocr_enabled={self.ocr_enabled}")
|
||||
logger.info(f" force_full_page_ocr={self.force_full_page_ocr}")
|
||||
logger.info(f" use_rapidocr={self.use_rapidocr}")
|
||||
logger.info(f" ocr_engine={self.ocr_engine or settings.OCR_ENGINE}")
|
||||
|
||||
if importlib.util.find_spec("docling.document_converter") is None:
|
||||
raise ImportError(
|
||||
@@ -414,34 +575,11 @@ class DoclingParser(BaseParser):
|
||||
"ocr_enabled": self.ocr_enabled,
|
||||
"table_structure": self.table_structure,
|
||||
"export_format": self.export_format,
|
||||
"use_rapidocr": self.use_rapidocr,
|
||||
"ocr_engine": self.ocr_engine,
|
||||
"ocr_languages": self.ocr_languages,
|
||||
"force_full_page_ocr": self.force_full_page_ocr,
|
||||
}
|
||||
|
||||
def _get_ocr_options(self):
|
||||
"""Get OCR options based on configuration.
|
||||
|
||||
Returns RapidOcrOptions if use_rapidocr is True and available,
|
||||
otherwise returns None to use docling defaults.
|
||||
"""
|
||||
if not self.use_rapidocr:
|
||||
return None
|
||||
|
||||
try:
|
||||
from docling.datamodel.pipeline_options import RapidOcrOptions
|
||||
|
||||
return RapidOcrOptions(
|
||||
lang=self.ocr_languages,
|
||||
force_full_page_ocr=self.force_full_page_ocr,
|
||||
)
|
||||
except ImportError as e:
|
||||
logger.warning(f"Failed to import RapidOcrOptions: {e}")
|
||||
return None
|
||||
except Exception as e:
|
||||
logger.error(f"Error creating RapidOcrOptions: {e}")
|
||||
return None
|
||||
|
||||
def _export_content(self, document) -> str:
|
||||
"""Export document content in the configured format.
|
||||
|
||||
@@ -651,7 +789,7 @@ class DoclingParser(BaseParser):
|
||||
|
||||
|
||||
class DoclingPDFParser(DoclingParser):
|
||||
"""Docling-based PDF parser with advanced features and RapidOCR support.
|
||||
"""Docling-based PDF parser with advanced features and configurable OCR.
|
||||
|
||||
Uses hybrid OCR approach by default:
|
||||
- Text regions: Direct PDF text extraction (fast)
|
||||
@@ -664,7 +802,7 @@ class DoclingPDFParser(DoclingParser):
|
||||
self,
|
||||
ocr_enabled: bool = True,
|
||||
table_structure: bool = True,
|
||||
use_rapidocr: bool = True,
|
||||
ocr_engine: Optional[str] = None,
|
||||
ocr_languages: Optional[List[str]] = None,
|
||||
force_full_page_ocr: bool = False,
|
||||
):
|
||||
@@ -672,7 +810,7 @@ class DoclingPDFParser(DoclingParser):
|
||||
ocr_enabled=ocr_enabled,
|
||||
table_structure=table_structure,
|
||||
export_format="markdown",
|
||||
use_rapidocr=use_rapidocr,
|
||||
ocr_engine=ocr_engine,
|
||||
ocr_languages=ocr_languages,
|
||||
force_full_page_ocr=force_full_page_ocr,
|
||||
)
|
||||
@@ -723,7 +861,7 @@ class DoclingHTMLParser(DoclingParser):
|
||||
|
||||
|
||||
class DoclingImageParser(DoclingParser):
|
||||
"""Docling-based image parser with OCR and RapidOCR support.
|
||||
"""Docling-based image parser with configurable OCR.
|
||||
|
||||
For images, force_full_page_ocr=True is used since images are entirely
|
||||
visual and require full OCR to extract any text.
|
||||
@@ -732,14 +870,14 @@ class DoclingImageParser(DoclingParser):
|
||||
def __init__(
|
||||
self,
|
||||
ocr_enabled: bool = True,
|
||||
use_rapidocr: bool = True,
|
||||
ocr_engine: Optional[str] = None,
|
||||
ocr_languages: Optional[List[str]] = None,
|
||||
force_full_page_ocr: bool = True,
|
||||
):
|
||||
super().__init__(
|
||||
ocr_enabled=ocr_enabled,
|
||||
export_format="markdown",
|
||||
use_rapidocr=use_rapidocr,
|
||||
ocr_engine=ocr_engine,
|
||||
ocr_languages=ocr_languages,
|
||||
force_full_page_ocr=force_full_page_ocr,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,177 @@
|
||||
"""Trust checks for PDF text extraction.
|
||||
|
||||
anydoc silently drops text from fonts it cannot map to Unicode — seen with
|
||||
non-embedded composite (Type0/CID) fonts relying on predefined CMaps, where
|
||||
the EN/ZH NDA corpus file loses its whole Chinese column with no error, and
|
||||
with Adobe-CNS1 fonts in HK legal PDFs where a valid ToUnicode exists but
|
||||
extraction still fails. These dependency-free checks scan the PDF's raw
|
||||
objects (including Flate-compressed object streams) so the pipeline can
|
||||
route such files to a heavier parser, or at least mark the output as
|
||||
unverified, instead of trusting silent partial text.
|
||||
|
||||
Two stages:
|
||||
|
||||
* ``check_pdf_fonts`` — pre-flight on the bytes alone: composite fonts
|
||||
without an embedded ToUnicode CMap, and whether the PDF declares CJK font
|
||||
resources at all.
|
||||
* ``verify_extraction`` — pre-flight plus the cross-check the font scan
|
||||
cannot do: a PDF that declares CJK fonts whose extracted text contains
|
||||
almost no CJK characters is not to be trusted.
|
||||
|
||||
A flag means "verify or route to a fallback", not "the text is wrong": tools
|
||||
shipping Adobe's predefined CMaps (docling's parser does) often extract such
|
||||
files fine. On the benchmark corpus the checks caught both known
|
||||
silent-drop cases with zero false positives on the other 14 PDFs, and cost
|
||||
~30 ms per scanned MB (92 ms on a 3 MB, 150-page annual report).
|
||||
"""
|
||||
import re
|
||||
import zlib
|
||||
from pathlib import Path
|
||||
from typing import Dict, Iterator, List, Union
|
||||
|
||||
_OBJ = re.compile(rb"\d+\s+\d+\s+obj(.*?)endobj", re.DOTALL)
|
||||
_STREAM = re.compile(rb"stream\r?\n(.*?)\r?\nendstream", re.DOTALL)
|
||||
_TYPE0 = re.compile(rb"/Subtype\s*/Type0")
|
||||
# The CJK expectation is keyed on CIDSystemInfo /Ordering ALONE — the
|
||||
# authoritative "this PDF maps text through a CJK character collection"
|
||||
# signal, present in every known silent-drop case. Matching CJK font *names*
|
||||
# (SimSun, MS-Gothic, MSungHK, ...) was tried and rejected: Word-exported
|
||||
# Latin PDFs routinely embed an MS-Gothic subset for a stray full-width
|
||||
# character (a 150-page English annual report in the benchmark corpus does),
|
||||
# and substring matching also catches Latin faces like FranklinGothic —
|
||||
# either way English-only documents would be rerouted to the heavy engine.
|
||||
_CJK_ORDERING = re.compile(rb"/Ordering\s*\((GB1|CNS1|Japan1|Japan2|KR|Korea1)\)")
|
||||
# Bytes kept around a Type0 marker found inside a decompressed object
|
||||
# stream: object streams hold many dicts with no obj/endobj markers, and a
|
||||
# font dict's keys sit close to its /Subtype entry.
|
||||
_WINDOW_BEFORE, _WINDOW_AFTER = 200, 800
|
||||
|
||||
# Extracted text with fewer CJK characters than this, from a PDF that
|
||||
# declares CJK fonts, is treated as a silent drop.
|
||||
_MIN_CJK_CHARS = 10
|
||||
|
||||
_CJK_RANGES = (
|
||||
("一", "鿿"), # CJK Unified Ideographs
|
||||
("", "ヿ"), # Hiragana + Katakana
|
||||
("가", ""), # Hangul syllables
|
||||
)
|
||||
|
||||
|
||||
# Per-stream decompression cap. Flate reaches ~1000:1, so uncapped inflation
|
||||
# of a crafted (or merely image-heavy) PDF inside the upload cap could balloon
|
||||
# to gigabytes and OOM the ingest worker. Font dicts and object streams — the
|
||||
# only things the signals live in — are far smaller than this.
|
||||
_STREAM_INFLATE_CAP = 4_000_000
|
||||
|
||||
|
||||
def _decompressed_streams(raw: bytes) -> Iterator[bytes]:
|
||||
"""Each Flate stream in ``raw`` that decompresses, one at a time, capped."""
|
||||
for match in _STREAM.finditer(raw):
|
||||
try:
|
||||
inflated = zlib.decompressobj().decompress(
|
||||
match.group(1), _STREAM_INFLATE_CAP
|
||||
)
|
||||
except zlib.error:
|
||||
continue
|
||||
yield inflated
|
||||
|
||||
|
||||
def _cjk_chars(text: str, up_to: int) -> int:
|
||||
"""Count CJK characters in ``text``, stopping once ``up_to`` is reached."""
|
||||
count = 0
|
||||
for char in text:
|
||||
for low, high in _CJK_RANGES:
|
||||
if low <= char <= high:
|
||||
count += 1
|
||||
break
|
||||
if count >= up_to:
|
||||
break
|
||||
return count
|
||||
|
||||
|
||||
def check_pdf_fonts(data: bytes) -> Dict[str, Union[int, bool]]:
|
||||
"""Pre-flight font scan of a PDF's bytes.
|
||||
|
||||
Args:
|
||||
data: The complete PDF file contents.
|
||||
|
||||
Returns:
|
||||
Dict with ``type0`` (composite fonts seen), ``type0_no_tounicode``
|
||||
(those without an embedded ToUnicode CMap), ``expects_cjk`` (the PDF
|
||||
declares CJK font resources), ``has_fonts``, and ``flagged`` — True
|
||||
when extraction should not be trusted unverified.
|
||||
"""
|
||||
type0 = bare = 0
|
||||
expects_cjk = bool(_CJK_ORDERING.search(data))
|
||||
has_fonts = b"/Font" in data
|
||||
|
||||
def count_font(unit: bytes) -> None:
|
||||
"""One analysis unit holding a Type0 font dict."""
|
||||
nonlocal type0, bare
|
||||
type0 += 1
|
||||
if b"/ToUnicode" not in unit:
|
||||
bare += 1
|
||||
|
||||
for match in _OBJ.finditer(data):
|
||||
body = match.group(1)
|
||||
if _TYPE0.search(body):
|
||||
count_font(body)
|
||||
# Each decompressed stream is scanned for every signal in one pass and
|
||||
# then dropped; nothing inflated is retained past its loop iteration.
|
||||
for stream in _decompressed_streams(data):
|
||||
expects_cjk = expects_cjk or bool(_CJK_ORDERING.search(stream))
|
||||
has_fonts = has_fonts or b"/Font" in stream
|
||||
for type0_match in _TYPE0.finditer(stream):
|
||||
start = type0_match.start()
|
||||
count_font(stream[max(0, start - _WINDOW_BEFORE): start + _WINDOW_AFTER])
|
||||
return {
|
||||
"type0": type0,
|
||||
"type0_no_tounicode": bare,
|
||||
"expects_cjk": expects_cjk,
|
||||
"has_fonts": has_fonts,
|
||||
"flagged": bare > 0 or not has_fonts,
|
||||
}
|
||||
|
||||
|
||||
def verify_extraction(data: bytes, markdown: str) -> List[str]:
|
||||
"""Reasons the extracted ``markdown`` of the PDF ``data`` shouldn't be trusted.
|
||||
|
||||
Combines the pre-flight font scan with the post-conversion cross-check it
|
||||
cannot do alone: fonts with a valid ToUnicode that the converter
|
||||
nevertheless failed to extract still show up as missing CJK output.
|
||||
|
||||
Args:
|
||||
data: The complete PDF file contents.
|
||||
markdown: The text a converter extracted from it.
|
||||
|
||||
Returns:
|
||||
Human-readable problem strings; empty when the output looks sound.
|
||||
"""
|
||||
problems: List[str] = []
|
||||
result = check_pdf_fonts(data)
|
||||
if result["type0_no_tounicode"]:
|
||||
problems.append(
|
||||
f"{result['type0_no_tounicode']} composite (Type0) font(s) carry no "
|
||||
"ToUnicode map; extracted text may silently omit glyphs"
|
||||
)
|
||||
elif not result["has_fonts"]:
|
||||
problems.append(
|
||||
"no font resources detected; text extraction cannot be verified"
|
||||
)
|
||||
if result["expects_cjk"]:
|
||||
cjk = _cjk_chars(markdown, _MIN_CJK_CHARS)
|
||||
if cjk < _MIN_CJK_CHARS:
|
||||
problems.append(
|
||||
f"PDF declares CJK fonts but the extracted text contains only "
|
||||
f"{cjk} CJK character(s)"
|
||||
)
|
||||
return problems
|
||||
|
||||
|
||||
def verify_pdf_file(file: Path, markdown: str) -> List[str]:
|
||||
"""``verify_extraction`` for a file on disk; unreadable files trust the output."""
|
||||
try:
|
||||
data = Path(file).read_bytes()
|
||||
except OSError:
|
||||
return []
|
||||
return verify_extraction(data, markdown)
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Reconstruct typographic tables in anydoc's PDF Markdown output.
|
||||
|
||||
anydoc converts PDFs straight to Markdown with no document model, so tables
|
||||
drawn with dot leaders or bare whitespace alignment (financial statements,
|
||||
tables of contents) come out as flat text lines — values intact, structure
|
||||
lost. This post-processor detects runs of such lines and rewrites them as
|
||||
GFM tables: the lightweight alternative to a table-structure model for this
|
||||
layout family. (The robust fix is upstream in anydoc's Rust PDF code, where
|
||||
glyph x-positions exist; this is the recoverable-downstream version.)
|
||||
|
||||
Deliberately conservative — a run converts only when it has at least
|
||||
``min_rows`` consecutive lines that each parse as ``label [leaders]
|
||||
numeric-columns`` with the *same* column count; anything else passes through
|
||||
untouched. Gated by ``ANYDOC_TABLEIZE`` (off by default) and applied only to
|
||||
anydoc's own PDF output, never to docling's.
|
||||
"""
|
||||
import re
|
||||
from typing import List, Optional, Tuple
|
||||
|
||||
# label ...... 1,234 (56) — dot-leader row
|
||||
_LEADER = re.compile(r"^(.*?)\s*\.{3,}\s*(.+)$")
|
||||
# numeric-ish token: $ 1,234 / (30) / 83,431 / 12.5% / —
|
||||
_NUM_TOKEN = re.compile(r"\$?\(?-?\d[\d,.]*\)?%?|—")
|
||||
# row without leaders: label text, then a trailing run of numeric tokens
|
||||
_TRAILING = re.compile(r"^(.*?[^\s\d,.])\s+([\d$(].*)$")
|
||||
|
||||
|
||||
def _merge_currency(tokens: List[str]) -> List[str]:
|
||||
"""Join a free-standing ``$`` onto the number that follows it."""
|
||||
out: List[str] = []
|
||||
for token in tokens:
|
||||
if out and out[-1] == "$":
|
||||
out[-1] = "$" + token
|
||||
else:
|
||||
out.append(token)
|
||||
return out
|
||||
|
||||
|
||||
def _parse_row(line: str) -> Optional[Tuple[str, List[str]]]:
|
||||
"""``(label, values)`` when the line looks like a typographic table row, else None."""
|
||||
match = _LEADER.match(line) or _TRAILING.match(line)
|
||||
if not match:
|
||||
return None
|
||||
label, rest = match.group(1).strip(" ."), match.group(2)
|
||||
values = _merge_currency(rest.split())
|
||||
if not values or not all(_NUM_TOKEN.fullmatch(v.lstrip("$")) for v in values):
|
||||
return None
|
||||
if not label or _NUM_TOKEN.fullmatch(label):
|
||||
return None
|
||||
return label, values
|
||||
|
||||
|
||||
def tableize(markdown: str, min_rows: int = 3) -> str:
|
||||
"""Rewrite runs of typographic table rows in ``markdown`` as GFM tables.
|
||||
|
||||
Args:
|
||||
markdown: Converter output to post-process.
|
||||
min_rows: Minimum consecutive, same-width rows for a run to convert.
|
||||
|
||||
Returns:
|
||||
``markdown`` with qualifying runs rewritten; everything else verbatim.
|
||||
"""
|
||||
out: List[str] = []
|
||||
run: List[Tuple[str, List[str], str]] = [] # (label, values, original line)
|
||||
|
||||
def flush() -> None:
|
||||
nonlocal run
|
||||
widths = {len(values) for _, values, _ in run}
|
||||
if len(run) >= min_rows and len(widths) == 1:
|
||||
ncols = widths.pop()
|
||||
out.append("")
|
||||
out.append("| " + " | ".join([" "] + [f"col{i + 1}" for i in range(ncols)]) + " |")
|
||||
out.append("|" + " --- |" * (ncols + 1))
|
||||
for label, values, _ in run:
|
||||
# Only the label can carry a '|' (values are numeric tokens);
|
||||
# unescaped it would add a cell and mis-column the row.
|
||||
cells = [label.replace("|", "\\|")] + values
|
||||
out.append("| " + " | ".join(cells) + " |")
|
||||
out.append("")
|
||||
else:
|
||||
out.extend(original for _, _, original in run)
|
||||
run = []
|
||||
|
||||
for line in markdown.splitlines():
|
||||
parsed = _parse_row(line.strip()) if line.strip() else None
|
||||
if parsed:
|
||||
run.append((*parsed, line))
|
||||
else:
|
||||
if run:
|
||||
flush()
|
||||
out.append(line)
|
||||
if run:
|
||||
flush()
|
||||
return "\n".join(out)
|
||||
@@ -1777,7 +1777,7 @@ def attachment_worker(self, file_info, user):
|
||||
parser_metadata = {
|
||||
key: value
|
||||
for key, value in (attachment_document.extra_info or {}).items()
|
||||
if key.startswith("transcript_")
|
||||
if key.startswith("transcript_") or key == "parse_warnings"
|
||||
}
|
||||
if parser_metadata:
|
||||
metadata = {**metadata, **parser_metadata}
|
||||
|
||||
@@ -213,6 +213,12 @@ for the engines and flows.
|
||||
| `DOCLING_TABULAR_MAX_BYTES` | `2000000` | Docling engine: CSV/XLSX larger than this (by content) use the lightweight tabular parsers. |
|
||||
| `DOCLING_MARKUP_MAX_BYTES` | `8000000` | Docling engine: HTML/VTT larger than this are head-truncated before parsing. |
|
||||
| `MARKUP_MAX_BYTES` | `8000000` | anydoc engine: HTML/XHTML larger than this are head-truncated before the `markdownify` parser runs (its tree costs ~50x the input). `0` disables the gate. |
|
||||
| `PDF_TRUST_CHECK` | `true` | Trust-check anydoc's PDF output for the two silent-loss classes (Type0 fonts without ToUnicode, CJK-declaring PDFs with CJK-less text). Flagged files re-parse on Docling when installed, else carry `parse_warnings` metadata. |
|
||||
| `ANYDOC_TABLEIZE` | `false` | Rewrite dot-leader / whitespace-aligned table runs in anydoc's PDF markdown into GFM tables. |
|
||||
| `OCR_ENGINE` | `tesseract` | OCR engine when OCR is on: `tesseract` (recommended), `auto`, `ocrmac`, `rapidocr`, or `deepseek` (VLM endpoint). Unavailable engines degrade to `auto` with a warning. See the [OCR guide](/Guides/ocr). |
|
||||
| `OCR_LANGS` | `eng` | Tesseract language packs, `+`-separated (e.g. `eng+chi_sim+deu`). |
|
||||
| `OCR_DEEPSEEK_URL` | `http://localhost:11434/v1/chat/completions` | OpenAI-compatible endpoint for `OCR_ENGINE=deepseek` (Ollama or vLLM). |
|
||||
| `OCR_DEEPSEEK_MODEL` | `deepseek-ocr:3b` | Model name at that endpoint. |
|
||||
|
||||
## Speech-to-Text Settings
|
||||
|
||||
|
||||
@@ -16,7 +16,7 @@ Training on other documentation sources can greatly enhance the versatility and
|
||||
Make sure you have the document on which you want to train on ready with you on the device which you are using .You can also use links to the documentation to train on.
|
||||
|
||||
<Callout type="warning" emoji="⚠️">
|
||||
Note: Supported file formats include .pdf, .txt, .rst, .docx, .md, .mdx, .csv, .epub, .html, .json, .xlsx, .pptx, .png, .jpg, .jpeg, and audio files (.wav, .mp3, .m4a, .ogg, .webm). You can also train using the link of the documentation.
|
||||
Note: Supported file formats include .pdf, .txt, .rst, .md, .mdx, .html, .xhtml, .json, .epub, Office documents (.docx, .doc, .docm, .odt, .rtf), spreadsheets (.xlsx, .xls, .xlsm, .xlsb, .ods, .csv), presentations (.pptx, .ppt, .pptm, .pps, .ppsx, .ppsm, .pot, .odp), images (.png, .jpg, .jpeg), and audio files (.wav, .mp3, .m4a, .ogg, .webm). You can also train using the link of the documentation.
|
||||
|
||||
</Callout>
|
||||
|
||||
|
||||
@@ -66,15 +66,52 @@ docling is picked up as the fallback engine as soon as it is importable, and
|
||||
|
||||
## Docling OCR
|
||||
|
||||
OCR is optional, Docling-only, and controlled by two settings:
|
||||
OCR is optional, Docling-only, and controlled by two on/off settings plus an
|
||||
engine choice:
|
||||
|
||||
```env
|
||||
DOCLING_OCR_ENABLED=false
|
||||
DOCLING_OCR_ATTACHMENTS_ENABLED=false
|
||||
OCR_ENGINE=tesseract
|
||||
```
|
||||
|
||||
- `DOCLING_OCR_ENABLED`: OCR behavior for Source Docs ingestion.
|
||||
- `DOCLING_OCR_ATTACHMENTS_ENABLED`: OCR behavior for chat attachments uploaded from the message box.
|
||||
- `OCR_ENGINE`: which engine performs the OCR when it is on (next section).
|
||||
|
||||
Under the default anydoc engine, a scanned PDF reaches OCR through anydoc's
|
||||
own detection: anydoc refuses it ("OCR is required") and the docling fallback
|
||||
takes over. If that fallback also extracts almost nothing — OCR off, or no
|
||||
docling installed — the upload now fails with a clear message instead of
|
||||
silently indexing an empty document.
|
||||
|
||||
## Choosing the OCR engine
|
||||
|
||||
Benchmarked 2026-08 on English, bilingual EN/ZH, table-heavy and degraded
|
||||
scans (all engines driven through docling so layout handling is identical):
|
||||
|
||||
| `OCR_ENGINE` | Role | Notes |
|
||||
|---|---|---|
|
||||
| `tesseract` | **recommended default** | Best classic-engine accuracy in the bench: perfect EN word recall on all docs, 0.000 CER on the bilingual page, 100% table cells, robust to mild degradation. ~35 MB of *system* packages, CPU-only. Needs the `tesseract` binary + language packs in the image — installed automatically when the Docker image is built with `INSTALL_DOCLING=true`; set languages via `OCR_LANGS` (e.g. `eng+chi_sim`). |
|
||||
| `auto` | convenience | docling picks: `ocrmac` on macOS (excellent, ~1 s/page), `rapidocr` on Linux — see below before relying on it server-side. Also the automatic fallback whenever the selected engine is not installed. |
|
||||
| `ocrmac` | macOS only | Best raw accuracy and fastest of all classic engines; irrelevant for Linux deploys. |
|
||||
| `rapidocr` | pip-only fallback | No system packages needed, perfect on tables/CJK — but it silently shreds some long text lines into garbage at every setting tried, which is content loss for RAG ingestion. Avoid as a server default until fixed upstream. |
|
||||
| `deepseek` | best quality, heavy on the *server* | DeepSeek-OCR through docling's VLM pipeline. Only engine that reconstructs totals rows as table rows; near-perfect CJK; barely affected by degradation. The ingestion worker gets *lighter* (~370 MB RSS — no layout models); the model runs in Ollama or vLLM. Costs: a GPU/Apple-Silicon endpoint, ~seconds per page, and occasional silent drops of page-level elements (titles). |
|
||||
|
||||
For `deepseek`, point the worker at an OpenAI-compatible endpoint:
|
||||
|
||||
```env
|
||||
OCR_ENGINE=deepseek
|
||||
OCR_DEEPSEEK_URL=http://localhost:11434/v1/chat/completions # Ollama default
|
||||
OCR_DEEPSEEK_MODEL=deepseek-ocr:3b
|
||||
```
|
||||
|
||||
Ollama works out of the box (`ollama pull deepseek-ocr:3b`); for real
|
||||
throughput serve `deepseek-ai/DeepSeek-OCR` with vLLM on a GPU and set the
|
||||
URL accordingly.
|
||||
|
||||
A selected engine that is not available (no tesseract binary, no macOS for
|
||||
ocrmac) degrades to `auto` with a warning rather than failing parses.
|
||||
|
||||
## Processing Flow
|
||||
|
||||
@@ -104,10 +141,17 @@ Docling OCR behavior is different for PDFs vs images:
|
||||
- bitmap/image regions: OCR only where needed
|
||||
- Image parser defaults to full-page OCR (the whole image is visual content).
|
||||
|
||||
By default, Docling parser classes use RapidOCR options (language default: `english`).
|
||||
The engine and its languages come from `OCR_ENGINE` and `OCR_LANGS` (see the
|
||||
table above). The Docker image ships only the English tesseract pack; for
|
||||
other languages install their packs in the image (e.g.
|
||||
`apt-get install tesseract-ocr-chi-sim`) and list them in `OCR_LANGS`
|
||||
(`eng+chi_sim`).
|
||||
|
||||
<Callout type="info" emoji="ℹ️">
|
||||
Parser internals like OCR language and force-full-page OCR are currently set by code defaults, not separate `.env` settings.
|
||||
<Callout type="warning" emoji="⚠️">
|
||||
Upgrading from a RapidOCR-based deployment? RapidOCR covered English *and*
|
||||
Chinese with no configuration. The tesseract default only OCRs the languages
|
||||
in `OCR_LANGS` (`eng` out of the box), so CJK scans stop ingesting until
|
||||
their packs are installed and listed.
|
||||
</Callout>
|
||||
|
||||
### Model compilation
|
||||
|
||||
@@ -26,6 +26,23 @@ export const FILE_UPLOAD_ACCEPT: Record<string, string[]> = {
|
||||
'application/vnd.openxmlformats-officedocument.presentationml.presentation': [
|
||||
'.pptx',
|
||||
],
|
||||
'application/xhtml+xml': ['.xhtml'],
|
||||
'application/msword': ['.doc'],
|
||||
'application/vnd.ms-word.document.macroEnabled.12': ['.docm'],
|
||||
'application/vnd.oasis.opendocument.text': ['.odt'],
|
||||
'application/rtf': ['.rtf'],
|
||||
'text/rtf': ['.rtf'],
|
||||
'application/vnd.ms-powerpoint': ['.ppt', '.pps', '.pot'],
|
||||
'application/vnd.ms-powerpoint.presentation.macroEnabled.12': ['.pptm'],
|
||||
'application/vnd.openxmlformats-officedocument.presentationml.slideshow': [
|
||||
'.ppsx',
|
||||
],
|
||||
'application/vnd.ms-powerpoint.slideshow.macroEnabled.12': ['.ppsm'],
|
||||
'application/vnd.oasis.opendocument.presentation': ['.odp'],
|
||||
'application/vnd.ms-excel': ['.xls'],
|
||||
'application/vnd.ms-excel.sheet.macroEnabled.12': ['.xlsm'],
|
||||
'application/vnd.ms-excel.sheet.binary.macroEnabled.12': ['.xlsb'],
|
||||
'application/vnd.oasis.opendocument.spreadsheet': ['.ods'],
|
||||
'image/png': ['.png'],
|
||||
'image/jpeg': ['.jpeg'],
|
||||
'image/jpg': ['.jpg'],
|
||||
@@ -45,6 +62,22 @@ export const FILE_UPLOAD_ACCEPT_ATTR = [
|
||||
'.epub',
|
||||
'.xlsx',
|
||||
'.pptx',
|
||||
'.xhtml',
|
||||
'.doc',
|
||||
'.docm',
|
||||
'.odt',
|
||||
'.rtf',
|
||||
'.ppt',
|
||||
'.pptm',
|
||||
'.pps',
|
||||
'.ppsx',
|
||||
'.ppsm',
|
||||
'.pot',
|
||||
'.odp',
|
||||
'.xls',
|
||||
'.xlsm',
|
||||
'.xlsb',
|
||||
'.ods',
|
||||
'.png',
|
||||
'.jpeg',
|
||||
'.jpg',
|
||||
@@ -76,6 +109,22 @@ export const SOURCE_FILE_TREE_ACCEPT_ATTR = [
|
||||
'.json',
|
||||
'.xlsx',
|
||||
'.pptx',
|
||||
'.xhtml',
|
||||
'.doc',
|
||||
'.docm',
|
||||
'.odt',
|
||||
'.rtf',
|
||||
'.ppt',
|
||||
'.pptm',
|
||||
'.pps',
|
||||
'.ppsx',
|
||||
'.ppsm',
|
||||
'.pot',
|
||||
'.odp',
|
||||
'.xls',
|
||||
'.xlsm',
|
||||
'.xlsb',
|
||||
'.ods',
|
||||
'.png',
|
||||
'.jpg',
|
||||
'.jpeg',
|
||||
|
||||
@@ -765,7 +765,7 @@
|
||||
"start": "Chat starten",
|
||||
"name": "Name",
|
||||
"choose": "Dateien auswählen",
|
||||
"info": "Bitte lade .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip hoch (max. 25 MB)",
|
||||
"info": "Bitte lade .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip hoch (max. 25 MB)",
|
||||
"uploadedFiles": "Hochgeladene Dateien",
|
||||
"cancel": "Abbrechen",
|
||||
"train": "Trainieren",
|
||||
|
||||
@@ -770,7 +770,7 @@
|
||||
"start": "Start Chatting",
|
||||
"name": "Name",
|
||||
"choose": "Choose Files",
|
||||
"info": "Please upload .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip limited to 25mb",
|
||||
"info": "Please upload .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip limited to 25mb",
|
||||
"uploadedFiles": "Uploaded Files",
|
||||
"cancel": "Cancel",
|
||||
"train": "Train",
|
||||
|
||||
@@ -765,7 +765,7 @@
|
||||
"start": "Comenzar a chatear",
|
||||
"name": "Nombre",
|
||||
"choose": "Seleccionar Archivos",
|
||||
"info": "Por favor, sube archivos .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip limitados a 25MB",
|
||||
"info": "Por favor, sube archivos .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip limitados a 25MB",
|
||||
"uploadedFiles": "Archivos Subidos",
|
||||
"cancel": "Cancelar",
|
||||
"train": "Entrenar",
|
||||
|
||||
@@ -765,7 +765,7 @@
|
||||
"start": "チャットを開始する",
|
||||
"name": "名前",
|
||||
"choose": "ファイルを選択",
|
||||
"info": "25MBまでの.pdf、.txt、.rst、.csv、.xlsx、.docx、.md、.html、.epub、.json、.pptx、.zipファイルをアップロードしてください",
|
||||
"info": "25MBまでの.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.epub、.json、.pptx、.ppt、.odp、.zipファイルをアップロードしてください",
|
||||
"uploadedFiles": "アップロードされたファイル",
|
||||
"cancel": "キャンセル",
|
||||
"train": "トレーニング",
|
||||
|
||||
@@ -765,7 +765,7 @@
|
||||
"start": "Начать чат",
|
||||
"name": "Имя",
|
||||
"choose": "Выбрать файлы",
|
||||
"info": "Пожалуйста, загрузите файлы .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip размером до 25 МБ",
|
||||
"info": "Пожалуйста, загрузите файлы .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip размером до 25 МБ",
|
||||
"uploadedFiles": "Загруженные файлы",
|
||||
"cancel": "Отмена",
|
||||
"train": "Тренировка",
|
||||
|
||||
@@ -765,7 +765,7 @@
|
||||
"start": "開始對話",
|
||||
"name": "名稱",
|
||||
"choose": "選擇檔案",
|
||||
"info": "請上傳限制為25MB的.pdf、.txt、.rst、.csv、.xlsx、.docx、.md、.html、.epub、.json、.pptx、.zip檔案",
|
||||
"info": "請上傳限制為25MB的.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.epub、.json、.pptx、.ppt、.odp、.zip檔案",
|
||||
"uploadedFiles": "已上傳檔案",
|
||||
"cancel": "取消",
|
||||
"train": "訓練",
|
||||
|
||||
@@ -765,7 +765,7 @@
|
||||
"start": "开始聊天",
|
||||
"name": "名称",
|
||||
"choose": "选择文件",
|
||||
"info": "请上传限制为25MB的.pdf、.txt、.rst、.csv、.xlsx、.docx、.md、.html、.epub、.json、.pptx、.zip文件",
|
||||
"info": "请上传限制为25MB的.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.epub、.json、.pptx、.ppt、.odp、.zip文件",
|
||||
"uploadedFiles": "已上传文件",
|
||||
"cancel": "取消",
|
||||
"train": "训练",
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
%PDF-1.4
|
||||
%“Œ‹ž ReportLab Generated PDF document (opensource)
|
||||
1 0 obj
|
||||
<<
|
||||
/F1 2 0 R /F2 3 0 R /F3 4 0 R /F4 5 0 R
|
||||
>>
|
||||
endobj
|
||||
2 0 obj
|
||||
<<
|
||||
/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font
|
||||
>>
|
||||
endobj
|
||||
3 0 obj
|
||||
<<
|
||||
/BaseFont /Times-Bold /Encoding /WinAnsiEncoding /Name /F2 /Subtype /Type1 /Type /Font
|
||||
>>
|
||||
endobj
|
||||
4 0 obj
|
||||
<<
|
||||
/BaseFont /STSong-Light /DescendantFonts [ <<
|
||||
/BaseFont /STSong-Light /CIDSystemInfo <<
|
||||
/Ordering (GB1) /Registry (Adobe) /Supplement 0
|
||||
>> /DW 1000 /FontDescriptor <<
|
||||
/Ascent 752 /CapHeight 737 /Descent -271 /Flags 6 /FontBBox [ -25 -254 1000 880 ] /FontName /STSongStd-Light
|
||||
/ItalicAngle 0 /Leading 148 /MaxWidth 1000 /MissingWidth 500 /StemH 91 /StemV 58
|
||||
/Type /FontDescriptor /XHeight 553
|
||||
>> /Subtype /CIDFontType0 /Type /Font
|
||||
/W [ 1 [ 207 270 342 467 462 797 710 239 374 ] 10 [ 374 423 605 238 375 238 334 462 ] 18 26 462 27 28 238
|
||||
29 31 605 32 [ 344 748 684 560 695 739 563 511 729 793
|
||||
318 312 666 526 896 758 772 544 772 628
|
||||
465 607 753 711 972 647 620 607 374 333
|
||||
374 606 500 239 417 503 427 529 415 264
|
||||
444 518 241 230 495 228 793 527 524 ] 81 [ 524 504 338 336 277 517 450 652 466 452
|
||||
407 370 258 370 605 ] ]
|
||||
>> ] /Encoding /UniGB-UCS2-H /Name /F3 /Subtype /Type0 /Type /Font
|
||||
>>
|
||||
endobj
|
||||
5 0 obj
|
||||
<<
|
||||
/BaseFont /Times-Roman /Encoding /WinAnsiEncoding /Name /F4 /Subtype /Type1 /Type /Font
|
||||
>>
|
||||
endobj
|
||||
6 0 obj
|
||||
<<
|
||||
/Contents 10 0 R /MediaBox [ 0 0 595.2756 841.8898 ] /Parent 9 0 R /Resources <<
|
||||
/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ]
|
||||
>> /Rotate 0 /Trans <<
|
||||
|
||||
>>
|
||||
/Type /Page
|
||||
>>
|
||||
endobj
|
||||
7 0 obj
|
||||
<<
|
||||
/PageMode /UseNone /Pages 9 0 R /Type /Catalog
|
||||
>>
|
||||
endobj
|
||||
8 0 obj
|
||||
<<
|
||||
/Author (\(anonymous\)) /CreationDate (D:20260822115950+02'00') /Creator (\(unspecified\)) /Keywords () /ModDate (D:20260822115950+02'00') /Producer (ReportLab PDF Library - \(opensource\))
|
||||
/Subject (\(unspecified\)) /Title (\(anonymous\)) /Trapped /False
|
||||
>>
|
||||
endobj
|
||||
9 0 obj
|
||||
<<
|
||||
/Count 1 /Kids [ 6 0 R ] /Type /Pages
|
||||
>>
|
||||
endobj
|
||||
10 0 obj
|
||||
<<
|
||||
/Filter [ /ASCII85Decode /FlateDecode ] /Length 1647
|
||||
>>
|
||||
stream
|
||||
Gatm<gN):C&:O:SoKt6S$HmLQ&3Rqa-FV*"%Nm-aMW-4=#+,j_=@nmi2Z)t-)*e6=$[Ld,&=Q79:X>PPVJQYC5?'k&+i&d>B>799%,;FkdkJW*5heHuF@91dXh%AJZ6fb3i'sKhCnMdY'0VkEO4GA.p;"DJY62!"YAIse&.WT60%ek]="]O%+5Ot<#Fl%^o]il7fm2dGp=dm)m^:gDD9"d?pER14WDHmu']GE^2k$Vs`2KagWk")2f26\M]Bs+K@uNQtga3OQkLQ7H-hY4>1!4.4XoK+nl.ZlK#$d"IN+=9q.'8T'8cF3g*[-3:LS5c>fi!l0mSVa4QpjYgA<$0LC`2-c:@^;pCa)ekN>!I%&S3he7oPPIfI^JJ8"'-"K,>QL6DaesnS'ThOk/)63Q<B9b;^E^!3N(^L"g8T-_[_e%ED-JKcd/L:T#=nOk`'\l#7Q=,(RK;`0#S<T4BU:`HhPGE"*QrTjD+LciNhe\[arIY[:*6r!Z(b46[T;c]'eT2ib4:#PYiS4KbfJkaWh^D3%L,2kV2tNtL-eVVV!V=WcZC*LmZ1JF,Q`_WpI/i"9jUr1Gd(ISB[Id?[]g@-RLQD"d?"iI[81]"&B'nLTX>D<J'G5>]1d]+<&nMsjUuVm%-`_Dn^N7Y3@(%aO`>?OaQR7"S`I)*4n^gDS'@E((C?3'qDA#_!0S=@#*`D<h%J/PD"f$4CY;-(VXO;.Ngc^I+,-r\]qpUQHC*nl`bn!js*R3SkJqAMhH<a9;!@-5%1$9AH:,n;b&'!\_DddM7mc;^=s`,BOb&%SnDA7Rl;HLN*t%(`jGJS&".8T-M[a.`>blJkd8M[QOjG@Lffk@WSsIihjLDN`q?@n!YaFRc6etUSZ3TeZg#m?mL-MTMO5V)b]\Em&*E2^jCc`Xsk1<"JLgnj,EWu^Kg\"3.u<u5YCaES/ZDf5DpeMd24qDipat:7Oab*_jqT_s6MJ(i_J60;p<ik,/MGX2+u90)Xt!desS2A+)qeT6sUK/IJ'C=1<0u0>SaqZ<FhNoMfaf*!B2`NiLP;eB4?[`;/Y%o&Gm)'qO1JK0uckb!lh3>=t73#j/BK3<dLaZd96%gq]>W+8,V.,0D<(=M`#OG+,1-Fh</5[IM)#q0>k-!*=d[eI7&!b)UOUK$3XO.Y5HWgM[D<SAa`SCR9MN]H2Q[$5(+ZDUq;e-9V7T89Qo[Rm`4fG0o4@V.p48_A)/:trd:.`nh*S%-%oo[?Go6OFLjPL]dfcTW*/@I<%u%sqi96p7;=;:bYI+:AU5':P>$c<"Wh_I_.r/0:j+[",VmP^3"R7]Vuj1Bju_M4fQpgg+K1-[?e$i#W^p3[3m^DdEM$S$d\G^r%4.<YC,K9<kt81fnor-c7+%aA'BW<5M_bT$PN_@IR!Yssf=Ie",juu4f]0+Uq".M]_sN3S\fN4OR!lJL2(a=:b4Ff&7E;q)O_P#WqT?;^@'e:m]2q4Y\']#rYS`0J,.E<b/Dai'5?mlc6AhJna\JGg_:Yc:98:nsRLal!#c+EsM(<tb(J^cTEVjq!Ll`V)/=\n1?2VP!nUibVne=IuE;BNdBFoimp4D_?%=$%6*-QW3f0rnX&<H85'1fg=@QlClK`^N0$X\\#1DFp#6kjo8*>*l66RfIdu>~>endstream
|
||||
endobj
|
||||
xref
|
||||
0 11
|
||||
0000000000 65535 f
|
||||
0000000061 00000 n
|
||||
0000000122 00000 n
|
||||
0000000229 00000 n
|
||||
0000000337 00000 n
|
||||
0000001270 00000 n
|
||||
0000001379 00000 n
|
||||
0000001583 00000 n
|
||||
0000001651 00000 n
|
||||
0000001931 00000 n
|
||||
0000001990 00000 n
|
||||
trailer
|
||||
<<
|
||||
/ID
|
||||
[<6167f87b12d550221d4edd2d2c131d94><6167f87b12d550221d4edd2d2c131d94>]
|
||||
% ReportLab generated PDF document -- digest (opensource)
|
||||
|
||||
/Info 8 0 R
|
||||
/Root 7 0 R
|
||||
/Size 11
|
||||
>>
|
||||
startxref
|
||||
3729
|
||||
%%EOF
|
||||
@@ -27,10 +27,14 @@ from application.parser.file.anydoc_parser import ( # noqa: E402 — after impo
|
||||
)
|
||||
|
||||
|
||||
# Long enough to clear the scanned-PDF near-empty guard (_MIN_SCAN_FALLBACK_CHARS).
|
||||
FALLBACK_TEXT = "fallback parser text output, long enough to clear the scanned-PDF guard."
|
||||
|
||||
|
||||
class _RecordingFallback(BaseParser):
|
||||
"""Stands in for docling / a legacy parser so delegation is observable."""
|
||||
|
||||
def __init__(self, result="fallback text"):
|
||||
def __init__(self, result=FALLBACK_TEXT):
|
||||
super().__init__(parser_config={})
|
||||
self.calls = []
|
||||
self._result = result
|
||||
@@ -146,7 +150,7 @@ def test_scanned_pdf_delegates_to_fallback(tmp_path):
|
||||
|
||||
out = parser.parse_file(path)
|
||||
|
||||
assert out == "fallback text"
|
||||
assert out == FALLBACK_TEXT
|
||||
assert fallback.calls == [path]
|
||||
assert parser.last_engine == "_RecordingFallback"
|
||||
|
||||
@@ -229,7 +233,7 @@ def test_typed_convert_error_delegates(tmp_path, monkeypatch):
|
||||
path = tmp_path / "x.pdf"
|
||||
path.write_bytes(b"%PDF-1.4")
|
||||
|
||||
assert AnydocParser(fallback_parser=fallback).parse_file(path) == "fallback text"
|
||||
assert AnydocParser(fallback_parser=fallback).parse_file(path) == FALLBACK_TEXT
|
||||
|
||||
|
||||
def test_typed_convert_error_message_reaches_the_user(tmp_path, monkeypatch):
|
||||
@@ -311,3 +315,222 @@ def test_init_parser_imports_for_real_not_just_find_spec(monkeypatch):
|
||||
|
||||
with pytest.raises(ImportError, match="firecrawl-anydoc"):
|
||||
AnydocParser().init_parser()
|
||||
|
||||
|
||||
# --- PDF trust check + tableize wiring (PR 3) -----------------------------------
|
||||
|
||||
from pathlib import Path as _Path
|
||||
|
||||
FIXTURES = _Path(__file__).parent / "fixtures"
|
||||
CID_PDF = FIXTURES / "nda_en_zh_cid_font.pdf"
|
||||
|
||||
|
||||
# Long enough to clear the near-empty guard a trust-check re-parse must pass.
|
||||
_REROUTE_TEXT = "docling reroute text, long enough to clear the near-empty reroute guard"
|
||||
|
||||
|
||||
class _FakeDoclingFallback:
|
||||
"""Registered as a DoclingParser subclass so ``_is_docling_backed`` is True."""
|
||||
|
||||
def __new__(cls):
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
class _Inner(DoclingParser):
|
||||
def __init__(self):
|
||||
self._parser_config = {}
|
||||
self.calls = []
|
||||
self.last_engine = None
|
||||
|
||||
def parse_file(self, file, errors="ignore"):
|
||||
self.calls.append(file)
|
||||
return _REROUTE_TEXT
|
||||
|
||||
return _Inner()
|
||||
|
||||
|
||||
def test_trust_flagged_pdf_reroutes_to_docling_fallback():
|
||||
fallback = _FakeDoclingFallback()
|
||||
parser = AnydocParser(fallback_parser=fallback)
|
||||
|
||||
out = parser.parse_file(CID_PDF)
|
||||
|
||||
assert out == _REROUTE_TEXT
|
||||
assert fallback.calls == [CID_PDF]
|
||||
assert parser.last_engine == "_Inner"
|
||||
assert parser.get_file_metadata(CID_PDF) == {} # rerouted, nothing to warn about
|
||||
|
||||
|
||||
def test_trust_flagged_pdf_without_docling_keeps_output_and_warns():
|
||||
parser = AnydocParser(fallback_parser=_RecordingFallback()) # not docling-backed
|
||||
|
||||
out = parser.parse_file(CID_PDF)
|
||||
|
||||
assert "NON-DISCLOSURE" in out.upper() or len(out) > 50 # anydoc's own output kept
|
||||
meta = parser.get_file_metadata(CID_PDF)
|
||||
assert "parse_warnings" in meta
|
||||
assert any("ToUnicode" in w for w in meta["parse_warnings"])
|
||||
assert any("CJK" in w for w in meta["parse_warnings"])
|
||||
assert parser.last_engine == "anydoc"
|
||||
|
||||
|
||||
def test_trust_check_disabled_stamps_nothing(monkeypatch):
|
||||
from application.parser.file import anydoc_parser as ap
|
||||
|
||||
monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False)
|
||||
parser = AnydocParser()
|
||||
|
||||
parser.parse_file(CID_PDF)
|
||||
|
||||
assert parser.get_file_metadata(CID_PDF) == {}
|
||||
|
||||
|
||||
def test_trust_reroute_failure_keeps_anydoc_output():
|
||||
fallback = _FakeDoclingFallback()
|
||||
|
||||
def _boom(file, errors="ignore"):
|
||||
raise RuntimeError("docling exploded")
|
||||
|
||||
fallback.parse_file = _boom
|
||||
parser = AnydocParser(fallback_parser=fallback)
|
||||
|
||||
out = parser.parse_file(CID_PDF)
|
||||
|
||||
assert "docling" not in out
|
||||
assert "parse_warnings" in parser.get_file_metadata(CID_PDF)
|
||||
assert parser.last_engine == "anydoc"
|
||||
|
||||
|
||||
def test_trust_reroute_near_empty_keeps_anydoc_output():
|
||||
"""A docling pipeline dropout ('' / '<!-- image -->') returns without
|
||||
raising; adopting it would swap anydoc's real text for an empty document."""
|
||||
fallback = _FakeDoclingFallback()
|
||||
fallback.parse_file = lambda file, errors="ignore": "<!-- image -->"
|
||||
parser = AnydocParser(fallback_parser=fallback)
|
||||
|
||||
out = parser.parse_file(CID_PDF)
|
||||
|
||||
assert "<!-- image -->" not in out
|
||||
assert "parse_warnings" in parser.get_file_metadata(CID_PDF)
|
||||
assert parser.last_engine == "anydoc"
|
||||
|
||||
|
||||
def test_warnings_reset_between_files(tmp_path):
|
||||
parser = AnydocParser()
|
||||
parser.parse_file(CID_PDF)
|
||||
assert parser.get_file_metadata(CID_PDF) != {}
|
||||
|
||||
clean = tmp_path / "clean.csv"
|
||||
clean.write_text("a,b\n1,2\n")
|
||||
parser.parse_file(clean)
|
||||
|
||||
assert parser.get_file_metadata(clean) == {}
|
||||
assert parser.get_file_metadata(CID_PDF) == {}
|
||||
|
||||
|
||||
def test_trust_check_errors_never_fail_the_parse(monkeypatch, tmp_path):
|
||||
def _explode(path, markdown):
|
||||
raise RuntimeError("scanner bug")
|
||||
|
||||
import application.parser.file.pdf_trust as pt
|
||||
|
||||
monkeypatch.setattr(pt, "verify_pdf_file", _explode)
|
||||
_fake_anydoc(monkeypatch, lambda path: "# converted fine")
|
||||
path = tmp_path / "x.pdf"
|
||||
path.write_bytes(b"%PDF-1.4 irrelevant")
|
||||
|
||||
assert AnydocParser().parse_file(path) == "# converted fine"
|
||||
|
||||
|
||||
def test_tableize_applied_when_enabled(monkeypatch, tmp_path):
|
||||
from application.parser.file import anydoc_parser as ap
|
||||
|
||||
monkeypatch.setattr(ap.settings, "ANYDOC_TABLEIZE", True)
|
||||
monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False)
|
||||
_fake_anydoc(
|
||||
monkeypatch,
|
||||
lambda path: "Cash ..... 1,234 900\nDebt ..... 2,000 1,500\nEquity ..... 900 800",
|
||||
)
|
||||
path = tmp_path / "x.pdf"
|
||||
path.write_bytes(b"%PDF-1.4 irrelevant")
|
||||
|
||||
out = AnydocParser().parse_file(path)
|
||||
|
||||
assert "| Cash | 1,234 | 900 |" in out
|
||||
|
||||
|
||||
def test_tableize_disabled_by_default(monkeypatch, tmp_path):
|
||||
from application.parser.file import anydoc_parser as ap
|
||||
|
||||
monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False)
|
||||
flat = "Cash ..... 1,234 900\nDebt ..... 2,000 1,500\nEquity ..... 900 800"
|
||||
_fake_anydoc(monkeypatch, lambda path: flat)
|
||||
path = tmp_path / "x.pdf"
|
||||
path.write_bytes(b"%PDF-1.4 irrelevant")
|
||||
|
||||
assert AnydocParser().parse_file(path) == flat
|
||||
|
||||
|
||||
def test_tableize_never_touches_docling_reroute(monkeypatch):
|
||||
from application.parser.file import anydoc_parser as ap
|
||||
|
||||
monkeypatch.setattr(ap.settings, "ANYDOC_TABLEIZE", True)
|
||||
fallback = _FakeDoclingFallback()
|
||||
parser = AnydocParser(fallback_parser=fallback)
|
||||
|
||||
out = parser.parse_file(CID_PDF)
|
||||
|
||||
assert out == _REROUTE_TEXT # verbatim, not post-processed
|
||||
|
||||
|
||||
# --- scanned-PDF near-empty guard (the OCR_ENGINE seam's loud-failure side) ------
|
||||
|
||||
|
||||
def test_scanned_pdf_with_near_empty_fallback_fails_loudly_when_ocr_off(tmp_path):
|
||||
"""A scan whose fallback (OCR off) extracts almost nothing must fail with
|
||||
an actionable message, not be stored as an empty document."""
|
||||
path = _scanned_pdf(tmp_path / "scan.pdf")
|
||||
fallback = _RecordingFallback(result=" ")
|
||||
fallback.ocr_enabled = False
|
||||
parser = AnydocParser(fallback_parser=fallback)
|
||||
|
||||
with pytest.raises(DocumentParseError, match="DOCLING_OCR_ENABLED"):
|
||||
parser.parse_file(path)
|
||||
assert parser.last_engine is None
|
||||
|
||||
|
||||
def test_scanned_pdf_with_near_empty_fallback_and_ocr_on_reports_it(tmp_path):
|
||||
path = _scanned_pdf(tmp_path / "scan.pdf")
|
||||
fallback = _RecordingFallback(result="x")
|
||||
fallback.ocr_enabled = True
|
||||
|
||||
with pytest.raises(DocumentParseError, match="even with OCR enabled"):
|
||||
AnydocParser(fallback_parser=fallback).parse_file(path)
|
||||
|
||||
|
||||
def test_scanned_pdf_with_substantial_fallback_output_passes(tmp_path):
|
||||
"""docling extracting a text layer anydoc refused (the HK-bill case) must
|
||||
keep working — the guard only fires on near-empty results."""
|
||||
path = _scanned_pdf(tmp_path / "scan.pdf")
|
||||
out = AnydocParser(fallback_parser=_RecordingFallback()).parse_file(path)
|
||||
assert out == FALLBACK_TEXT
|
||||
|
||||
|
||||
def test_non_scan_refusals_do_not_trigger_the_guard(tmp_path, monkeypatch):
|
||||
"""Only the OCR-required refusal implies 'content exists but needs OCR';
|
||||
a malformed file with a short fallback result stays a successful parse."""
|
||||
fake = _fake_anydoc(monkeypatch, None)
|
||||
|
||||
class MalformedError(fake.ConvertError):
|
||||
pass
|
||||
|
||||
fake.MalformedError = MalformedError
|
||||
|
||||
def _refuse(path):
|
||||
raise MalformedError("structurally unusable")
|
||||
|
||||
fake.to_markdown = _refuse
|
||||
path = tmp_path / "x.docx"
|
||||
path.write_bytes(b"irrelevant")
|
||||
|
||||
out = AnydocParser(fallback_parser=_RecordingFallback(result="tiny")).parse_file(path)
|
||||
assert out == "tiny"
|
||||
@@ -628,3 +628,115 @@ class TestParserEngineSwitch:
|
||||
assert ".xhtml" in extractor
|
||||
for suffix in (".pdf", ".docx", ".csv", ".xlsx", ".html", ".xhtml", ".pptx"):
|
||||
extractor[suffix].init_parser()
|
||||
|
||||
|
||||
# =====================================================================
|
||||
# Gained formats (anydoc-only: legacy/macro Office, OpenDocument, RTF)
|
||||
# =====================================================================
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestGainedFormats:
|
||||
def test_every_supported_extension_has_a_parser_under_both_engines(self):
|
||||
"""The invariant that keeps constants.py and the parser maps in lockstep.
|
||||
|
||||
Without it, an extension accepted at upload but missing from the map
|
||||
falls to SimpleDirectoryReader's plain-text read — for a binary
|
||||
format that means mojibake silently ingested and embedded.
|
||||
(`.txt` is the one deliberate plain-text read.)
|
||||
"""
|
||||
pytest.importorskip("anydoc")
|
||||
from application.parser.file.bulk import get_default_file_extractor
|
||||
from application.parser.file.constants import (
|
||||
SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS,
|
||||
)
|
||||
|
||||
for engine in ("anydoc", "docling"):
|
||||
extractor = get_default_file_extractor(engine=engine)
|
||||
missing = [
|
||||
suffix
|
||||
for suffix in SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS
|
||||
if suffix != ".txt" and suffix not in extractor
|
||||
]
|
||||
assert missing == [], (engine, missing)
|
||||
|
||||
def test_gained_formats_map_to_anydoc_under_both_engines(self):
|
||||
pytest.importorskip("anydoc")
|
||||
from application.parser.file.anydoc_parser import (
|
||||
ANYDOC_GAINED_SUFFIXES,
|
||||
AnydocParser,
|
||||
)
|
||||
from application.parser.file.bulk import get_default_file_extractor
|
||||
|
||||
for engine in ("anydoc", "docling"):
|
||||
extractor = get_default_file_extractor(engine=engine)
|
||||
for suffix in ANYDOC_GAINED_SUFFIXES:
|
||||
assert isinstance(extractor[suffix], AnydocParser), (engine, suffix)
|
||||
|
||||
def test_gained_formats_never_fall_back_to_anydoc_itself(self):
|
||||
pytest.importorskip("anydoc")
|
||||
from application.parser.file.anydoc_parser import AnydocParser
|
||||
from application.parser.file.bulk import get_default_file_extractor
|
||||
|
||||
extractor = get_default_file_extractor(engine="anydoc")
|
||||
assert extractor[".doc"].fallback_parser is None
|
||||
assert extractor[".rtf"].fallback_parser is None
|
||||
# ...while the core five keep their real fallback.
|
||||
assert extractor[".pdf"].fallback_parser is not None
|
||||
assert not isinstance(extractor[".pdf"].fallback_parser, AnydocParser)
|
||||
|
||||
def test_gained_entries_absent_without_anydoc(self, monkeypatch):
|
||||
import sys
|
||||
|
||||
from application.parser.file.bulk import get_default_file_extractor
|
||||
|
||||
monkeypatch.setitem(sys.modules, "anydoc", None)
|
||||
extractor = get_default_file_extractor(engine="docling")
|
||||
assert ".doc" not in extractor # degrades to the pre-anydoc map
|
||||
|
||||
def test_rtf_converts_end_to_end(self, tmp_path):
|
||||
pytest.importorskip("anydoc")
|
||||
from application.parser.file.bulk import get_default_file_extractor
|
||||
|
||||
path = tmp_path / "note.rtf"
|
||||
path.write_text(r"{\rtf1\ansi Hello {\b bold} world.\par Second paragraph.}")
|
||||
parser = get_default_file_extractor()[".rtf"]
|
||||
parser.init_parser()
|
||||
|
||||
out = parser.parse_file(path)
|
||||
|
||||
assert "Hello **bold** world." in out
|
||||
assert "Second paragraph." in out
|
||||
|
||||
def test_odt_converts_end_to_end(self, tmp_path):
|
||||
pytest.importorskip("anydoc")
|
||||
import zipfile
|
||||
|
||||
from application.parser.file.bulk import get_default_file_extractor
|
||||
|
||||
path = tmp_path / "doc.odt"
|
||||
with zipfile.ZipFile(path, "w") as z:
|
||||
z.writestr(
|
||||
"mimetype",
|
||||
"application/vnd.oasis.opendocument.text",
|
||||
compress_type=zipfile.ZIP_STORED,
|
||||
)
|
||||
z.writestr(
|
||||
"META-INF/manifest.xml",
|
||||
'<?xml version="1.0"?><manifest:manifest xmlns:manifest="urn:oasis:names:tc:opendocument:xmlns:manifest:1.0">'
|
||||
'<manifest:file-entry manifest:full-path="/" manifest:media-type="application/vnd.oasis.opendocument.text"/>'
|
||||
'<manifest:file-entry manifest:full-path="content.xml" manifest:media-type="text/xml"/></manifest:manifest>',
|
||||
)
|
||||
z.writestr(
|
||||
"content.xml",
|
||||
'<?xml version="1.0"?><office:document-content xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0" '
|
||||
'xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0"><office:body><office:text>'
|
||||
'<text:h text:outline-level="1">Title</text:h><text:p>Body text here.</text:p>'
|
||||
"</office:text></office:body></office:document-content>",
|
||||
)
|
||||
parser = get_default_file_extractor()[".odt"]
|
||||
|
||||
out = parser.parse_file(path)
|
||||
|
||||
assert "# Title" in out
|
||||
assert "Body text here." in out
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Comprehensive tests for application/parser/file/docling_parser.py
|
||||
|
||||
Covers: DoclingParser (init, _init_parser, _get_ocr_options, _export_content,
|
||||
Covers: DoclingParser (init, _init_parser, OCR engine selection, _export_content,
|
||||
parse_file), subclass initialization, error handling.
|
||||
"""
|
||||
|
||||
@@ -27,8 +27,8 @@ class TestDoclingParserInit:
|
||||
assert parser.ocr_enabled is True
|
||||
assert parser.table_structure is True
|
||||
assert parser.export_format == "markdown"
|
||||
assert parser.use_rapidocr is True
|
||||
assert parser.ocr_languages == ["english"]
|
||||
assert parser.ocr_engine is None # None -> settings.OCR_ENGINE at build
|
||||
assert parser.ocr_languages is None # None -> the engine's own default
|
||||
assert parser.force_full_page_ocr is False
|
||||
assert parser._converter is None
|
||||
|
||||
@@ -39,14 +39,14 @@ class TestDoclingParserInit:
|
||||
ocr_enabled=False,
|
||||
table_structure=False,
|
||||
export_format="text",
|
||||
use_rapidocr=False,
|
||||
ocr_engine="rapidocr",
|
||||
ocr_languages=["german"],
|
||||
force_full_page_ocr=True,
|
||||
)
|
||||
assert parser.ocr_enabled is False
|
||||
assert parser.table_structure is False
|
||||
assert parser.export_format == "text"
|
||||
assert parser.use_rapidocr is False
|
||||
assert parser.ocr_engine == "rapidocr"
|
||||
assert parser.ocr_languages == ["german"]
|
||||
assert parser.force_full_page_ocr is True
|
||||
|
||||
@@ -90,44 +90,169 @@ class TestDoclingParserInitParser:
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestGetOCROptions:
|
||||
class TestOcrEngineSelection:
|
||||
"""``OCR_ENGINE`` resolution and per-engine option building."""
|
||||
|
||||
def test_returns_none_when_rapidocr_disabled(self):
|
||||
@pytest.fixture
|
||||
def settings(self):
|
||||
from application.core.settings import settings
|
||||
|
||||
return settings
|
||||
|
||||
def test_default_setting_is_tesseract(self, settings):
|
||||
assert settings.OCR_ENGINE == "tesseract"
|
||||
assert settings.OCR_LANGS == "eng"
|
||||
|
||||
def test_none_reads_setting(self, settings, monkeypatch):
|
||||
from application.parser.file.docling_parser import _resolve_ocr_engine
|
||||
|
||||
monkeypatch.setattr(settings, "OCR_ENGINE", "auto")
|
||||
assert _resolve_ocr_engine(None) == "auto"
|
||||
|
||||
def test_unknown_engine_degrades_to_auto(self):
|
||||
from application.parser.file.docling_parser import _resolve_ocr_engine
|
||||
|
||||
assert _resolve_ocr_engine("easyocr") == "auto"
|
||||
|
||||
def test_tesseract_without_binary_degrades_to_auto(self, monkeypatch):
|
||||
import application.parser.file.docling_parser as dp
|
||||
|
||||
monkeypatch.setattr(dp.shutil, "which", lambda name: None)
|
||||
assert dp._resolve_ocr_engine("tesseract") == "auto"
|
||||
|
||||
def test_tesseract_with_binary_selected(self, monkeypatch):
|
||||
import application.parser.file.docling_parser as dp
|
||||
|
||||
monkeypatch.setattr(dp.shutil, "which", lambda name: "/usr/bin/tesseract")
|
||||
assert dp._resolve_ocr_engine("tesseract") == "tesseract"
|
||||
|
||||
def test_ocrmac_off_darwin_degrades_to_auto(self, monkeypatch):
|
||||
import application.parser.file.docling_parser as dp
|
||||
|
||||
monkeypatch.setattr(dp.sys, "platform", "linux")
|
||||
assert dp._resolve_ocr_engine("ocrmac") == "auto"
|
||||
|
||||
def test_rapidocr_missing_degrades_to_auto(self, monkeypatch):
|
||||
import sys
|
||||
|
||||
import application.parser.file.docling_parser as dp
|
||||
|
||||
monkeypatch.setitem(sys.modules, "rapidocr", None)
|
||||
assert dp._resolve_ocr_engine("rapidocr") == "auto"
|
||||
|
||||
def test_deepseek_passes_through(self):
|
||||
from application.parser.file.docling_parser import _resolve_ocr_engine
|
||||
|
||||
assert _resolve_ocr_engine("deepseek") == "deepseek"
|
||||
|
||||
def test_build_auto_returns_none(self):
|
||||
from application.parser.file.docling_parser import _build_ocr_options
|
||||
|
||||
assert _build_ocr_options("auto", None, True) is None
|
||||
|
||||
def test_build_tesseract_reads_ocr_langs(self, settings, monkeypatch):
|
||||
pytest.importorskip("docling")
|
||||
from application.parser.file.docling_parser import _build_ocr_options
|
||||
|
||||
monkeypatch.setattr(settings, "OCR_LANGS", "eng+chi_sim")
|
||||
options = _build_ocr_options("tesseract", None, True)
|
||||
|
||||
assert type(options).__name__ == "TesseractCliOcrOptions"
|
||||
assert options.lang == ["eng", "chi_sim"]
|
||||
assert options.force_full_page_ocr is True
|
||||
|
||||
def test_build_tesseract_explicit_languages_win(self):
|
||||
pytest.importorskip("docling")
|
||||
from application.parser.file.docling_parser import _build_ocr_options
|
||||
|
||||
options = _build_ocr_options("tesseract", ["deu"], False)
|
||||
assert options.lang == ["deu"]
|
||||
|
||||
def test_build_rapidocr(self):
|
||||
pytest.importorskip("docling")
|
||||
from application.parser.file.docling_parser import _build_ocr_options
|
||||
|
||||
options = _build_ocr_options("rapidocr", None, False)
|
||||
assert type(options).__name__ == "RapidOcrOptions"
|
||||
assert options.lang == ["english"]
|
||||
|
||||
def test_build_import_failure_returns_none(self, monkeypatch):
|
||||
import sys
|
||||
|
||||
from application.parser.file.docling_parser import _build_ocr_options
|
||||
|
||||
monkeypatch.setitem(sys.modules, "docling.datamodel.pipeline_options", None)
|
||||
assert _build_ocr_options("tesseract", ["eng"], False) is None
|
||||
assert _build_ocr_options("ocrmac", None, False) is None
|
||||
|
||||
def test_build_generic_failure_returns_none(self, monkeypatch):
|
||||
import sys
|
||||
import types
|
||||
|
||||
from application.parser.file.docling_parser import _build_ocr_options
|
||||
|
||||
fake = types.ModuleType("docling.datamodel.pipeline_options")
|
||||
|
||||
def _boom(**kwargs):
|
||||
raise RuntimeError("bad options")
|
||||
|
||||
fake.RapidOcrOptions = _boom
|
||||
monkeypatch.setitem(sys.modules, "docling.datamodel.pipeline_options", fake)
|
||||
assert _build_ocr_options("rapidocr", None, False) is None
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestDeepseekVlmConverter:
|
||||
"""OCR_ENGINE=deepseek swaps the whole pipeline for docling's VLM route."""
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _requires_docling(self):
|
||||
pytest.importorskip("docling")
|
||||
|
||||
def test_deepseek_builds_vlm_converter(self, monkeypatch):
|
||||
from application.core.settings import settings
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
parser = DoclingParser(use_rapidocr=False)
|
||||
assert parser._get_ocr_options() is None
|
||||
monkeypatch.setattr(settings, "OCR_DEEPSEEK_URL", "http://gpu-host:8000/v1/chat/completions")
|
||||
monkeypatch.setattr(settings, "OCR_DEEPSEEK_MODEL", "deepseek-ocr-x")
|
||||
|
||||
def test_returns_options_when_available(self):
|
||||
built = {}
|
||||
|
||||
def _capture_converter(format_options):
|
||||
built["format_options"] = format_options
|
||||
return MagicMock()
|
||||
|
||||
monkeypatch.setattr(
|
||||
"docling.document_converter.DocumentConverter", _capture_converter
|
||||
)
|
||||
parser = DoclingParser(ocr_enabled=True, ocr_engine="deepseek")
|
||||
parser._create_converter()
|
||||
|
||||
from docling.datamodel.base_models import InputFormat
|
||||
from docling.pipeline.vlm_pipeline import VlmPipeline
|
||||
|
||||
pdf_option = built["format_options"][InputFormat.PDF]
|
||||
image_option = built["format_options"][InputFormat.IMAGE]
|
||||
assert pdf_option.pipeline_cls is VlmPipeline
|
||||
assert image_option.pipeline_cls is VlmPipeline
|
||||
vlm = pdf_option.pipeline_options.vlm_options
|
||||
assert vlm.url == "http://gpu-host:8000/v1/chat/completions"
|
||||
assert vlm.params["model"] == "deepseek-ocr-x"
|
||||
assert pdf_option.pipeline_options.enable_remote_services is True
|
||||
|
||||
def test_deepseek_ignored_when_ocr_disabled(self, monkeypatch):
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
parser = DoclingParser(use_rapidocr=True, ocr_languages=["english"])
|
||||
vlm_called = []
|
||||
monkeypatch.setattr(
|
||||
DoclingParser,
|
||||
"_create_vlm_converter",
|
||||
lambda self: vlm_called.append(True),
|
||||
)
|
||||
monkeypatch.setattr("docling.document_converter.DocumentConverter", MagicMock())
|
||||
DoclingParser(ocr_enabled=False, ocr_engine="deepseek")._create_converter()
|
||||
|
||||
mock_options = MagicMock()
|
||||
with patch(
|
||||
"application.parser.file.docling_parser.DoclingParser._get_ocr_options",
|
||||
return_value=mock_options,
|
||||
):
|
||||
result = parser._get_ocr_options()
|
||||
assert result is mock_options
|
||||
|
||||
def test_returns_none_on_import_error(self):
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
parser = DoclingParser(use_rapidocr=True)
|
||||
|
||||
# Simulate the ImportError path
|
||||
original = parser._get_ocr_options
|
||||
|
||||
def patched_get_ocr():
|
||||
try:
|
||||
raise ImportError("No RapidOcrOptions")
|
||||
except ImportError:
|
||||
return None
|
||||
|
||||
parser._get_ocr_options = patched_get_ocr
|
||||
assert parser._get_ocr_options() is None
|
||||
parser._get_ocr_options = original
|
||||
assert vlm_called == []
|
||||
|
||||
|
||||
# =====================================================================
|
||||
@@ -431,31 +556,6 @@ class TestDoclingSubclasses:
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestDoclingParserGaps:
|
||||
def test_get_ocr_options_import_error_returns_none(self):
|
||||
"""Cover lines 148-150: ImportError returns None."""
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
parser = DoclingParser(ocr_enabled=True, use_rapidocr=True)
|
||||
with patch.dict("sys.modules", {"docling.datamodel.pipeline_options": None}):
|
||||
# Force re-import to trigger ImportError
|
||||
with patch(
|
||||
"builtins.__import__", side_effect=ImportError("no module")
|
||||
):
|
||||
result = parser._get_ocr_options()
|
||||
assert result is None
|
||||
|
||||
def test_get_ocr_options_generic_error_returns_none(self):
|
||||
"""Cover lines 151-153: generic Exception returns None."""
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
parser = DoclingParser(ocr_enabled=True, use_rapidocr=True)
|
||||
with patch(
|
||||
"builtins.__import__",
|
||||
side_effect=RuntimeError("unexpected"),
|
||||
):
|
||||
result = parser._get_ocr_options()
|
||||
assert result is None
|
||||
|
||||
def test_csv_parser_init(self):
|
||||
"""Cover line 289: DoclingCSVParser.__init__ calls super."""
|
||||
from application.parser.file.docling_parser import DoclingCSVParser
|
||||
@@ -1442,11 +1542,11 @@ class TestForceFullPageOCRWiring:
|
||||
DoclingParser(**kwargs)._create_converter()
|
||||
return built[0]
|
||||
|
||||
def test_forced_without_rapidocr(self, monkeypatch):
|
||||
def test_forced_with_default_auto_options(self, monkeypatch):
|
||||
options = self._pipeline_options(
|
||||
monkeypatch,
|
||||
ocr_enabled=True,
|
||||
use_rapidocr=False,
|
||||
ocr_engine="auto",
|
||||
force_full_page_ocr=True,
|
||||
)
|
||||
assert options.ocr_options.force_full_page_ocr is True
|
||||
@@ -1455,7 +1555,7 @@ class TestForceFullPageOCRWiring:
|
||||
options = self._pipeline_options(
|
||||
monkeypatch,
|
||||
ocr_enabled=True,
|
||||
use_rapidocr=False,
|
||||
ocr_engine="auto",
|
||||
force_full_page_ocr=False,
|
||||
)
|
||||
assert options.ocr_options.force_full_page_ocr is False
|
||||
@@ -0,0 +1,148 @@
|
||||
"""Tests for the PDF trust checks behind ``PDF_TRUST_CHECK``.
|
||||
|
||||
The checks target the two classes where anydoc drops text *silently*:
|
||||
composite (Type0) fonts without a ToUnicode map, and CJK-declaring PDFs
|
||||
whose extracted text carries almost no CJK. The scanner is byte-level and
|
||||
regex-based, so the synthetic fixtures below are minimal PDF fragments, not
|
||||
well-formed files; the one real fixture is the corpus document that loses
|
||||
its whole Chinese column in anydoc with no error.
|
||||
"""
|
||||
import zlib
|
||||
from pathlib import Path
|
||||
|
||||
from application.parser.file.pdf_trust import (
|
||||
check_pdf_fonts,
|
||||
verify_extraction,
|
||||
verify_pdf_file,
|
||||
)
|
||||
|
||||
FIXTURES = Path(__file__).parent / "fixtures"
|
||||
|
||||
|
||||
def _obj(body: bytes) -> bytes:
|
||||
return b"1 0 obj" + body + b"endobj\n"
|
||||
|
||||
|
||||
def _stream(payload: bytes) -> bytes:
|
||||
return b"2 0 obj<</Filter/FlateDecode>>stream\n" + zlib.compress(payload) + b"\nendstream endobj\n"
|
||||
|
||||
|
||||
TYPE0_WITH_TOUNICODE = _obj(
|
||||
b"<</Type/Font/Subtype/Type0/BaseFont/AAAAAA+Simple/ToUnicode 9 0 R>>"
|
||||
)
|
||||
TYPE0_BARE = _obj(b"<</Type/Font/Subtype/Type0/BaseFont/AAAAAA+Bare>>")
|
||||
SIMPLE_FONT = _obj(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>")
|
||||
GB1_ORDERING = _obj(b"<</Registry(Adobe)/Ordering(GB1)/Supplement 5>>")
|
||||
|
||||
|
||||
class TestCheckPdfFonts:
|
||||
def test_type0_with_tounicode_not_flagged(self):
|
||||
result = check_pdf_fonts(TYPE0_WITH_TOUNICODE)
|
||||
assert result["type0"] == 1
|
||||
assert result["type0_no_tounicode"] == 0
|
||||
assert not result["flagged"]
|
||||
|
||||
def test_type0_without_tounicode_flagged(self):
|
||||
result = check_pdf_fonts(TYPE0_BARE)
|
||||
assert result["type0_no_tounicode"] == 1
|
||||
assert result["flagged"]
|
||||
|
||||
def test_simple_font_not_flagged(self):
|
||||
result = check_pdf_fonts(SIMPLE_FONT)
|
||||
assert result["type0"] == 0
|
||||
assert result["has_fonts"]
|
||||
assert not result["flagged"]
|
||||
|
||||
def test_no_fonts_flagged(self):
|
||||
result = check_pdf_fonts(_obj(b"<</Type/Page>>"))
|
||||
assert not result["has_fonts"]
|
||||
assert result["flagged"]
|
||||
|
||||
def test_type0_inside_object_stream_is_seen(self):
|
||||
"""Object streams hold dicts with no obj markers; the scan must decompress them."""
|
||||
data = _stream(b"<</Type/Font/Subtype/Type0/BaseFont/BBBBBB+Packed>>")
|
||||
result = check_pdf_fonts(data)
|
||||
assert result["type0"] == 1
|
||||
assert result["type0_no_tounicode"] == 1
|
||||
assert result["flagged"]
|
||||
|
||||
def test_font_inside_object_stream_counts_as_has_fonts(self):
|
||||
"""A fully-compressed PDF with only simple fonts must not false-positive."""
|
||||
data = _stream(b"<</Type/Font/Subtype/TrueType/BaseFont/Arial>>")
|
||||
result = check_pdf_fonts(data)
|
||||
assert result["has_fonts"]
|
||||
assert not result["flagged"]
|
||||
|
||||
def test_cjk_ordering_detected_in_stream(self):
|
||||
data = _stream(b"<</Registry(Adobe)/Ordering(CNS1)/Supplement 4>>") + SIMPLE_FONT
|
||||
assert check_pdf_fonts(data)["expects_cjk"]
|
||||
|
||||
def test_cjk_font_names_alone_do_not_set_expectation(self):
|
||||
"""Regression for the Berkshire false positive: Word-exported Latin PDFs
|
||||
embed MS-Gothic subsets for stray full-width characters, and substring
|
||||
matching also hits FranklinGothic — neither means CJK *content*, and a
|
||||
false expectation reroutes a 150-page English report to the heavy
|
||||
engine. Only CIDSystemInfo /Ordering may set the expectation."""
|
||||
data = _obj(
|
||||
b"<</Type/Font/Subtype/TrueType/BaseFont/AKBHWD+MS-Gothic>>"
|
||||
) + _obj(b"<</Type/Font/Subtype/TrueType/BaseFont/KMBGKH+FranklinGothic-Roman>>")
|
||||
result = check_pdf_fonts(data)
|
||||
assert not result["expects_cjk"]
|
||||
assert not result["flagged"]
|
||||
|
||||
|
||||
class TestVerifyExtraction:
|
||||
def test_clean_pdf_and_output(self):
|
||||
assert verify_extraction(SIMPLE_FONT, "Plain English text.") == []
|
||||
|
||||
def test_bare_type0_reported(self):
|
||||
problems = verify_extraction(TYPE0_BARE + SIMPLE_FONT, "some text")
|
||||
assert len(problems) == 1
|
||||
assert "ToUnicode" in problems[0]
|
||||
|
||||
def test_cjk_expected_but_missing(self):
|
||||
problems = verify_extraction(GB1_ORDERING + SIMPLE_FONT, "English only output")
|
||||
assert len(problems) == 1
|
||||
assert "CJK" in problems[0]
|
||||
|
||||
def test_cjk_expected_and_present(self):
|
||||
text = "标题:本协议由双方共同签署生效" # 14 CJK chars
|
||||
assert verify_extraction(GB1_ORDERING + SIMPLE_FONT, text) == []
|
||||
|
||||
def test_both_problems_reported(self):
|
||||
problems = verify_extraction(TYPE0_BARE + GB1_ORDERING, "english")
|
||||
assert len(problems) == 2
|
||||
|
||||
|
||||
class TestRealFixture:
|
||||
"""The corpus PDF where anydoc silently drops the entire Chinese column."""
|
||||
|
||||
def test_cid_font_nda_is_flagged(self):
|
||||
data = (FIXTURES / "nda_en_zh_cid_font.pdf").read_bytes()
|
||||
result = check_pdf_fonts(data)
|
||||
assert result["type0_no_tounicode"] >= 1
|
||||
assert result["expects_cjk"]
|
||||
assert result["flagged"]
|
||||
|
||||
def test_cjkless_extraction_reports_both(self):
|
||||
data = (FIXTURES / "nda_en_zh_cid_font.pdf").read_bytes()
|
||||
problems = verify_extraction(data, "MUTUAL NON-DISCLOSURE AGREEMENT ...")
|
||||
assert any("ToUnicode" in p for p in problems)
|
||||
assert any("CJK" in p for p in problems)
|
||||
|
||||
|
||||
def test_flate_bomb_streams_are_capped():
|
||||
"""Flate reaches ~1000:1, so per-stream inflation must be capped — an
|
||||
uncapped decompress of a crafted PDF would OOM the ingest worker."""
|
||||
from application.parser.file.pdf_trust import _decompressed_streams, _STREAM_INFLATE_CAP
|
||||
|
||||
bomb = _stream(b"\0" * (_STREAM_INFLATE_CAP * 4))
|
||||
chunks = list(_decompressed_streams(bomb))
|
||||
assert chunks
|
||||
assert all(len(chunk) <= _STREAM_INFLATE_CAP for chunk in chunks)
|
||||
# The capped scan still completes and reports sanely alongside real objects.
|
||||
assert not check_pdf_fonts(SIMPLE_FONT + bomb)["flagged"]
|
||||
|
||||
|
||||
def test_verify_pdf_file_unreadable_trusts_output(tmp_path):
|
||||
assert verify_pdf_file(tmp_path / "missing.pdf", "text") == []
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Tests for the dot-leader/whitespace table reconstruction (``ANYDOC_TABLEIZE``)."""
|
||||
from application.parser.file.tableize import tableize
|
||||
|
||||
DOT_LEADER = """Revenues
|
||||
Insurance premiums ............ 83,431 77,731
|
||||
Sales and service revenues ........ 24,660 23,406
|
||||
Freight rail transportation ......... 22,341 23,852
|
||||
Total revenues 130,432 124,989
|
||||
"""
|
||||
|
||||
|
||||
def test_dot_leader_run_becomes_table():
|
||||
out = tableize(DOT_LEADER)
|
||||
assert "| Insurance premiums | 83,431 | 77,731 |" in out
|
||||
assert "| Total revenues | 130,432 | 124,989 |" in out
|
||||
assert "| --- |" in out
|
||||
assert "Revenues" in out # heading line untouched
|
||||
|
||||
|
||||
def test_whitespace_only_rows_convert_too():
|
||||
md = "Alpha 1 2\nBeta 3 4\nGamma 5 6\n"
|
||||
out = tableize(md)
|
||||
assert "| Alpha | 1 | 2 |" in out
|
||||
|
||||
|
||||
def test_currency_symbol_merges_with_number():
|
||||
md = "Cash $ 1,234 900\nDebt $ 2,000 1,500\nEquity $ 900 800\n"
|
||||
out = tableize(md)
|
||||
assert "| Cash | $1,234 | 900 |" in out
|
||||
|
||||
|
||||
def test_parenthesised_negatives_and_dash():
|
||||
md = "Losses (30) —\nGains (12) —\nNet (42) —\n"
|
||||
out = tableize(md)
|
||||
assert "| Losses | (30) | — |" in out
|
||||
|
||||
|
||||
def test_short_run_is_left_alone():
|
||||
md = "Alpha 1 2\nBeta 3 4\n"
|
||||
assert "|" not in tableize(md)
|
||||
|
||||
|
||||
def test_mixed_widths_are_left_alone():
|
||||
md = "Alpha 1 2\nBeta 3\nGamma 5 6\n"
|
||||
assert "|" not in tableize(md)
|
||||
|
||||
|
||||
def test_prose_with_numbers_is_left_alone():
|
||||
md = "The division shipped 47 releases in 2023.\nIt hired 12 engineers this year alone.\nRevenue grew by a factor of 3 since 2019.\n"
|
||||
assert "|" not in tableize(md)
|
||||
|
||||
|
||||
def test_min_rows_boundary():
|
||||
two = "A 1 2\nB 3 4\n"
|
||||
three = two + "C 5 6\n"
|
||||
assert "|" not in tableize(two, min_rows=3)
|
||||
assert "|" in tableize(three, min_rows=3)
|
||||
|
||||
|
||||
def test_pipe_in_label_is_escaped():
|
||||
md = "Assets | current 1,234 900\nDebt 2,000 1,500\nEquity 900 800\n"
|
||||
out = tableize(md)
|
||||
assert "| Assets \\| current | 1,234 | 900 |" in out
|
||||
|
||||
|
||||
def test_dollar_prefixed_word_is_not_a_value():
|
||||
md = "Alpha $TBD 2\nBeta $TBD 4\nGamma $TBD 6\n"
|
||||
assert "|" not in tableize(md)
|
||||
|
||||
|
||||
def test_non_table_text_passes_through_verbatim():
|
||||
"""Only the trailing newline may differ (splitlines/join round-trip)."""
|
||||
md = "# Heading\n\nA paragraph with no numbers.\n\n- a list item\n"
|
||||
assert tableize(md) == md.rstrip("\n")
|
||||
Reference in new issue
Block a user