ocr update

This commit is contained in:
Pavel committed 2026-08-27 23:46:12 +04:00
1 parent d7a7d4d084
commit d4e92be0ab
27 files changed
+1656 -134

No files matched your search

+10
View File
@@ -66,6 +66,16 @@ RUN apt-get update && \
ln -s /usr/bin/python3.12 /usr/bin/python && \
rm -rf /var/lib/apt/lists/*
# The recommended OCR engine (OCR_ENGINE=tesseract) shells out to the system
# tesseract binary, so ship it alongside the optional docling engine. Extra
# language packs are a deployment concern (apt: tesseract-ocr-<lang>).
ARG INSTALL_DOCLING=false
RUN if [ "$INSTALL_DOCLING" = "true" ]; then \
apt-get update && \
apt-get install -y --no-install-recommends tesseract-ocr tesseract-ocr-eng && \
rm -rf /var/lib/apt/lists/*; \
fi
# Set working directory
WORKDIR /app
+15 -1
View File
@@ -272,7 +272,21 @@ class UploadFile(Resource):
# Office/e-book containers are ZIP-based formats but must
# be parsed as documents, not expanded as user archives.
is_office_format = safe_file.lower().endswith(
(".docx", ".xlsx", ".pptx", ".odt", ".ods", ".odp", ".epub")
(
".docx",
".docm",
".xlsx",
".xlsm",
".xlsb",
".pptx",
".pptm",
".ppsx",
".ppsm",
".odt",
".ods",
".odp",
".epub",
)
)
if zipfile.is_zipfile(temp_file_path) and not is_office_format:
extract_dir = os.path.join(upload_dir, "extracted")
+34
View File
@@ -129,6 +129,40 @@ class Settings(BaseSettings):
# and the upload cap is 100 MB, so the gate is what keeps one upload from
# taking the ingest worker down. 0 disables it.
MARKUP_MAX_BYTES: int = 8_000_000
# Trust-check anydoc's PDF output (application/parser/file/pdf_trust.py):
# flag composite (Type0) fonts without a ToUnicode map, and CJK-declaring
# PDFs whose extracted text has almost no CJK — the two classes where
# anydoc drops text silently. A flagged file re-parses on the docling
# fallback when docling is installed; otherwise the anydoc output is kept
# and the document gets extra_info["parse_warnings"]. ~30 ms per scanned MB.
PDF_TRUST_CHECK: bool = True
# Rewrite dot-leader / whitespace-aligned table runs in anydoc's PDF
# markdown into GFM tables (application/parser/file/tableize.py). Off by
# default: it rewrites content on a heuristic (>=3 uniform label+numbers
# lines) validated only on a small corpus so far.
ANYDOC_TABLEIZE: bool = False
# OCR engine used by the docling parsers when OCR is on
# (DOCLING_OCR_ENABLED / DOCLING_OCR_ATTACHMENTS_ENABLED). Benched 2026-08
# on EN/ZH/table/degraded scans (docs/Guides/ocr has the menu):
# tesseract — recommended: best classic-engine accuracy (perfect EN word
# recall, 0.000 bilingual CER, 100% table cells), ~35 MB, CPU-only.
# Needs the system binary + language packs (the Docker image installs
# them when built with INSTALL_DOCLING=true).
# auto — docling's pick: ocrmac on macOS (excellent), rapidocr on Linux
# (silently shreds some long text lines — avoid as a server default).
# ocrmac | rapidocr — force one of those.
# deepseek — DeepSeek-OCR through docling's VLM pipeline against an
# Ollama/vLLM endpoint (OCR_DEEPSEEK_*). Best table/CJK quality; the
# worker stays light (~370 MB, no layout models) but each page costs
# seconds on the model server.
# An engine that is not installed degrades to "auto" with a warning
# instead of failing the parse.
OCR_ENGINE: str = "tesseract"
# Tesseract language packs, "+"-separated (e.g. "eng+chi_sim+deu"). Other
# engines keep their own defaults — their language codes differ.
OCR_LANGS: str = "eng"
OCR_DEEPSEEK_URL: str = "http://localhost:11434/v1/chat/completions"
OCR_DEEPSEEK_MODEL: str = "deepseek-ocr:3b"
# Chars-per-page floor below which an OCR'd PDF/image parse is treated as a docling
# pipeline dropout (long-running workers were observed returning zero characters for
# every scanned page after a long scanned PDF, with no error) rather than as content.
+166 -3
View File
@@ -18,6 +18,7 @@ import logging
from pathlib import Path
from typing import Dict, List, Optional, Tuple, Union
from application.core.settings import settings
from application.parser.file.base_parser import (
BaseParser,
DocumentParseError,
@@ -27,6 +28,11 @@ from application.parser.file.base_parser import (
logger = logging.getLogger(__name__)
# A fallback parse (scanned-PDF delegation or trust-check re-parse) yielding
# fewer stripped characters than this is an empty document in the making,
# not content.
_MIN_SCAN_FALLBACK_CHARS = 50
# Suffixes anydoc 0.2.3 converts, verified with ``anydoc.format_from_extension``.
# ``.epub`` is deliberately absent: it stays on ``EpubParser``.
ANYDOC_SUFFIXES: Tuple[str, ...] = (
@@ -56,11 +62,39 @@ ANYDOC_SUFFIXES: Tuple[str, ...] = (
)
# The five formats every engine has its own parser for (docling / legacy);
# under the anydoc engine they get that parser as the fallback. Everything
# else in ``ANYDOC_SUFFIXES`` is anydoc-only ("gained") and is mapped to a
# fallback-less ``AnydocParser`` in every parser map.
_CORE_SUFFIXES = frozenset({".pdf", ".docx", ".pptx", ".xlsx", ".csv"})
ANYDOC_GAINED_SUFFIXES: Tuple[str, ...] = tuple(
suffix for suffix in ANYDOC_SUFFIXES if suffix not in _CORE_SUFFIXES
)
def anydoc_available() -> bool:
"""Whether the ``anydoc`` extension can be imported."""
return module_available("anydoc")
def _result_chars(result: Union[str, List[str]]) -> int:
"""Stripped character count of a parser result (str or list of rows)."""
if isinstance(result, list):
return sum(len(str(part).strip()) for part in result)
return len(str(result).strip())
def _is_docling_backed(parser: Optional[BaseParser]) -> bool:
"""True when ``parser`` is a docling parser (worth a trust-check re-parse)."""
if parser is None:
return False
try:
from application.parser.file.docling_parser import DoclingParser
except ImportError:
return False
return isinstance(parser, DoclingParser)
class AnydocParser(BaseParser):
"""Markdown conversion via ``anydoc.to_markdown`` with a typed fallback.
@@ -82,6 +116,10 @@ class AnydocParser(BaseParser):
super().__init__(parser_config)
self.fallback_parser = fallback_parser
self.last_engine: Optional[str] = None
# (path, problems) from the most recent parse whose output the trust
# check flagged but kept — surfaced via ``get_file_metadata`` as
# ``parse_warnings`` so the document carries its own caveat.
self._last_warnings: Optional[Tuple[Path, List[str]]] = None
def _init_parser(self) -> Dict:
# Import for real rather than trusting ``find_spec``: a wheel whose
@@ -117,6 +155,7 @@ class AnydocParser(BaseParser):
import anydoc
path = Path(file)
self._last_warnings = None
try:
content = anydoc.to_markdown(str(path))
except anydoc.ResourceLimitError as exc:
@@ -132,7 +171,10 @@ class AnydocParser(BaseParser):
# Typed, per-document failures another engine may still handle:
# UnsupportedError (scanned PDF / unknown format), MalformedError,
# EncryptedError, MissingPartError.
return self._delegate(path, errors, f"{type(exc).__name__}: {exc}")
ocr_needed = isinstance(exc, anydoc.UnsupportedError) and "OCR" in str(exc)
return self._delegate(
path, errors, f"{type(exc).__name__}: {exc}", ocr_needed=ocr_needed
)
except OSError as exc:
raise DocumentParseError(
f"Failed to parse {path.name}: the file could not be read."
@@ -147,10 +189,117 @@ class AnydocParser(BaseParser):
return self._delegate(path, errors, "anydoc produced no text")
self.last_engine = "anydoc"
if path.suffix.lower() == ".pdf":
content = self._finish_pdf(path, content, errors)
return content
def _delegate(self, path: Path, errors: str, reason: str) -> Union[str, List[str]]:
"""Hand ``path`` to the fallback parser, or fail loudly without one."""
def _finish_pdf(self, path: Path, content: str, errors: str) -> Union[str, List[str]]:
"""Post-process a successful anydoc PDF conversion.
Two PDF-only steps:
1. Trust check (``PDF_TRUST_CHECK``): the two classes where anydoc
drops text *silently* — Type0 fonts without ToUnicode, and
CJK-declaring PDFs with CJK-less output. A flagged file re-parses
on the docling fallback when one is wired; otherwise the anydoc
output is kept and the problems ride along as ``parse_warnings``
metadata.
2. ``ANYDOC_TABLEIZE`` (off by default): rewrite dot-leader /
whitespace table runs as GFM tables. Never applied to a docling
re-parse — docling emits real tables already.
"""
problems = self._trust_problems(path, content)
if problems:
rerouted = self._reroute_flagged(path, errors, problems)
if rerouted is not None:
return rerouted
self._last_warnings = (path, problems)
logger.warning(
f"anydoc output for {path.name} failed the PDF trust check "
f"({'; '.join(problems)}); keeping it with parse_warnings"
)
if settings.ANYDOC_TABLEIZE:
from application.parser.file.tableize import tableize
content = tableize(content)
return content
def _trust_problems(self, path: Path, content: str) -> List[str]:
"""Trust-check findings for ``content``; [] when disabled, clean, or the check errors."""
if not settings.PDF_TRUST_CHECK:
return []
try:
from application.parser.file.pdf_trust import verify_pdf_file
return verify_pdf_file(path, content)
except Exception:
# The check exists to catch silent loss; it must never turn a
# successful parse into a failure.
logger.warning(
f"PDF trust check errored on {path.name}; trusting the output",
exc_info=True,
)
return []
def _reroute_flagged(
self, path: Path, errors: str, problems: List[str]
) -> Optional[Union[str, List[str]]]:
"""Re-parse a trust-flagged PDF with a docling-backed fallback.
Returns the fallback's output, or None when there is no docling
fallback, it fails, or it comes back near-empty — the caller then
keeps the anydoc output.
Only docling is worth the re-parse: it ships Adobe's predefined
CMaps, which is exactly what the flagged font class needs; the
legacy pypdf parser does not.
"""
fallback = self.fallback_parser
if not _is_docling_backed(fallback):
return None
logger.warning(
f"anydoc output for {path.name} failed the PDF trust check "
f"({'; '.join(problems)}); re-parsing with {type(fallback).__name__}"
)
try:
result = delegate_parse(fallback, path, errors)
except DocumentParseError:
logger.warning(
f"docling re-parse of trust-flagged {path.name} failed; "
"keeping the anydoc output",
exc_info=True,
)
return None
if _result_chars(result) < _MIN_SCAN_FALLBACK_CHARS:
# A docling pipeline dropout ('' / '<!-- image -->') returns
# without raising; adopting it would swap anydoc's real text for
# an empty document.
logger.warning(
f"docling re-parse of trust-flagged {path.name} returned almost "
"no text; keeping the anydoc output"
)
return None
self.last_engine = getattr(fallback, "last_engine", None) or type(fallback).__name__
return result
def get_file_metadata(self, file: Path) -> Dict:
"""Surface trust-check findings for the file just parsed, if any."""
last = self._last_warnings
if last is not None and last[0] == Path(file):
return {"parse_warnings": list(last[1])}
return {}
def _delegate(
self, path: Path, errors: str, reason: str, ocr_needed: bool = False
) -> Union[str, List[str]]:
"""Hand ``path`` to the fallback parser, or fail loudly without one.
``ocr_needed`` marks anydoc's scanned-PDF refusal ("OCR is required").
The fallback still runs first — docling extracts text layers anydoc
refuses (seen on degenerate CJK text layers) even with OCR off — but
when it comes back near-empty the parse fails loudly, telling the
user how to get OCR, instead of silently storing an empty document
for a scan.
"""
fallback = self.fallback_parser
if fallback is None:
self.last_engine = None
@@ -160,5 +309,19 @@ class AnydocParser(BaseParser):
f"falling back to {type(fallback).__name__}"
)
result = delegate_parse(fallback, path, errors)
if ocr_needed and _result_chars(result) < _MIN_SCAN_FALLBACK_CHARS:
self.last_engine = None
if getattr(fallback, "ocr_enabled", False):
hint = " even with OCR enabled."
else:
hint = (
". Enable OCR to ingest scans: set DOCLING_OCR_ENABLED=true "
"(and DOCLING_OCR_ATTACHMENTS_ENABLED for attachments) with "
"the docling extra installed."
)
raise DocumentParseError(
f"{path.name} appears to be a scanned PDF (no text layer), and "
f"{type(fallback).__name__} extracted almost nothing{hint}"
)
self.last_engine = getattr(fallback, "last_engine", None) or type(fallback).__name__
return result
+29 -1
View File
@@ -49,6 +49,25 @@ def _wrap_pdf_fast_path(pdf_parser: BaseParser) -> BaseParser:
)
def _gained_format_entries() -> Dict[str, BaseParser]:
"""Anydoc-only formats (legacy/macro Office, OpenDocument, RTF) for every map.
anydoc is a core dependency, so these suffixes are parseable under both
engines. When anydoc is somehow missing the entries are omitted, and such
files fall to ``SimpleDirectoryReader``'s plain-text read — the
pre-existing behaviour for unmapped suffixes.
"""
from application.parser.file.anydoc_parser import (
ANYDOC_GAINED_SUFFIXES,
AnydocParser,
anydoc_available,
)
if not anydoc_available():
return {}
return {suffix: AnydocParser() for suffix in ANYDOC_GAINED_SUFFIXES}
def _legacy_file_extractor(pdf_text_fast_path: bool = False) -> Dict[str, BaseParser]:
"""Parser map that needs neither docling nor anydoc.
@@ -79,6 +98,7 @@ def _legacy_file_extractor(pdf_text_fast_path: bool = False) -> Dict[str, BasePa
".jpg": ImageParser(),
".jpeg": ImageParser(),
**_build_audio_parser_mapping(),
**_gained_format_entries(),
}
@@ -165,6 +185,8 @@ def _docling_file_extractor(
".xml": DoclingXMLParser(),
# Formats docling doesn't support - use standard parsers
".epub": EpubParser(),
# Formats only anydoc reads (legacy/macro Office, OpenDocument, RTF)
**_gained_format_entries(),
}
@@ -211,7 +233,13 @@ def _anydoc_file_extractor(ocr_enabled: bool, pdf_text_fast_path: bool = False)
)
extractor = dict(base)
for suffix in ANYDOC_SUFFIXES:
extractor[suffix] = AnydocParser(fallback_parser=base.get(suffix))
fallback = base.get(suffix)
if isinstance(fallback, AnydocParser):
# A gained-format entry from the base map is already a
# fallback-less AnydocParser — anydoc delegating to anydoc would
# just repeat the same failure.
continue
extractor[suffix] = AnydocParser(fallback_parser=fallback)
extractor[".html"] = HTMLMarkdownParser()
extractor[".xhtml"] = HTMLMarkdownParser()
return extractor
+7
View File
@@ -1,5 +1,6 @@
"""Shared file-extension constants for parsing and ingestion flows."""
from application.parser.file.anydoc_parser import ANYDOC_GAINED_SUFFIXES
from application.stt.constants import SUPPORTED_AUDIO_EXTENSIONS
@@ -16,6 +17,12 @@ SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS = (
".json",
".xlsx",
".pptx",
# Read by the HTML parsers on every engine.
".xhtml",
# Read by the anydoc engine (legacy/macro Office, OpenDocument, RTF).
# Parseable regardless of DOC_PARSER_ENGINE: anydoc is a core dependency,
# and both parser maps route these suffixes to it.
*ANYDOC_GAINED_SUFFIXES,
)
SUPPORTED_SOURCE_IMAGE_EXTENSIONS = (".png", ".jpg", ".jpeg")
+187 -49
View File
@@ -10,6 +10,8 @@ import importlib.util
import logging
import os
import re
import shutil
import sys
import tempfile
import zipfile
from pathlib import Path
@@ -71,6 +73,112 @@ def _apply_inference_settings() -> None:
inference.compile_torch_models = settings.DOCLING_COMPILE_TORCH_MODELS
_VALID_OCR_ENGINES = ("tesseract", "auto", "ocrmac", "rapidocr", "deepseek")
def _resolve_ocr_engine(requested: Optional[str]) -> str:
"""Resolve the OCR engine to build, degrading to ``auto`` when unavailable.
``auto`` is docling's own selection (ocrmac on macOS, rapidocr on a
typical Linux server). The degradation is deliberate: OCR is switched on
by deployments that expect scans to work, so a missing engine must warn
and OCR with what exists rather than fail every parse.
Args:
requested: Engine name, or None to read ``settings.OCR_ENGINE``.
Returns:
One of ``_VALID_OCR_ENGINES``, guaranteed buildable here.
"""
from application.core.settings import settings
from application.parser.file.base_parser import module_available
engine = str(requested or settings.OCR_ENGINE or "auto").strip().lower()
if engine not in _VALID_OCR_ENGINES:
logger.warning(
f"Unknown OCR_ENGINE {engine!r}; using docling auto-selection"
)
return "auto"
if engine == "tesseract" and shutil.which("tesseract") is None:
logger.warning(
"OCR_ENGINE=tesseract but no tesseract binary is on PATH (install "
"tesseract-ocr plus language packs, or build the Docker image with "
"INSTALL_DOCLING=true); using docling auto-selection"
)
return "auto"
if engine == "ocrmac" and (
sys.platform != "darwin" or not module_available("ocrmac")
):
logger.warning(
"OCR_ENGINE=ocrmac needs macOS with the ocrmac package; "
"using docling auto-selection"
)
return "auto"
if engine == "rapidocr" and not module_available("rapidocr"):
logger.warning(
"OCR_ENGINE=rapidocr but rapidocr is not installed; "
"using docling auto-selection"
)
return "auto"
return engine
def _build_ocr_options(
engine: str, languages: Optional[List[str]], force_full_page_ocr: bool
):
"""docling OCR options for a resolved classic engine.
Returns None for ``auto`` (docling's default pipeline options already run
auto-selection) and on any build failure — the parse then proceeds on the
default engine rather than failing; the caller re-applies
``force_full_page_ocr`` onto whatever options end up active.
Args:
engine: A ``_resolve_ocr_engine`` result other than ``deepseek``.
languages: Engine-specific language list; None uses the engine's
default (tesseract reads ``settings.OCR_LANGS``).
force_full_page_ocr: OCR whole pages instead of only bitmap regions.
"""
if engine == "auto":
return None
from application.core.settings import settings
try:
if engine == "tesseract":
from docling.datamodel.pipeline_options import TesseractCliOcrOptions
langs = (
languages
or [lang.strip() for lang in settings.OCR_LANGS.split("+") if lang.strip()]
or ["eng"]
)
return TesseractCliOcrOptions(
lang=langs, force_full_page_ocr=force_full_page_ocr
)
if engine == "rapidocr":
from docling.datamodel.pipeline_options import RapidOcrOptions
return RapidOcrOptions(
lang=languages or ["english"],
force_full_page_ocr=force_full_page_ocr,
)
if engine == "ocrmac":
from docling.datamodel.pipeline_options import OcrMacOptions
if languages:
return OcrMacOptions(
lang=languages, force_full_page_ocr=force_full_page_ocr
)
return OcrMacOptions(force_full_page_ocr=force_full_page_ocr)
except ImportError as e:
logger.warning(f"Failed to build {engine} OCR options: {e}")
return None
except Exception as e:
logger.error(f"Error building {engine} OCR options: {e}")
return None
return None
def _tabular_content_size(file: Path) -> int:
"""Effective content size of a tabular file, in bytes.
@@ -303,7 +411,7 @@ class DoclingParser(BaseParser):
- Advanced PDF layout analysis
- Table structure recognition
- Reading order detection
- OCR for scanned documents (supports RapidOCR)
- OCR for scanned documents (engine chosen by ``OCR_ENGINE``)
- Unified DoclingDocument format
- Export to Markdown
@@ -317,7 +425,7 @@ class DoclingParser(BaseParser):
ocr_enabled: bool = True,
table_structure: bool = True,
export_format: str = "markdown",
use_rapidocr: bool = True,
ocr_engine: Optional[str] = None,
ocr_languages: Optional[List[str]] = None,
force_full_page_ocr: bool = False,
):
@@ -327,29 +435,33 @@ class DoclingParser(BaseParser):
ocr_enabled: Enable OCR for bitmap/image regions in documents
table_structure: Enable table structure recognition
export_format: Output format ('markdown', 'text', 'html')
use_rapidocr: Use RapidOCR engine (default True, works well in Docker)
ocr_languages: List of OCR languages (default: ['english'])
ocr_engine: OCR engine when OCR is enabled — one of
``tesseract | auto | ocrmac | rapidocr | deepseek``. None
reads ``settings.OCR_ENGINE`` at converter build time; an
unavailable engine degrades to docling's auto-selection with
a warning.
ocr_languages: Engine-specific language list; None keeps the
engine's own default (tesseract reads ``settings.OCR_LANGS``).
force_full_page_ocr: Force OCR on entire page (False = smart hybrid OCR)
"""
super().__init__()
self.ocr_enabled = ocr_enabled
self.table_structure = table_structure
self.export_format = export_format
self.use_rapidocr = use_rapidocr
self.ocr_languages = ocr_languages or ["english"]
self.ocr_engine = ocr_engine
self.ocr_languages = ocr_languages
self.force_full_page_ocr = force_full_page_ocr
self._converter = None
def _create_converter(self):
"""Create a docling converter with hybrid OCR configuration.
"""Create a docling converter for the configured OCR engine.
Uses smart OCR approach:
- When ocr_enabled=True and force_full_page_ocr=False (default):
Layout model detects text vs bitmap regions, OCR only runs on bitmaps
- When ocr_enabled=True and force_full_page_ocr=True:
OCR runs on entire page (for scanned documents/images)
- When ocr_enabled=False:
No OCR, only native text extraction
- ``ocr_enabled=False``: no OCR, native text extraction only.
- Classic engines (tesseract / auto / ocrmac / rapidocr): the standard
PDF pipeline (layout + TableFormer) with that engine's OCR options.
``force_full_page_ocr=False`` (default) OCRs only the bitmap regions
the layout model finds; True routes whole pages through OCR.
- ``deepseek``: the VLM pipeline instead (``_create_vlm_converter``).
Returns:
DocumentConverter instance
@@ -364,6 +476,10 @@ class DoclingParser(BaseParser):
_apply_inference_settings()
engine = _resolve_ocr_engine(self.ocr_engine) if self.ocr_enabled else None
if engine == "deepseek":
return self._create_vlm_converter()
pipeline_options = PdfPipelineOptions(
do_ocr=self.ocr_enabled,
do_table_structure=self.table_structure,
@@ -371,13 +487,15 @@ class DoclingParser(BaseParser):
_apply_pipeline_caps(pipeline_options)
if self.ocr_enabled:
ocr_options = self._get_ocr_options()
ocr_options = _build_ocr_options(
engine, self.ocr_languages, self.force_full_page_ocr
)
if ocr_options is not None:
pipeline_options.ocr_options = ocr_options
# Docling's *default* OCR options carry their own flag, so without
# this the setting was silently dropped whenever `_get_ocr_options`
# returned None (use_rapidocr=False) — including the dropout retry,
# whose whole point is forcing full-page OCR.
# this the setting was silently dropped whenever no explicit
# options were built (engine=auto, or a build failure) — including
# the dropout retry, whose whole point is forcing full-page OCR.
active_ocr_options = getattr(pipeline_options, "ocr_options", None)
if hasattr(active_ocr_options, "force_full_page_ocr"):
active_ocr_options.force_full_page_ocr = self.force_full_page_ocr
@@ -393,12 +511,55 @@ class DoclingParser(BaseParser):
}
)
def _create_vlm_converter(self):
"""DeepSeek-OCR converter via docling's VLM pipeline.
Each page goes to an OpenAI-compatible endpoint (Ollama or vLLM;
``OCR_DEEPSEEK_URL`` / ``OCR_DEEPSEEK_MODEL``) and the grounded output
is parsed back into a DoclingDocument. This replaces the *entire*
classic pipeline — no layout/TableFormer/OCR models load in the worker
(~370 MB RSS vs 1.0-1.6 GB measured), the compute lives in the model
server. Bench trade-offs (2026-08): best table/CJK/degraded-scan
quality of every engine tried, ~10-20 s/page on modest hardware, and
occasional silent drops of page-level elements (titles).
"""
from docling.datamodel import vlm_model_specs
from docling.datamodel.pipeline_options import VlmPipelineOptions
from docling.document_converter import (
DocumentConverter,
ImageFormatOption,
InputFormat,
PdfFormatOption,
)
from docling.pipeline.vlm_pipeline import VlmPipeline
from application.core.settings import settings
vlm_options = vlm_model_specs.DEEPSEEKOCR_OLLAMA.model_copy(deep=True)
vlm_options.url = settings.OCR_DEEPSEEK_URL
vlm_options.params["model"] = settings.OCR_DEEPSEEK_MODEL
pipeline_options = VlmPipelineOptions(
vlm_options=vlm_options, enable_remote_services=True
)
return DocumentConverter(
format_options={
InputFormat.PDF: PdfFormatOption(
pipeline_cls=VlmPipeline, pipeline_options=pipeline_options
),
InputFormat.IMAGE: ImageFormatOption(
pipeline_cls=VlmPipeline, pipeline_options=pipeline_options
),
}
)
def _init_parser(self) -> Dict:
"""Initialize the docling converter with hybrid OCR."""
from application.core.settings import settings
logger.info("Initializing DoclingParser...")
logger.info(f" ocr_enabled={self.ocr_enabled}")
logger.info(f" force_full_page_ocr={self.force_full_page_ocr}")
logger.info(f" use_rapidocr={self.use_rapidocr}")
logger.info(f" ocr_engine={self.ocr_engine or settings.OCR_ENGINE}")
if importlib.util.find_spec("docling.document_converter") is None:
raise ImportError(
@@ -414,34 +575,11 @@ class DoclingParser(BaseParser):
"ocr_enabled": self.ocr_enabled,
"table_structure": self.table_structure,
"export_format": self.export_format,
"use_rapidocr": self.use_rapidocr,
"ocr_engine": self.ocr_engine,
"ocr_languages": self.ocr_languages,
"force_full_page_ocr": self.force_full_page_ocr,
}
def _get_ocr_options(self):
"""Get OCR options based on configuration.
Returns RapidOcrOptions if use_rapidocr is True and available,
otherwise returns None to use docling defaults.
"""
if not self.use_rapidocr:
return None
try:
from docling.datamodel.pipeline_options import RapidOcrOptions
return RapidOcrOptions(
lang=self.ocr_languages,
force_full_page_ocr=self.force_full_page_ocr,
)
except ImportError as e:
logger.warning(f"Failed to import RapidOcrOptions: {e}")
return None
except Exception as e:
logger.error(f"Error creating RapidOcrOptions: {e}")
return None
def _export_content(self, document) -> str:
"""Export document content in the configured format.
@@ -651,7 +789,7 @@ class DoclingParser(BaseParser):
class DoclingPDFParser(DoclingParser):
"""Docling-based PDF parser with advanced features and RapidOCR support.
"""Docling-based PDF parser with advanced features and configurable OCR.
Uses hybrid OCR approach by default:
- Text regions: Direct PDF text extraction (fast)
@@ -664,7 +802,7 @@ class DoclingPDFParser(DoclingParser):
self,
ocr_enabled: bool = True,
table_structure: bool = True,
use_rapidocr: bool = True,
ocr_engine: Optional[str] = None,
ocr_languages: Optional[List[str]] = None,
force_full_page_ocr: bool = False,
):
@@ -672,7 +810,7 @@ class DoclingPDFParser(DoclingParser):
ocr_enabled=ocr_enabled,
table_structure=table_structure,
export_format="markdown",
use_rapidocr=use_rapidocr,
ocr_engine=ocr_engine,
ocr_languages=ocr_languages,
force_full_page_ocr=force_full_page_ocr,
)
@@ -723,7 +861,7 @@ class DoclingHTMLParser(DoclingParser):
class DoclingImageParser(DoclingParser):
"""Docling-based image parser with OCR and RapidOCR support.
"""Docling-based image parser with configurable OCR.
For images, force_full_page_ocr=True is used since images are entirely
visual and require full OCR to extract any text.
@@ -732,14 +870,14 @@ class DoclingImageParser(DoclingParser):
def __init__(
self,
ocr_enabled: bool = True,
use_rapidocr: bool = True,
ocr_engine: Optional[str] = None,
ocr_languages: Optional[List[str]] = None,
force_full_page_ocr: bool = True,
):
super().__init__(
ocr_enabled=ocr_enabled,
export_format="markdown",
use_rapidocr=use_rapidocr,
ocr_engine=ocr_engine,
ocr_languages=ocr_languages,
force_full_page_ocr=force_full_page_ocr,
)
+177
View File
@@ -0,0 +1,177 @@
"""Trust checks for PDF text extraction.
anydoc silently drops text from fonts it cannot map to Unicode — seen with
non-embedded composite (Type0/CID) fonts relying on predefined CMaps, where
the EN/ZH NDA corpus file loses its whole Chinese column with no error, and
with Adobe-CNS1 fonts in HK legal PDFs where a valid ToUnicode exists but
extraction still fails. These dependency-free checks scan the PDF's raw
objects (including Flate-compressed object streams) so the pipeline can
route such files to a heavier parser, or at least mark the output as
unverified, instead of trusting silent partial text.
Two stages:
* ``check_pdf_fonts`` — pre-flight on the bytes alone: composite fonts
without an embedded ToUnicode CMap, and whether the PDF declares CJK font
resources at all.
* ``verify_extraction`` — pre-flight plus the cross-check the font scan
cannot do: a PDF that declares CJK fonts whose extracted text contains
almost no CJK characters is not to be trusted.
A flag means "verify or route to a fallback", not "the text is wrong": tools
shipping Adobe's predefined CMaps (docling's parser does) often extract such
files fine. On the benchmark corpus the checks caught both known
silent-drop cases with zero false positives on the other 14 PDFs, and cost
~30 ms per scanned MB (92 ms on a 3 MB, 150-page annual report).
"""
import re
import zlib
from pathlib import Path
from typing import Dict, Iterator, List, Union
_OBJ = re.compile(rb"\d+\s+\d+\s+obj(.*?)endobj", re.DOTALL)
_STREAM = re.compile(rb"stream\r?\n(.*?)\r?\nendstream", re.DOTALL)
_TYPE0 = re.compile(rb"/Subtype\s*/Type0")
# The CJK expectation is keyed on CIDSystemInfo /Ordering ALONE — the
# authoritative "this PDF maps text through a CJK character collection"
# signal, present in every known silent-drop case. Matching CJK font *names*
# (SimSun, MS-Gothic, MSungHK, ...) was tried and rejected: Word-exported
# Latin PDFs routinely embed an MS-Gothic subset for a stray full-width
# character (a 150-page English annual report in the benchmark corpus does),
# and substring matching also catches Latin faces like FranklinGothic —
# either way English-only documents would be rerouted to the heavy engine.
_CJK_ORDERING = re.compile(rb"/Ordering\s*\((GB1|CNS1|Japan1|Japan2|KR|Korea1)\)")
# Bytes kept around a Type0 marker found inside a decompressed object
# stream: object streams hold many dicts with no obj/endobj markers, and a
# font dict's keys sit close to its /Subtype entry.
_WINDOW_BEFORE, _WINDOW_AFTER = 200, 800
# Extracted text with fewer CJK characters than this, from a PDF that
# declares CJK fonts, is treated as a silent drop.
_MIN_CJK_CHARS = 10
_CJK_RANGES = (
("一", "鿿"), # CJK Unified Ideographs
("぀", "ヿ"), # Hiragana + Katakana
("가", "힯"), # Hangul syllables
)
# Per-stream decompression cap. Flate reaches ~1000:1, so uncapped inflation
# of a crafted (or merely image-heavy) PDF inside the upload cap could balloon
# to gigabytes and OOM the ingest worker. Font dicts and object streams — the
# only things the signals live in — are far smaller than this.
_STREAM_INFLATE_CAP = 4_000_000
def _decompressed_streams(raw: bytes) -> Iterator[bytes]:
"""Each Flate stream in ``raw`` that decompresses, one at a time, capped."""
for match in _STREAM.finditer(raw):
try:
inflated = zlib.decompressobj().decompress(
match.group(1), _STREAM_INFLATE_CAP
)
except zlib.error:
continue
yield inflated
def _cjk_chars(text: str, up_to: int) -> int:
"""Count CJK characters in ``text``, stopping once ``up_to`` is reached."""
count = 0
for char in text:
for low, high in _CJK_RANGES:
if low <= char <= high:
count += 1
break
if count >= up_to:
break
return count
def check_pdf_fonts(data: bytes) -> Dict[str, Union[int, bool]]:
"""Pre-flight font scan of a PDF's bytes.
Args:
data: The complete PDF file contents.
Returns:
Dict with ``type0`` (composite fonts seen), ``type0_no_tounicode``
(those without an embedded ToUnicode CMap), ``expects_cjk`` (the PDF
declares CJK font resources), ``has_fonts``, and ``flagged`` — True
when extraction should not be trusted unverified.
"""
type0 = bare = 0
expects_cjk = bool(_CJK_ORDERING.search(data))
has_fonts = b"/Font" in data
def count_font(unit: bytes) -> None:
"""One analysis unit holding a Type0 font dict."""
nonlocal type0, bare
type0 += 1
if b"/ToUnicode" not in unit:
bare += 1
for match in _OBJ.finditer(data):
body = match.group(1)
if _TYPE0.search(body):
count_font(body)
# Each decompressed stream is scanned for every signal in one pass and
# then dropped; nothing inflated is retained past its loop iteration.
for stream in _decompressed_streams(data):
expects_cjk = expects_cjk or bool(_CJK_ORDERING.search(stream))
has_fonts = has_fonts or b"/Font" in stream
for type0_match in _TYPE0.finditer(stream):
start = type0_match.start()
count_font(stream[max(0, start - _WINDOW_BEFORE): start + _WINDOW_AFTER])
return {
"type0": type0,
"type0_no_tounicode": bare,
"expects_cjk": expects_cjk,
"has_fonts": has_fonts,
"flagged": bare > 0 or not has_fonts,
}
def verify_extraction(data: bytes, markdown: str) -> List[str]:
"""Reasons the extracted ``markdown`` of the PDF ``data`` shouldn't be trusted.
Combines the pre-flight font scan with the post-conversion cross-check it
cannot do alone: fonts with a valid ToUnicode that the converter
nevertheless failed to extract still show up as missing CJK output.
Args:
data: The complete PDF file contents.
markdown: The text a converter extracted from it.
Returns:
Human-readable problem strings; empty when the output looks sound.
"""
problems: List[str] = []
result = check_pdf_fonts(data)
if result["type0_no_tounicode"]:
problems.append(
f"{result['type0_no_tounicode']} composite (Type0) font(s) carry no "
"ToUnicode map; extracted text may silently omit glyphs"
)
elif not result["has_fonts"]:
problems.append(
"no font resources detected; text extraction cannot be verified"
)
if result["expects_cjk"]:
cjk = _cjk_chars(markdown, _MIN_CJK_CHARS)
if cjk < _MIN_CJK_CHARS:
problems.append(
f"PDF declares CJK fonts but the extracted text contains only "
f"{cjk} CJK character(s)"
)
return problems
def verify_pdf_file(file: Path, markdown: str) -> List[str]:
"""``verify_extraction`` for a file on disk; unreadable files trust the output."""
try:
data = Path(file).read_bytes()
except OSError:
return []
return verify_extraction(data, markdown)
+94
View File
@@ -0,0 +1,94 @@
"""Reconstruct typographic tables in anydoc's PDF Markdown output.
anydoc converts PDFs straight to Markdown with no document model, so tables
drawn with dot leaders or bare whitespace alignment (financial statements,
tables of contents) come out as flat text lines — values intact, structure
lost. This post-processor detects runs of such lines and rewrites them as
GFM tables: the lightweight alternative to a table-structure model for this
layout family. (The robust fix is upstream in anydoc's Rust PDF code, where
glyph x-positions exist; this is the recoverable-downstream version.)
Deliberately conservative — a run converts only when it has at least
``min_rows`` consecutive lines that each parse as ``label [leaders]
numeric-columns`` with the *same* column count; anything else passes through
untouched. Gated by ``ANYDOC_TABLEIZE`` (off by default) and applied only to
anydoc's own PDF output, never to docling's.
"""
import re
from typing import List, Optional, Tuple
# label ...... 1,234 (56) — dot-leader row
_LEADER = re.compile(r"^(.*?)\s*\.{3,}\s*(.+)$")
# numeric-ish token: $ 1,234 / (30) / 83,431 / 12.5% / —
_NUM_TOKEN = re.compile(r"\$?\(?-?\d[\d,.]*\)?%?|—")
# row without leaders: label text, then a trailing run of numeric tokens
_TRAILING = re.compile(r"^(.*?[^\s\d,.])\s+([\d$(].*)$")
def _merge_currency(tokens: List[str]) -> List[str]:
"""Join a free-standing ``$`` onto the number that follows it."""
out: List[str] = []
for token in tokens:
if out and out[-1] == "$":
out[-1] = "$" + token
else:
out.append(token)
return out
def _parse_row(line: str) -> Optional[Tuple[str, List[str]]]:
"""``(label, values)`` when the line looks like a typographic table row, else None."""
match = _LEADER.match(line) or _TRAILING.match(line)
if not match:
return None
label, rest = match.group(1).strip(" ."), match.group(2)
values = _merge_currency(rest.split())
if not values or not all(_NUM_TOKEN.fullmatch(v.lstrip("$")) for v in values):
return None
if not label or _NUM_TOKEN.fullmatch(label):
return None
return label, values
def tableize(markdown: str, min_rows: int = 3) -> str:
"""Rewrite runs of typographic table rows in ``markdown`` as GFM tables.
Args:
markdown: Converter output to post-process.
min_rows: Minimum consecutive, same-width rows for a run to convert.
Returns:
``markdown`` with qualifying runs rewritten; everything else verbatim.
"""
out: List[str] = []
run: List[Tuple[str, List[str], str]] = [] # (label, values, original line)
def flush() -> None:
nonlocal run
widths = {len(values) for _, values, _ in run}
if len(run) >= min_rows and len(widths) == 1:
ncols = widths.pop()
out.append("")
out.append("| " + " | ".join([" "] + [f"col{i + 1}" for i in range(ncols)]) + " |")
out.append("|" + " --- |" * (ncols + 1))
for label, values, _ in run:
# Only the label can carry a '|' (values are numeric tokens);
# unescaped it would add a cell and mis-column the row.
cells = [label.replace("|", "\\|")] + values
out.append("| " + " | ".join(cells) + " |")
out.append("")
else:
out.extend(original for _, _, original in run)
run = []
for line in markdown.splitlines():
parsed = _parse_row(line.strip()) if line.strip() else None
if parsed:
run.append((*parsed, line))
else:
if run:
flush()
out.append(line)
if run:
flush()
return "\n".join(out)
+1 -1
View File
@@ -1777,7 +1777,7 @@ def attachment_worker(self, file_info, user):
parser_metadata = {
key: value
for key, value in (attachment_document.extra_info or {}).items()
if key.startswith("transcript_")
if key.startswith("transcript_") or key == "parse_warnings"
}
if parser_metadata:
metadata = {**metadata, **parser_metadata}
@@ -213,6 +213,12 @@ for the engines and flows.
| `DOCLING_TABULAR_MAX_BYTES` | `2000000` | Docling engine: CSV/XLSX larger than this (by content) use the lightweight tabular parsers. |
| `DOCLING_MARKUP_MAX_BYTES` | `8000000` | Docling engine: HTML/VTT larger than this are head-truncated before parsing. |
| `MARKUP_MAX_BYTES` | `8000000` | anydoc engine: HTML/XHTML larger than this are head-truncated before the `markdownify` parser runs (its tree costs ~50x the input). `0` disables the gate. |
| `PDF_TRUST_CHECK` | `true` | Trust-check anydoc's PDF output for the two silent-loss classes (Type0 fonts without ToUnicode, CJK-declaring PDFs with CJK-less text). Flagged files re-parse on Docling when installed, else carry `parse_warnings` metadata. |
| `ANYDOC_TABLEIZE` | `false` | Rewrite dot-leader / whitespace-aligned table runs in anydoc's PDF markdown into GFM tables. |
| `OCR_ENGINE` | `tesseract` | OCR engine when OCR is on: `tesseract` (recommended), `auto`, `ocrmac`, `rapidocr`, or `deepseek` (VLM endpoint). Unavailable engines degrade to `auto` with a warning. See the [OCR guide](/Guides/ocr). |
| `OCR_LANGS` | `eng` | Tesseract language packs, `+`-separated (e.g. `eng+chi_sim+deu`). |
| `OCR_DEEPSEEK_URL` | `http://localhost:11434/v1/chat/completions` | OpenAI-compatible endpoint for `OCR_ENGINE=deepseek` (Ollama or vLLM). |
| `OCR_DEEPSEEK_MODEL` | `deepseek-ocr:3b` | Model name at that endpoint. |
## Speech-to-Text Settings
@@ -16,7 +16,7 @@ Training on other documentation sources can greatly enhance the versatility and
Make sure you have the document on which you want to train on ready with you on the device which you are using .You can also use links to the documentation to train on.
<Callout type="warning" emoji="⚠️">
Note: Supported file formats include .pdf, .txt, .rst, .docx, .md, .mdx, .csv, .epub, .html, .json, .xlsx, .pptx, .png, .jpg, .jpeg, and audio files (.wav, .mp3, .m4a, .ogg, .webm). You can also train using the link of the documentation.
Note: Supported file formats include .pdf, .txt, .rst, .md, .mdx, .html, .xhtml, .json, .epub, Office documents (.docx, .doc, .docm, .odt, .rtf), spreadsheets (.xlsx, .xls, .xlsm, .xlsb, .ods, .csv), presentations (.pptx, .ppt, .pptm, .pps, .ppsx, .ppsm, .pot, .odp), images (.png, .jpg, .jpeg), and audio files (.wav, .mp3, .m4a, .ogg, .webm). You can also train using the link of the documentation.
</Callout>
+48 -4
View File
@@ -66,15 +66,52 @@ docling is picked up as the fallback engine as soon as it is importable, and
## Docling OCR
OCR is optional, Docling-only, and controlled by two settings:
OCR is optional, Docling-only, and controlled by two on/off settings plus an
engine choice:
```env
DOCLING_OCR_ENABLED=false
DOCLING_OCR_ATTACHMENTS_ENABLED=false
OCR_ENGINE=tesseract
```
- `DOCLING_OCR_ENABLED`: OCR behavior for Source Docs ingestion.
- `DOCLING_OCR_ATTACHMENTS_ENABLED`: OCR behavior for chat attachments uploaded from the message box.
- `OCR_ENGINE`: which engine performs the OCR when it is on (next section).
Under the default anydoc engine, a scanned PDF reaches OCR through anydoc's
own detection: anydoc refuses it ("OCR is required") and the docling fallback
takes over. If that fallback also extracts almost nothing — OCR off, or no
docling installed — the upload now fails with a clear message instead of
silently indexing an empty document.
## Choosing the OCR engine
Benchmarked 2026-08 on English, bilingual EN/ZH, table-heavy and degraded
scans (all engines driven through docling so layout handling is identical):
| `OCR_ENGINE` | Role | Notes |
|---|---|---|
| `tesseract` | **recommended default** | Best classic-engine accuracy in the bench: perfect EN word recall on all docs, 0.000 CER on the bilingual page, 100% table cells, robust to mild degradation. ~35 MB of *system* packages, CPU-only. Needs the `tesseract` binary + language packs in the image — installed automatically when the Docker image is built with `INSTALL_DOCLING=true`; set languages via `OCR_LANGS` (e.g. `eng+chi_sim`). |
| `auto` | convenience | docling picks: `ocrmac` on macOS (excellent, ~1 s/page), `rapidocr` on Linux — see below before relying on it server-side. Also the automatic fallback whenever the selected engine is not installed. |
| `ocrmac` | macOS only | Best raw accuracy and fastest of all classic engines; irrelevant for Linux deploys. |
| `rapidocr` | pip-only fallback | No system packages needed, perfect on tables/CJK — but it silently shreds some long text lines into garbage at every setting tried, which is content loss for RAG ingestion. Avoid as a server default until fixed upstream. |
| `deepseek` | best quality, heavy on the *server* | DeepSeek-OCR through docling's VLM pipeline. Only engine that reconstructs totals rows as table rows; near-perfect CJK; barely affected by degradation. The ingestion worker gets *lighter* (~370 MB RSS — no layout models); the model runs in Ollama or vLLM. Costs: a GPU/Apple-Silicon endpoint, ~seconds per page, and occasional silent drops of page-level elements (titles). |
For `deepseek`, point the worker at an OpenAI-compatible endpoint:
```env
OCR_ENGINE=deepseek
OCR_DEEPSEEK_URL=http://localhost:11434/v1/chat/completions # Ollama default
OCR_DEEPSEEK_MODEL=deepseek-ocr:3b
```
Ollama works out of the box (`ollama pull deepseek-ocr:3b`); for real
throughput serve `deepseek-ai/DeepSeek-OCR` with vLLM on a GPU and set the
URL accordingly.
A selected engine that is not available (no tesseract binary, no macOS for
ocrmac) degrades to `auto` with a warning rather than failing parses.
## Processing Flow
@@ -104,10 +141,17 @@ Docling OCR behavior is different for PDFs vs images:
- bitmap/image regions: OCR only where needed
- Image parser defaults to full-page OCR (the whole image is visual content).
By default, Docling parser classes use RapidOCR options (language default: `english`).
The engine and its languages come from `OCR_ENGINE` and `OCR_LANGS` (see the
table above). The Docker image ships only the English tesseract pack; for
other languages install their packs in the image (e.g.
`apt-get install tesseract-ocr-chi-sim`) and list them in `OCR_LANGS`
(`eng+chi_sim`).
<Callout type="info" emoji="ℹ️">
Parser internals like OCR language and force-full-page OCR are currently set by code defaults, not separate `.env` settings.
<Callout type="warning" emoji="⚠️">
Upgrading from a RapidOCR-based deployment? RapidOCR covered English *and*
Chinese with no configuration. The tesseract default only OCRs the languages
in `OCR_LANGS` (`eng` out of the box), so CJK scans stop ingesting until
their packs are installed and listed.
</Callout>
### Model compilation
+49
View File
@@ -26,6 +26,23 @@ export const FILE_UPLOAD_ACCEPT: Record<string, string[]> = {
'application/vnd.openxmlformats-officedocument.presentationml.presentation': [
'.pptx',
],
'application/xhtml+xml': ['.xhtml'],
'application/msword': ['.doc'],
'application/vnd.ms-word.document.macroEnabled.12': ['.docm'],
'application/vnd.oasis.opendocument.text': ['.odt'],
'application/rtf': ['.rtf'],
'text/rtf': ['.rtf'],
'application/vnd.ms-powerpoint': ['.ppt', '.pps', '.pot'],
'application/vnd.ms-powerpoint.presentation.macroEnabled.12': ['.pptm'],
'application/vnd.openxmlformats-officedocument.presentationml.slideshow': [
'.ppsx',
],
'application/vnd.ms-powerpoint.slideshow.macroEnabled.12': ['.ppsm'],
'application/vnd.oasis.opendocument.presentation': ['.odp'],
'application/vnd.ms-excel': ['.xls'],
'application/vnd.ms-excel.sheet.macroEnabled.12': ['.xlsm'],
'application/vnd.ms-excel.sheet.binary.macroEnabled.12': ['.xlsb'],
'application/vnd.oasis.opendocument.spreadsheet': ['.ods'],
'image/png': ['.png'],
'image/jpeg': ['.jpeg'],
'image/jpg': ['.jpg'],
@@ -45,6 +62,22 @@ export const FILE_UPLOAD_ACCEPT_ATTR = [
'.epub',
'.xlsx',
'.pptx',
'.xhtml',
'.doc',
'.docm',
'.odt',
'.rtf',
'.ppt',
'.pptm',
'.pps',
'.ppsx',
'.ppsm',
'.pot',
'.odp',
'.xls',
'.xlsm',
'.xlsb',
'.ods',
'.png',
'.jpeg',
'.jpg',
@@ -76,6 +109,22 @@ export const SOURCE_FILE_TREE_ACCEPT_ATTR = [
'.json',
'.xlsx',
'.pptx',
'.xhtml',
'.doc',
'.docm',
'.odt',
'.rtf',
'.ppt',
'.pptm',
'.pps',
'.ppsx',
'.ppsm',
'.pot',
'.odp',
'.xls',
'.xlsm',
'.xlsb',
'.ods',
'.png',
'.jpg',
'.jpeg',
+1 -1
View File
@@ -765,7 +765,7 @@
"start": "Chat starten",
"name": "Name",
"choose": "Dateien auswählen",
"info": "Bitte lade .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip hoch (max. 25 MB)",
"info": "Bitte lade .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip hoch (max. 25 MB)",
"uploadedFiles": "Hochgeladene Dateien",
"cancel": "Abbrechen",
"train": "Trainieren",
+1 -1
View File
@@ -770,7 +770,7 @@
"start": "Start Chatting",
"name": "Name",
"choose": "Choose Files",
"info": "Please upload .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip limited to 25mb",
"info": "Please upload .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip limited to 25mb",
"uploadedFiles": "Uploaded Files",
"cancel": "Cancel",
"train": "Train",
+1 -1
View File
@@ -765,7 +765,7 @@
"start": "Comenzar a chatear",
"name": "Nombre",
"choose": "Seleccionar Archivos",
"info": "Por favor, sube archivos .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip limitados a 25MB",
"info": "Por favor, sube archivos .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip limitados a 25MB",
"uploadedFiles": "Archivos Subidos",
"cancel": "Cancelar",
"train": "Entrenar",
+1 -1
View File
@@ -765,7 +765,7 @@
"start": "チャットを開始する",
"name": "名前",
"choose": "ファイルを選択",
"info": "25MBまでの.pdf、.txt、.rst、.csv、.xlsx、.docx、.md、.html、.epub、.json、.pptx、.zipファイルをアップロードしてください",
"info": "25MBまでの.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.epub、.json、.pptx、.ppt、.odp、.zipファイルをアップロードしてください",
"uploadedFiles": "アップロードされたファイル",
"cancel": "キャンセル",
"train": "トレーニング",
+1 -1
View File
@@ -765,7 +765,7 @@
"start": "Начать чат",
"name": "Имя",
"choose": "Выбрать файлы",
"info": "Пожалуйста, загрузите файлы .pdf, .txt, .rst, .csv, .xlsx, .docx, .md, .html, .epub, .json, .pptx, .zip размером до 25 МБ",
"info": "Пожалуйста, загрузите файлы .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .epub, .json, .pptx, .ppt, .odp, .zip размером до 25 МБ",
"uploadedFiles": "Загруженные файлы",
"cancel": "Отмена",
"train": "Тренировка",
+1 -1
View File
@@ -765,7 +765,7 @@
"start": "開始對話",
"name": "名稱",
"choose": "選擇檔案",
"info": "請上傳限制為25MB的.pdf、.txt、.rst、.csv、.xlsx、.docx、.md、.html、.epub、.json、.pptx、.zip檔案",
"info": "請上傳限制為25MB的.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.epub、.json、.pptx、.ppt、.odp、.zip檔案",
"uploadedFiles": "已上傳檔案",
"cancel": "取消",
"train": "訓練",
+1 -1
View File
@@ -765,7 +765,7 @@
"start": "开始聊天",
"name": "名称",
"choose": "选择文件",
"info": "请上传限制为25MB的.pdf、.txt、.rst、.csv、.xlsx、.docx、.md、.html、.epub、.json、.pptx、.zip文件",
"info": "请上传限制为25MB的.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.epub、.json、.pptx、.ppt、.odp、.zip文件",
"uploadedFiles": "已上传文件",
"cancel": "取消",
"train": "训练",
@@ -0,0 +1,101 @@
%PDF-1.4
%“Œ‹ž ReportLab Generated PDF document (opensource)
1 0 obj
<<
/F1 2 0 R /F2 3 0 R /F3 4 0 R /F4 5 0 R
>>
endobj
2 0 obj
<<
/BaseFont /Helvetica /Encoding /WinAnsiEncoding /Name /F1 /Subtype /Type1 /Type /Font
>>
endobj
3 0 obj
<<
/BaseFont /Times-Bold /Encoding /WinAnsiEncoding /Name /F2 /Subtype /Type1 /Type /Font
>>
endobj
4 0 obj
<<
/BaseFont /STSong-Light /DescendantFonts [ <<
/BaseFont /STSong-Light /CIDSystemInfo <<
/Ordering (GB1) /Registry (Adobe) /Supplement 0
>> /DW 1000 /FontDescriptor <<
/Ascent 752 /CapHeight 737 /Descent -271 /Flags 6 /FontBBox [ -25 -254 1000 880 ] /FontName /STSongStd-Light
/ItalicAngle 0 /Leading 148 /MaxWidth 1000 /MissingWidth 500 /StemH 91 /StemV 58
/Type /FontDescriptor /XHeight 553
>> /Subtype /CIDFontType0 /Type /Font
/W [ 1 [ 207 270 342 467 462 797 710 239 374 ] 10 [ 374 423 605 238 375 238 334 462 ] 18 26 462 27 28 238
29 31 605 32 [ 344 748 684 560 695 739 563 511 729 793
318 312 666 526 896 758 772 544 772 628
465 607 753 711 972 647 620 607 374 333
374 606 500 239 417 503 427 529 415 264
444 518 241 230 495 228 793 527 524 ] 81 [ 524 504 338 336 277 517 450 652 466 452
407 370 258 370 605 ] ]
>> ] /Encoding /UniGB-UCS2-H /Name /F3 /Subtype /Type0 /Type /Font
>>
endobj
5 0 obj
<<
/BaseFont /Times-Roman /Encoding /WinAnsiEncoding /Name /F4 /Subtype /Type1 /Type /Font
>>
endobj
6 0 obj
<<
/Contents 10 0 R /MediaBox [ 0 0 595.2756 841.8898 ] /Parent 9 0 R /Resources <<
/Font 1 0 R /ProcSet [ /PDF /Text /ImageB /ImageC /ImageI ]
>> /Rotate 0 /Trans <<
>>
/Type /Page
>>
endobj
7 0 obj
<<
/PageMode /UseNone /Pages 9 0 R /Type /Catalog
>>
endobj
8 0 obj
<<
/Author (\(anonymous\)) /CreationDate (D:20260822115950+02'00') /Creator (\(unspecified\)) /Keywords () /ModDate (D:20260822115950+02'00') /Producer (ReportLab PDF Library - \(opensource\))
/Subject (\(unspecified\)) /Title (\(anonymous\)) /Trapped /False
>>
endobj
9 0 obj
<<
/Count 1 /Kids [ 6 0 R ] /Type /Pages
>>
endobj
10 0 obj
<<
/Filter [ /ASCII85Decode /FlateDecode ] /Length 1647
>>
stream
Gatm<gN):C&:O:SoKt6S$HmLQ&3Rqa-FV*"%Nm-aMW-4=#+,j_=@nmi2Z)t-)*e6=$[Ld,&=Q79:X>PPVJQYC5?'k&+i&d>B>799%,;FkdkJW*5heHuF@91dXh%AJZ6fb3i'sKhCnMdY'0VkEO4GA.p;"DJY62!"YAIse&.WT60%ek]="]O%+5Ot<#Fl%^o]il7fm2dGp=dm)m^:gDD9"d?pER14WDHmu']GE^2k$Vs`2KagWk")2f26\M]Bs+K@uNQtga3OQkLQ7H-hY4>1!4.4XoK+nl.ZlK#$d"IN+=9q.'8T'8cF3g*[-3:LS5c>fi!l0mSVa4QpjYgA<$0LC`2-c:@^;pCa)ekN>!I%&S3he7oPPIfI^JJ8"'-"K,>QL6DaesnS'ThOk/)63Q<B9b;^E^!3N(^L"g8T-_[_e%ED-JKcd/L:T#=nOk`'\l#7Q=,(RK;`0#S<T4BU:`HhPGE"*QrTjD+LciNhe\[arIY[:*6r!Z(b46[T;c]'eT2ib4:#PYiS4KbfJkaWh^D3%L,2kV2tNtL-eVVV!V=WcZC*LmZ1JF,Q`_WpI/i"9jUr1Gd(ISB[Id?[]g@-RLQD"d?"iI[81]"&B'nLTX>D<J'G5>]1d]+<&nMsjUuVm%-`_Dn^N7Y3@(%aO`>?OaQR7"S`I)*4n^gDS'@E((C?3'qDA#_!0S=@#*`D<h%J/PD"f$4CY;-(VXO;.Ngc^I+,-r\]qpUQHC*nl`bn!js*R3SkJqAMhH<a9;!@-5%1$9AH:,n;b&'!\_DddM7mc;^=s`,BOb&%SnDA7Rl;HLN*t%(`jGJS&".8T-M[a.`>blJkd8M[QOjG@Lffk@WSsIihjLDN`q?@n!YaFRc6etUSZ3TeZg#m?mL-MTMO5V)b]\Em&*E2^jCc`Xsk1<"JLgnj,EWu^Kg\"3.u<u5YCaES/ZDf5DpeMd24qDipat:7Oab*_jqT_s6MJ(i_J60;p<ik,/MGX2+u90)Xt!desS2A+)qeT6sUK/IJ'C=1<0u0>SaqZ<FhNoMfaf*!B2`NiLP;eB4?[`;/Y%o&Gm)'qO1JK0uckb!lh3>=t73#j/BK3<dLaZd96%gq]>W+8,V.,0D<(=M`#OG+,1-Fh</5[IM)#q0>k-!*=d[eI7&!b)UOUK$3XO.Y5HWgM[D<SAa`SCR9MN]H2Q[$5(+ZDUq;e-9V7T89Qo[Rm`4fG0o4@V.p48_A)/:trd:.`nh*S%-%oo[?Go6OFLjPL]dfcTW*/@I<%u%sqi96p7;=;:bYI+:AU5':P>$c<"Wh_I_.r/0:j+[",VmP^3"R7]Vuj1Bju_M4fQpgg+K1-[?e$i#W^p3[3m^DdEM$S$d\G^r%4.<YC,K9<kt81fnor-c7+%aA'BW<5M_bT$PN_@IR!Yssf=Ie",juu4&#3f]0+Uq".M]_sN3S\fN4OR!lJL2(a=:b4Ff&7E;q)O_P#WqT?;^@'e:m]2q4Y\']#rYS`0J,.E<b/Dai'5?mlc6AhJna\JGg_:Yc:98:nsRLal!#c+EsM(<tb(J^cTEVjq!Ll`V)/=\n1?2VP!nUibVne=IuE;BNdBFoimp4D_?%=$%6*-QW3f0rnX&<H85'1fg=@QlClK`^N0$X\\#1DFp#6kjo8*>*l66RfIdu>~>endstream
endobj
xref
0 11
0000000000 65535 f
0000000061 00000 n
0000000122 00000 n
0000000229 00000 n
0000000337 00000 n
0000001270 00000 n
0000001379 00000 n
0000001583 00000 n
0000001651 00000 n
0000001931 00000 n
0000001990 00000 n
trailer
<<
/ID
[<6167f87b12d550221d4edd2d2c131d94><6167f87b12d550221d4edd2d2c131d94>]
% ReportLab generated PDF document -- digest (opensource)
/Info 8 0 R
/Root 7 0 R
/Size 11
>>
startxref
3729
%%EOF
+226 -3
View File
@@ -27,10 +27,14 @@ from application.parser.file.anydoc_parser import ( # noqa: E402 — after impo
)
# Long enough to clear the scanned-PDF near-empty guard (_MIN_SCAN_FALLBACK_CHARS).
FALLBACK_TEXT = "fallback parser text output, long enough to clear the scanned-PDF guard."
class _RecordingFallback(BaseParser):
"""Stands in for docling / a legacy parser so delegation is observable."""
def __init__(self, result="fallback text"):
def __init__(self, result=FALLBACK_TEXT):
super().__init__(parser_config={})
self.calls = []
self._result = result
@@ -146,7 +150,7 @@ def test_scanned_pdf_delegates_to_fallback(tmp_path):
out = parser.parse_file(path)
assert out == "fallback text"
assert out == FALLBACK_TEXT
assert fallback.calls == [path]
assert parser.last_engine == "_RecordingFallback"
@@ -229,7 +233,7 @@ def test_typed_convert_error_delegates(tmp_path, monkeypatch):
path = tmp_path / "x.pdf"
path.write_bytes(b"%PDF-1.4")
assert AnydocParser(fallback_parser=fallback).parse_file(path) == "fallback text"
assert AnydocParser(fallback_parser=fallback).parse_file(path) == FALLBACK_TEXT
def test_typed_convert_error_message_reaches_the_user(tmp_path, monkeypatch):
@@ -311,3 +315,222 @@ def test_init_parser_imports_for_real_not_just_find_spec(monkeypatch):
with pytest.raises(ImportError, match="firecrawl-anydoc"):
AnydocParser().init_parser()
# --- PDF trust check + tableize wiring (PR 3) -----------------------------------
from pathlib import Path as _Path
FIXTURES = _Path(__file__).parent / "fixtures"
CID_PDF = FIXTURES / "nda_en_zh_cid_font.pdf"
# Long enough to clear the near-empty guard a trust-check re-parse must pass.
_REROUTE_TEXT = "docling reroute text, long enough to clear the near-empty reroute guard"
class _FakeDoclingFallback:
"""Registered as a DoclingParser subclass so ``_is_docling_backed`` is True."""
def __new__(cls):
from application.parser.file.docling_parser import DoclingParser
class _Inner(DoclingParser):
def __init__(self):
self._parser_config = {}
self.calls = []
self.last_engine = None
def parse_file(self, file, errors="ignore"):
self.calls.append(file)
return _REROUTE_TEXT
return _Inner()
def test_trust_flagged_pdf_reroutes_to_docling_fallback():
fallback = _FakeDoclingFallback()
parser = AnydocParser(fallback_parser=fallback)
out = parser.parse_file(CID_PDF)
assert out == _REROUTE_TEXT
assert fallback.calls == [CID_PDF]
assert parser.last_engine == "_Inner"
assert parser.get_file_metadata(CID_PDF) == {} # rerouted, nothing to warn about
def test_trust_flagged_pdf_without_docling_keeps_output_and_warns():
parser = AnydocParser(fallback_parser=_RecordingFallback()) # not docling-backed
out = parser.parse_file(CID_PDF)
assert "NON-DISCLOSURE" in out.upper() or len(out) > 50 # anydoc's own output kept
meta = parser.get_file_metadata(CID_PDF)
assert "parse_warnings" in meta
assert any("ToUnicode" in w for w in meta["parse_warnings"])
assert any("CJK" in w for w in meta["parse_warnings"])
assert parser.last_engine == "anydoc"
def test_trust_check_disabled_stamps_nothing(monkeypatch):
from application.parser.file import anydoc_parser as ap
monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False)
parser = AnydocParser()
parser.parse_file(CID_PDF)
assert parser.get_file_metadata(CID_PDF) == {}
def test_trust_reroute_failure_keeps_anydoc_output():
fallback = _FakeDoclingFallback()
def _boom(file, errors="ignore"):
raise RuntimeError("docling exploded")
fallback.parse_file = _boom
parser = AnydocParser(fallback_parser=fallback)
out = parser.parse_file(CID_PDF)
assert "docling" not in out
assert "parse_warnings" in parser.get_file_metadata(CID_PDF)
assert parser.last_engine == "anydoc"
def test_trust_reroute_near_empty_keeps_anydoc_output():
"""A docling pipeline dropout ('' / '<!-- image -->') returns without
raising; adopting it would swap anydoc's real text for an empty document."""
fallback = _FakeDoclingFallback()
fallback.parse_file = lambda file, errors="ignore": "<!-- image -->"
parser = AnydocParser(fallback_parser=fallback)
out = parser.parse_file(CID_PDF)
assert "<!-- image -->" not in out
assert "parse_warnings" in parser.get_file_metadata(CID_PDF)
assert parser.last_engine == "anydoc"
def test_warnings_reset_between_files(tmp_path):
parser = AnydocParser()
parser.parse_file(CID_PDF)
assert parser.get_file_metadata(CID_PDF) != {}
clean = tmp_path / "clean.csv"
clean.write_text("a,b\n1,2\n")
parser.parse_file(clean)
assert parser.get_file_metadata(clean) == {}
assert parser.get_file_metadata(CID_PDF) == {}
def test_trust_check_errors_never_fail_the_parse(monkeypatch, tmp_path):
def _explode(path, markdown):
raise RuntimeError("scanner bug")
import application.parser.file.pdf_trust as pt
monkeypatch.setattr(pt, "verify_pdf_file", _explode)
_fake_anydoc(monkeypatch, lambda path: "# converted fine")
path = tmp_path / "x.pdf"
path.write_bytes(b"%PDF-1.4 irrelevant")
assert AnydocParser().parse_file(path) == "# converted fine"
def test_tableize_applied_when_enabled(monkeypatch, tmp_path):
from application.parser.file import anydoc_parser as ap
monkeypatch.setattr(ap.settings, "ANYDOC_TABLEIZE", True)
monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False)
_fake_anydoc(
monkeypatch,
lambda path: "Cash ..... 1,234 900\nDebt ..... 2,000 1,500\nEquity ..... 900 800",
)
path = tmp_path / "x.pdf"
path.write_bytes(b"%PDF-1.4 irrelevant")
out = AnydocParser().parse_file(path)
assert "| Cash | 1,234 | 900 |" in out
def test_tableize_disabled_by_default(monkeypatch, tmp_path):
from application.parser.file import anydoc_parser as ap
monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False)
flat = "Cash ..... 1,234 900\nDebt ..... 2,000 1,500\nEquity ..... 900 800"
_fake_anydoc(monkeypatch, lambda path: flat)
path = tmp_path / "x.pdf"
path.write_bytes(b"%PDF-1.4 irrelevant")
assert AnydocParser().parse_file(path) == flat
def test_tableize_never_touches_docling_reroute(monkeypatch):
from application.parser.file import anydoc_parser as ap
monkeypatch.setattr(ap.settings, "ANYDOC_TABLEIZE", True)
fallback = _FakeDoclingFallback()
parser = AnydocParser(fallback_parser=fallback)
out = parser.parse_file(CID_PDF)
assert out == _REROUTE_TEXT # verbatim, not post-processed
# --- scanned-PDF near-empty guard (the OCR_ENGINE seam's loud-failure side) ------
def test_scanned_pdf_with_near_empty_fallback_fails_loudly_when_ocr_off(tmp_path):
"""A scan whose fallback (OCR off) extracts almost nothing must fail with
an actionable message, not be stored as an empty document."""
path = _scanned_pdf(tmp_path / "scan.pdf")
fallback = _RecordingFallback(result=" ")
fallback.ocr_enabled = False
parser = AnydocParser(fallback_parser=fallback)
with pytest.raises(DocumentParseError, match="DOCLING_OCR_ENABLED"):
parser.parse_file(path)
assert parser.last_engine is None
def test_scanned_pdf_with_near_empty_fallback_and_ocr_on_reports_it(tmp_path):
path = _scanned_pdf(tmp_path / "scan.pdf")
fallback = _RecordingFallback(result="x")
fallback.ocr_enabled = True
with pytest.raises(DocumentParseError, match="even with OCR enabled"):
AnydocParser(fallback_parser=fallback).parse_file(path)
def test_scanned_pdf_with_substantial_fallback_output_passes(tmp_path):
"""docling extracting a text layer anydoc refused (the HK-bill case) must
keep working — the guard only fires on near-empty results."""
path = _scanned_pdf(tmp_path / "scan.pdf")
out = AnydocParser(fallback_parser=_RecordingFallback()).parse_file(path)
assert out == FALLBACK_TEXT
def test_non_scan_refusals_do_not_trigger_the_guard(tmp_path, monkeypatch):
"""Only the OCR-required refusal implies 'content exists but needs OCR';
a malformed file with a short fallback result stays a successful parse."""
fake = _fake_anydoc(monkeypatch, None)
class MalformedError(fake.ConvertError):
pass
fake.MalformedError = MalformedError
def _refuse(path):
raise MalformedError("structurally unusable")
fake.to_markdown = _refuse
path = tmp_path / "x.docx"
path.write_bytes(b"irrelevant")
out = AnydocParser(fallback_parser=_RecordingFallback(result="tiny")).parse_file(path)
assert out == "tiny"
+112
View File
@@ -628,3 +628,115 @@ class TestParserEngineSwitch:
assert ".xhtml" in extractor
for suffix in (".pdf", ".docx", ".csv", ".xlsx", ".html", ".xhtml", ".pptx"):
extractor[suffix].init_parser()
# =====================================================================
# Gained formats (anydoc-only: legacy/macro Office, OpenDocument, RTF)
# =====================================================================
@pytest.mark.unit
class TestGainedFormats:
def test_every_supported_extension_has_a_parser_under_both_engines(self):
"""The invariant that keeps constants.py and the parser maps in lockstep.
Without it, an extension accepted at upload but missing from the map
falls to SimpleDirectoryReader's plain-text read — for a binary
format that means mojibake silently ingested and embedded.
(`.txt` is the one deliberate plain-text read.)
"""
pytest.importorskip("anydoc")
from application.parser.file.bulk import get_default_file_extractor
from application.parser.file.constants import (
SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS,
)
for engine in ("anydoc", "docling"):
extractor = get_default_file_extractor(engine=engine)
missing = [
suffix
for suffix in SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS
if suffix != ".txt" and suffix not in extractor
]
assert missing == [], (engine, missing)
def test_gained_formats_map_to_anydoc_under_both_engines(self):
pytest.importorskip("anydoc")
from application.parser.file.anydoc_parser import (
ANYDOC_GAINED_SUFFIXES,
AnydocParser,
)
from application.parser.file.bulk import get_default_file_extractor
for engine in ("anydoc", "docling"):
extractor = get_default_file_extractor(engine=engine)
for suffix in ANYDOC_GAINED_SUFFIXES:
assert isinstance(extractor[suffix], AnydocParser), (engine, suffix)
def test_gained_formats_never_fall_back_to_anydoc_itself(self):
pytest.importorskip("anydoc")
from application.parser.file.anydoc_parser import AnydocParser
from application.parser.file.bulk import get_default_file_extractor
extractor = get_default_file_extractor(engine="anydoc")
assert extractor[".doc"].fallback_parser is None
assert extractor[".rtf"].fallback_parser is None
# ...while the core five keep their real fallback.
assert extractor[".pdf"].fallback_parser is not None
assert not isinstance(extractor[".pdf"].fallback_parser, AnydocParser)
def test_gained_entries_absent_without_anydoc(self, monkeypatch):
import sys
from application.parser.file.bulk import get_default_file_extractor
monkeypatch.setitem(sys.modules, "anydoc", None)
extractor = get_default_file_extractor(engine="docling")
assert ".doc" not in extractor # degrades to the pre-anydoc map
def test_rtf_converts_end_to_end(self, tmp_path):
pytest.importorskip("anydoc")
from application.parser.file.bulk import get_default_file_extractor
path = tmp_path / "note.rtf"
path.write_text(r"{\rtf1\ansi Hello {\b bold} world.\par Second paragraph.}")
parser = get_default_file_extractor()[".rtf"]
parser.init_parser()
out = parser.parse_file(path)
assert "Hello **bold** world." in out
assert "Second paragraph." in out
def test_odt_converts_end_to_end(self, tmp_path):
pytest.importorskip("anydoc")
import zipfile
from application.parser.file.bulk import get_default_file_extractor
path = tmp_path / "doc.odt"
with zipfile.ZipFile(path, "w") as z:
z.writestr(
"mimetype",
"application/vnd.oasis.opendocument.text",
compress_type=zipfile.ZIP_STORED,
)
z.writestr(
"META-INF/manifest.xml",
'<?xml version="1.0"?><manifest:manifest xmlns:manifest="urn:oasis:names:tc:opendocument:xmlns:manifest:1.0">'
'<manifest:file-entry manifest:full-path="/" manifest:media-type="application/vnd.oasis.opendocument.text"/>'
'<manifest:file-entry manifest:full-path="content.xml" manifest:media-type="text/xml"/></manifest:manifest>',
)
z.writestr(
"content.xml",
'<?xml version="1.0"?><office:document-content xmlns:office="urn:oasis:names:tc:opendocument:xmlns:office:1.0" '
'xmlns:text="urn:oasis:names:tc:opendocument:xmlns:text:1.0"><office:body><office:text>'
'<text:h text:outline-level="1">Title</text:h><text:p>Body text here.</text:p>'
"</office:text></office:body></office:document-content>",
)
parser = get_default_file_extractor()[".odt"]
out = parser.parse_file(path)
assert "# Title" in out
assert "Body text here." in out
+164 -64
View File
@@ -1,6 +1,6 @@
"""Comprehensive tests for application/parser/file/docling_parser.py
Covers: DoclingParser (init, _init_parser, _get_ocr_options, _export_content,
Covers: DoclingParser (init, _init_parser, OCR engine selection, _export_content,
parse_file), subclass initialization, error handling.
"""
@@ -27,8 +27,8 @@ class TestDoclingParserInit:
assert parser.ocr_enabled is True
assert parser.table_structure is True
assert parser.export_format == "markdown"
assert parser.use_rapidocr is True
assert parser.ocr_languages == ["english"]
assert parser.ocr_engine is None # None -> settings.OCR_ENGINE at build
assert parser.ocr_languages is None # None -> the engine's own default
assert parser.force_full_page_ocr is False
assert parser._converter is None
@@ -39,14 +39,14 @@ class TestDoclingParserInit:
ocr_enabled=False,
table_structure=False,
export_format="text",
use_rapidocr=False,
ocr_engine="rapidocr",
ocr_languages=["german"],
force_full_page_ocr=True,
)
assert parser.ocr_enabled is False
assert parser.table_structure is False
assert parser.export_format == "text"
assert parser.use_rapidocr is False
assert parser.ocr_engine == "rapidocr"
assert parser.ocr_languages == ["german"]
assert parser.force_full_page_ocr is True
@@ -90,44 +90,169 @@ class TestDoclingParserInitParser:
@pytest.mark.unit
class TestGetOCROptions:
class TestOcrEngineSelection:
"""``OCR_ENGINE`` resolution and per-engine option building."""
def test_returns_none_when_rapidocr_disabled(self):
@pytest.fixture
def settings(self):
from application.core.settings import settings
return settings
def test_default_setting_is_tesseract(self, settings):
assert settings.OCR_ENGINE == "tesseract"
assert settings.OCR_LANGS == "eng"
def test_none_reads_setting(self, settings, monkeypatch):
from application.parser.file.docling_parser import _resolve_ocr_engine
monkeypatch.setattr(settings, "OCR_ENGINE", "auto")
assert _resolve_ocr_engine(None) == "auto"
def test_unknown_engine_degrades_to_auto(self):
from application.parser.file.docling_parser import _resolve_ocr_engine
assert _resolve_ocr_engine("easyocr") == "auto"
def test_tesseract_without_binary_degrades_to_auto(self, monkeypatch):
import application.parser.file.docling_parser as dp
monkeypatch.setattr(dp.shutil, "which", lambda name: None)
assert dp._resolve_ocr_engine("tesseract") == "auto"
def test_tesseract_with_binary_selected(self, monkeypatch):
import application.parser.file.docling_parser as dp
monkeypatch.setattr(dp.shutil, "which", lambda name: "/usr/bin/tesseract")
assert dp._resolve_ocr_engine("tesseract") == "tesseract"
def test_ocrmac_off_darwin_degrades_to_auto(self, monkeypatch):
import application.parser.file.docling_parser as dp
monkeypatch.setattr(dp.sys, "platform", "linux")
assert dp._resolve_ocr_engine("ocrmac") == "auto"
def test_rapidocr_missing_degrades_to_auto(self, monkeypatch):
import sys
import application.parser.file.docling_parser as dp
monkeypatch.setitem(sys.modules, "rapidocr", None)
assert dp._resolve_ocr_engine("rapidocr") == "auto"
def test_deepseek_passes_through(self):
from application.parser.file.docling_parser import _resolve_ocr_engine
assert _resolve_ocr_engine("deepseek") == "deepseek"
def test_build_auto_returns_none(self):
from application.parser.file.docling_parser import _build_ocr_options
assert _build_ocr_options("auto", None, True) is None
def test_build_tesseract_reads_ocr_langs(self, settings, monkeypatch):
pytest.importorskip("docling")
from application.parser.file.docling_parser import _build_ocr_options
monkeypatch.setattr(settings, "OCR_LANGS", "eng+chi_sim")
options = _build_ocr_options("tesseract", None, True)
assert type(options).__name__ == "TesseractCliOcrOptions"
assert options.lang == ["eng", "chi_sim"]
assert options.force_full_page_ocr is True
def test_build_tesseract_explicit_languages_win(self):
pytest.importorskip("docling")
from application.parser.file.docling_parser import _build_ocr_options
options = _build_ocr_options("tesseract", ["deu"], False)
assert options.lang == ["deu"]
def test_build_rapidocr(self):
pytest.importorskip("docling")
from application.parser.file.docling_parser import _build_ocr_options
options = _build_ocr_options("rapidocr", None, False)
assert type(options).__name__ == "RapidOcrOptions"
assert options.lang == ["english"]
def test_build_import_failure_returns_none(self, monkeypatch):
import sys
from application.parser.file.docling_parser import _build_ocr_options
monkeypatch.setitem(sys.modules, "docling.datamodel.pipeline_options", None)
assert _build_ocr_options("tesseract", ["eng"], False) is None
assert _build_ocr_options("ocrmac", None, False) is None
def test_build_generic_failure_returns_none(self, monkeypatch):
import sys
import types
from application.parser.file.docling_parser import _build_ocr_options
fake = types.ModuleType("docling.datamodel.pipeline_options")
def _boom(**kwargs):
raise RuntimeError("bad options")
fake.RapidOcrOptions = _boom
monkeypatch.setitem(sys.modules, "docling.datamodel.pipeline_options", fake)
assert _build_ocr_options("rapidocr", None, False) is None
@pytest.mark.unit
class TestDeepseekVlmConverter:
"""OCR_ENGINE=deepseek swaps the whole pipeline for docling's VLM route."""
@pytest.fixture(autouse=True)
def _requires_docling(self):
pytest.importorskip("docling")
def test_deepseek_builds_vlm_converter(self, monkeypatch):
from application.core.settings import settings
from application.parser.file.docling_parser import DoclingParser
parser = DoclingParser(use_rapidocr=False)
assert parser._get_ocr_options() is None
monkeypatch.setattr(settings, "OCR_DEEPSEEK_URL", "http://gpu-host:8000/v1/chat/completions")
monkeypatch.setattr(settings, "OCR_DEEPSEEK_MODEL", "deepseek-ocr-x")
def test_returns_options_when_available(self):
built = {}
def _capture_converter(format_options):
built["format_options"] = format_options
return MagicMock()
monkeypatch.setattr(
"docling.document_converter.DocumentConverter", _capture_converter
)
parser = DoclingParser(ocr_enabled=True, ocr_engine="deepseek")
parser._create_converter()
from docling.datamodel.base_models import InputFormat
from docling.pipeline.vlm_pipeline import VlmPipeline
pdf_option = built["format_options"][InputFormat.PDF]
image_option = built["format_options"][InputFormat.IMAGE]
assert pdf_option.pipeline_cls is VlmPipeline
assert image_option.pipeline_cls is VlmPipeline
vlm = pdf_option.pipeline_options.vlm_options
assert vlm.url == "http://gpu-host:8000/v1/chat/completions"
assert vlm.params["model"] == "deepseek-ocr-x"
assert pdf_option.pipeline_options.enable_remote_services is True
def test_deepseek_ignored_when_ocr_disabled(self, monkeypatch):
from application.parser.file.docling_parser import DoclingParser
parser = DoclingParser(use_rapidocr=True, ocr_languages=["english"])
vlm_called = []
monkeypatch.setattr(
DoclingParser,
"_create_vlm_converter",
lambda self: vlm_called.append(True),
)
monkeypatch.setattr("docling.document_converter.DocumentConverter", MagicMock())
DoclingParser(ocr_enabled=False, ocr_engine="deepseek")._create_converter()
mock_options = MagicMock()
with patch(
"application.parser.file.docling_parser.DoclingParser._get_ocr_options",
return_value=mock_options,
):
result = parser._get_ocr_options()
assert result is mock_options
def test_returns_none_on_import_error(self):
from application.parser.file.docling_parser import DoclingParser
parser = DoclingParser(use_rapidocr=True)
# Simulate the ImportError path
original = parser._get_ocr_options
def patched_get_ocr():
try:
raise ImportError("No RapidOcrOptions")
except ImportError:
return None
parser._get_ocr_options = patched_get_ocr
assert parser._get_ocr_options() is None
parser._get_ocr_options = original
assert vlm_called == []
# =====================================================================
@@ -431,31 +556,6 @@ class TestDoclingSubclasses:
@pytest.mark.unit
class TestDoclingParserGaps:
def test_get_ocr_options_import_error_returns_none(self):
"""Cover lines 148-150: ImportError returns None."""
from application.parser.file.docling_parser import DoclingParser
parser = DoclingParser(ocr_enabled=True, use_rapidocr=True)
with patch.dict("sys.modules", {"docling.datamodel.pipeline_options": None}):
# Force re-import to trigger ImportError
with patch(
"builtins.__import__", side_effect=ImportError("no module")
):
result = parser._get_ocr_options()
assert result is None
def test_get_ocr_options_generic_error_returns_none(self):
"""Cover lines 151-153: generic Exception returns None."""
from application.parser.file.docling_parser import DoclingParser
parser = DoclingParser(ocr_enabled=True, use_rapidocr=True)
with patch(
"builtins.__import__",
side_effect=RuntimeError("unexpected"),
):
result = parser._get_ocr_options()
assert result is None
def test_csv_parser_init(self):
"""Cover line 289: DoclingCSVParser.__init__ calls super."""
from application.parser.file.docling_parser import DoclingCSVParser
@@ -1442,11 +1542,11 @@ class TestForceFullPageOCRWiring:
DoclingParser(**kwargs)._create_converter()
return built[0]
def test_forced_without_rapidocr(self, monkeypatch):
def test_forced_with_default_auto_options(self, monkeypatch):
options = self._pipeline_options(
monkeypatch,
ocr_enabled=True,
use_rapidocr=False,
ocr_engine="auto",
force_full_page_ocr=True,
)
assert options.ocr_options.force_full_page_ocr is True
@@ -1455,7 +1555,7 @@ class TestForceFullPageOCRWiring:
options = self._pipeline_options(
monkeypatch,
ocr_enabled=True,
use_rapidocr=False,
ocr_engine="auto",
force_full_page_ocr=False,
)
assert options.ocr_options.force_full_page_ocr is False
+148
View File
@@ -0,0 +1,148 @@
"""Tests for the PDF trust checks behind ``PDF_TRUST_CHECK``.
The checks target the two classes where anydoc drops text *silently*:
composite (Type0) fonts without a ToUnicode map, and CJK-declaring PDFs
whose extracted text carries almost no CJK. The scanner is byte-level and
regex-based, so the synthetic fixtures below are minimal PDF fragments, not
well-formed files; the one real fixture is the corpus document that loses
its whole Chinese column in anydoc with no error.
"""
import zlib
from pathlib import Path
from application.parser.file.pdf_trust import (
check_pdf_fonts,
verify_extraction,
verify_pdf_file,
)
FIXTURES = Path(__file__).parent / "fixtures"
def _obj(body: bytes) -> bytes:
return b"1 0 obj" + body + b"endobj\n"
def _stream(payload: bytes) -> bytes:
return b"2 0 obj<</Filter/FlateDecode>>stream\n" + zlib.compress(payload) + b"\nendstream endobj\n"
TYPE0_WITH_TOUNICODE = _obj(
b"<</Type/Font/Subtype/Type0/BaseFont/AAAAAA+Simple/ToUnicode 9 0 R>>"
)
TYPE0_BARE = _obj(b"<</Type/Font/Subtype/Type0/BaseFont/AAAAAA+Bare>>")
SIMPLE_FONT = _obj(b"<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>")
GB1_ORDERING = _obj(b"<</Registry(Adobe)/Ordering(GB1)/Supplement 5>>")
class TestCheckPdfFonts:
def test_type0_with_tounicode_not_flagged(self):
result = check_pdf_fonts(TYPE0_WITH_TOUNICODE)
assert result["type0"] == 1
assert result["type0_no_tounicode"] == 0
assert not result["flagged"]
def test_type0_without_tounicode_flagged(self):
result = check_pdf_fonts(TYPE0_BARE)
assert result["type0_no_tounicode"] == 1
assert result["flagged"]
def test_simple_font_not_flagged(self):
result = check_pdf_fonts(SIMPLE_FONT)
assert result["type0"] == 0
assert result["has_fonts"]
assert not result["flagged"]
def test_no_fonts_flagged(self):
result = check_pdf_fonts(_obj(b"<</Type/Page>>"))
assert not result["has_fonts"]
assert result["flagged"]
def test_type0_inside_object_stream_is_seen(self):
"""Object streams hold dicts with no obj markers; the scan must decompress them."""
data = _stream(b"<</Type/Font/Subtype/Type0/BaseFont/BBBBBB+Packed>>")
result = check_pdf_fonts(data)
assert result["type0"] == 1
assert result["type0_no_tounicode"] == 1
assert result["flagged"]
def test_font_inside_object_stream_counts_as_has_fonts(self):
"""A fully-compressed PDF with only simple fonts must not false-positive."""
data = _stream(b"<</Type/Font/Subtype/TrueType/BaseFont/Arial>>")
result = check_pdf_fonts(data)
assert result["has_fonts"]
assert not result["flagged"]
def test_cjk_ordering_detected_in_stream(self):
data = _stream(b"<</Registry(Adobe)/Ordering(CNS1)/Supplement 4>>") + SIMPLE_FONT
assert check_pdf_fonts(data)["expects_cjk"]
def test_cjk_font_names_alone_do_not_set_expectation(self):
"""Regression for the Berkshire false positive: Word-exported Latin PDFs
embed MS-Gothic subsets for stray full-width characters, and substring
matching also hits FranklinGothic — neither means CJK *content*, and a
false expectation reroutes a 150-page English report to the heavy
engine. Only CIDSystemInfo /Ordering may set the expectation."""
data = _obj(
b"<</Type/Font/Subtype/TrueType/BaseFont/AKBHWD+MS-Gothic>>"
) + _obj(b"<</Type/Font/Subtype/TrueType/BaseFont/KMBGKH+FranklinGothic-Roman>>")
result = check_pdf_fonts(data)
assert not result["expects_cjk"]
assert not result["flagged"]
class TestVerifyExtraction:
def test_clean_pdf_and_output(self):
assert verify_extraction(SIMPLE_FONT, "Plain English text.") == []
def test_bare_type0_reported(self):
problems = verify_extraction(TYPE0_BARE + SIMPLE_FONT, "some text")
assert len(problems) == 1
assert "ToUnicode" in problems[0]
def test_cjk_expected_but_missing(self):
problems = verify_extraction(GB1_ORDERING + SIMPLE_FONT, "English only output")
assert len(problems) == 1
assert "CJK" in problems[0]
def test_cjk_expected_and_present(self):
text = "标题:本协议由双方共同签署生效" # 14 CJK chars
assert verify_extraction(GB1_ORDERING + SIMPLE_FONT, text) == []
def test_both_problems_reported(self):
problems = verify_extraction(TYPE0_BARE + GB1_ORDERING, "english")
assert len(problems) == 2
class TestRealFixture:
"""The corpus PDF where anydoc silently drops the entire Chinese column."""
def test_cid_font_nda_is_flagged(self):
data = (FIXTURES / "nda_en_zh_cid_font.pdf").read_bytes()
result = check_pdf_fonts(data)
assert result["type0_no_tounicode"] >= 1
assert result["expects_cjk"]
assert result["flagged"]
def test_cjkless_extraction_reports_both(self):
data = (FIXTURES / "nda_en_zh_cid_font.pdf").read_bytes()
problems = verify_extraction(data, "MUTUAL NON-DISCLOSURE AGREEMENT ...")
assert any("ToUnicode" in p for p in problems)
assert any("CJK" in p for p in problems)
def test_flate_bomb_streams_are_capped():
"""Flate reaches ~1000:1, so per-stream inflation must be capped — an
uncapped decompress of a crafted PDF would OOM the ingest worker."""
from application.parser.file.pdf_trust import _decompressed_streams, _STREAM_INFLATE_CAP
bomb = _stream(b"\0" * (_STREAM_INFLATE_CAP * 4))
chunks = list(_decompressed_streams(bomb))
assert chunks
assert all(len(chunk) <= _STREAM_INFLATE_CAP for chunk in chunks)
# The capped scan still completes and reports sanely alongside real objects.
assert not check_pdf_fonts(SIMPLE_FONT + bomb)["flagged"]
def test_verify_pdf_file_unreadable_trusts_output(tmp_path):
assert verify_pdf_file(tmp_path / "missing.pdf", "text") == []
+74
View File
@@ -0,0 +1,74 @@
"""Tests for the dot-leader/whitespace table reconstruction (``ANYDOC_TABLEIZE``)."""
from application.parser.file.tableize import tableize
DOT_LEADER = """Revenues
Insurance premiums ............ 83,431 77,731
Sales and service revenues ........ 24,660 23,406
Freight rail transportation ......... 22,341 23,852
Total revenues 130,432 124,989
"""
def test_dot_leader_run_becomes_table():
out = tableize(DOT_LEADER)
assert "| Insurance premiums | 83,431 | 77,731 |" in out
assert "| Total revenues | 130,432 | 124,989 |" in out
assert "| --- |" in out
assert "Revenues" in out # heading line untouched
def test_whitespace_only_rows_convert_too():
md = "Alpha 1 2\nBeta 3 4\nGamma 5 6\n"
out = tableize(md)
assert "| Alpha | 1 | 2 |" in out
def test_currency_symbol_merges_with_number():
md = "Cash $ 1,234 900\nDebt $ 2,000 1,500\nEquity $ 900 800\n"
out = tableize(md)
assert "| Cash | $1,234 | 900 |" in out
def test_parenthesised_negatives_and_dash():
md = "Losses (30) —\nGains (12) —\nNet (42) —\n"
out = tableize(md)
assert "| Losses | (30) | — |" in out
def test_short_run_is_left_alone():
md = "Alpha 1 2\nBeta 3 4\n"
assert "|" not in tableize(md)
def test_mixed_widths_are_left_alone():
md = "Alpha 1 2\nBeta 3\nGamma 5 6\n"
assert "|" not in tableize(md)
def test_prose_with_numbers_is_left_alone():
md = "The division shipped 47 releases in 2023.\nIt hired 12 engineers this year alone.\nRevenue grew by a factor of 3 since 2019.\n"
assert "|" not in tableize(md)
def test_min_rows_boundary():
two = "A 1 2\nB 3 4\n"
three = two + "C 5 6\n"
assert "|" not in tableize(two, min_rows=3)
assert "|" in tableize(three, min_rows=3)
def test_pipe_in_label_is_escaped():
md = "Assets | current 1,234 900\nDebt 2,000 1,500\nEquity 900 800\n"
out = tableize(md)
assert "| Assets \\| current | 1,234 | 900 |" in out
def test_dollar_prefixed_word_is_not_a_value():
md = "Alpha $TBD 2\nBeta $TBD 4\nGamma $TBD 6\n"
assert "|" not in tableize(md)
def test_non_table_text_passes_through_verbatim():
"""Only the trailing newline may differ (splitlines/join round-trip)."""
md = "# Heading\n\nA paragraph with no numbers.\n\n- a list item\n"
assert tableize(md) == md.rstrip("\n")