From ed0892b39bbdca438d7afafa2ccbb7b5ad2b42f7 Mon Sep 17 00:00:00 2001 From: Pavel Date: Thu, 3 Sep 2026 00:30:59 +0400 Subject: [PATCH] Batch fixes 2 --- application/parser/document_reader.py | 44 +++++++- application/parser/file/base_parser.py | 5 + application/parser/file/bulk.py | 16 ++- application/parser/file/docling_parser.py | 20 ++++ application/parser/file/html_parser.py | 26 ++++- application/parser/file/ocr_parser.py | 112 ++++++++++++++++-- application/parser/file/tableize.py | 13 ++- application/parser/remote/s3_loader.py | 113 ++++++++++--------- application/requirements-docling.txt | 2 +- deployment/docker-compose-azure.yaml | 12 +- docs/content/Deploying/Docker-Deploying.mdx | 2 +- docs/content/Deploying/DocsGPT-Settings.mdx | 2 +- docs/content/Guides/ocr.mdx | 31 ++--- frontend/src/locale/de.json | 2 +- frontend/src/locale/en.json | 2 +- frontend/src/locale/es.json | 2 +- frontend/src/locale/jp.json | 2 +- frontend/src/locale/ru.json | 2 +- frontend/src/locale/zh-TW.json | 2 +- frontend/src/locale/zh.json | 2 +- setup.ps1 | 13 ++- setup.sh | 2 +- tests/parser/file/test_anydoc_parser.py | 3 +- tests/parser/file/test_bulk.py | 13 +++ tests/parser/file/test_docling_parser.py | 45 +++++++- tests/parser/file/test_html_parser.py | 47 +++++++- tests/parser/file/test_ocr_parser.py | 119 +++++++++++++++++++- tests/parser/file/test_tableize.py | 20 ++++ tests/parser/remote/test_s3_loader.py | 11 ++ tests/parser/test_document_reader.py | 46 ++++++++ 30 files changed, 619 insertions(+), 112 deletions(-) diff --git a/application/parser/document_reader.py b/application/parser/document_reader.py index 450a8400..7ae4554c 100644 --- a/application/parser/document_reader.py +++ b/application/parser/document_reader.py @@ -196,18 +196,27 @@ def _pick_parser(suffix: str, *, ocr_enabled: bool, engine: str): def _legacy_parser_for(suffix: str): """Return a non-Docling parser for ``suffix`` (the ``fast`` engine), or None.""" from application.parser.file.docs_parser import DocxParser, PDFParser + from application.parser.file.epub_parser import EpubParser from application.parser.file.html_parser import HTMLParser + from application.parser.file.json_parser import JSONParser from application.parser.file.markdown_parser import MarkdownParser + from application.parser.file.pptx_parser import PPTXParser + from application.parser.file.rst_parser import RstParser from application.parser.file.tabular_parser import ExcelParser, PandasCSVParser legacy = { ".pdf": PDFParser, ".docx": DocxParser, + ".pptx": PPTXParser, ".csv": PandasCSVParser, ".xlsx": ExcelParser, ".html": HTMLParser, + ".xhtml": HTMLParser, + ".epub": EpubParser, ".md": MarkdownParser, ".mdx": MarkdownParser, + ".rst": RstParser, + ".json": JSONParser, } cls = legacy.get(suffix) return cls() if cls is not None else None @@ -223,6 +232,17 @@ def _parse_to_text(parser: Any, path: Path) -> str: return str(parsed) +def _is_native_ocr_parser(parser: Any) -> bool: + """True for the native OCR PDF parser (OCR_BACKEND=native): a docling table pass would OCR the scan again.""" + if parser is None: + return False + try: + from application.parser.file.ocr_parser import NativeOcrPdfParser + except Exception: + return False + return isinstance(parser, NativeOcrPdfParser) + + def _is_docling_parser(parser: Any) -> bool: """True when ``parser`` is Docling-backed (collecting tables would otherwise re-convert).""" if parser is None: @@ -510,10 +530,17 @@ def _shape( parser = _pick_parser(suffix, ocr_enabled=ocr_enabled, engine=engine) # Tables come from a Docling conversion, so they are only collected when - # Docling is the engine in play: under anydoc/fast a table pass would be - # a second, full docling conversion that costs more than the whole read. + # a Docling parser is actually in play: under anydoc/fast a table pass + # would be a second, full docling conversion that costs more than the + # whole read, and under OCR_BACKEND=native the PDF parser is the native + # OCR parser (docling only as its text_parser), where a vanilla + # DocumentConverter pass would OCR the scan a second time with docling's + # default engine, ignoring OCR_ENGINE. wants_tables = ( - include_tables and output != "chunks" and _effective_engine(engine) == "docling" + include_tables + and output != "chunks" + and _effective_engine(engine) == "docling" + and not _is_native_ocr_parser(parser) ) # A Docling-backed parser already converts the whole document to produce its text. @@ -536,7 +563,16 @@ def _shape( if parser is None: # A whitelisted extension with no dedicated parser (e.g. .txt) reads as plain - # text, matching SimpleDirectoryReader's standard-read fallback. + # text, matching SimpleDirectoryReader's standard-read fallback. Binary + # office formats that only anydoc reads must not: without anydoc they + # would come back as OLE/zip bytes decoded as text. + from application.parser.file.anydoc_parser import ANYDOC_GAINED_SUFFIXES + + if suffix in ANYDOC_GAINED_SUFFIXES: + return { + "error": f"No parser is available for {suffix} files: firecrawl-anydoc " + "is not installed (pip install firecrawl-anydoc)" + } text = path.read_text(errors="ignore") else: text = _parse_to_text(parser, path) diff --git a/application/parser/file/base_parser.py b/application/parser/file/base_parser.py index d0ae5ae4..7b8907d1 100644 --- a/application/parser/file/base_parser.py +++ b/application/parser/file/base_parser.py @@ -117,6 +117,11 @@ def delegate_parse( return parser.parse_file(file, errors) except DocumentParseError: raise + except ImportError: + # A fallback whose dependency is missing is a deployment problem, not + # a bad document: let it reach the ingest task's setup-error path + # instead of blaming every file with "could not be read". + raise except Exception as e: logger.error( f"Fallback parse of {Path(file).name} with " diff --git a/application/parser/file/bulk.py b/application/parser/file/bulk.py index 56ddfe4e..2630f122 100644 --- a/application/parser/file/bulk.py +++ b/application/parser/file/bulk.py @@ -53,9 +53,10 @@ def _gained_format_entries() -> Dict[str, BaseParser]: """Anydoc-only formats (legacy/macro Office, OpenDocument, RTF) for every map. anydoc is a core dependency, so these suffixes are parseable under both - engines. When anydoc is somehow missing the entries are omitted, and such - files fall to ``SimpleDirectoryReader``'s plain-text read — the - pre-existing behaviour for unmapped suffixes. + engines. When anydoc is somehow missing the entries are omitted and + ``SimpleDirectoryReader.load_data`` rejects such files with a + ``DocumentParseError`` naming the install, rather than reading OLE/zip + bytes as text. """ from application.parser.file.anydoc_parser import ( ANYDOC_GAINED_SUFFIXES, @@ -511,6 +512,15 @@ class SimpleDirectoryReader(BaseReader): data = parser.parse_file(input_file, errors=self.errors) parser_metadata = parser.get_file_metadata(input_file) else: + from application.parser.file.anydoc_parser import ANYDOC_GAINED_SUFFIXES + + if suffix_lower in ANYDOC_GAINED_SUFFIXES: + # Binary office formats only anydoc reads; without it + # the standard read would index OLE/zip bytes as text. + raise DocumentParseError( + f"No parser is available for {suffix_lower} files: " + "firecrawl-anydoc is not installed (pip install firecrawl-anydoc)" + ) # do standard read with open(input_file, "r", errors=self.errors) as f: data = f.read() diff --git a/application/parser/file/docling_parser.py b/application/parser/file/docling_parser.py index 365e24a4..8b157c5e 100644 --- a/application/parser/file/docling_parser.py +++ b/application/parser/file/docling_parser.py @@ -148,11 +148,31 @@ def _build_ocr_options( if engine == "tesseract": from docling.datamodel.pipeline_options import TesseractCliOcrOptions + from application.parser.file.ocr_parser import tesseract_languages + langs = ( languages or [lang.strip() for lang in settings.OCR_LANGS.split("+") if lang.strip()] or ["eng"] ) + # docling swallows tesseract's per-region failures: a language list + # with no installed pack logs an error per region and exports '' + # which the dropout guard then indexes as "text-sparse". Drop the + # packs that are not installed up front and say so. + installed = tesseract_languages() + if installed is not None: + missing = [lang for lang in langs if lang not in installed] + if missing: + kept = [lang for lang in langs if lang in installed] + if not kept and "eng" in installed: + kept = ["eng"] + logger.warning( + "OCR_LANGS lists tesseract language pack(s) that are not installed: %s; " + "OCR-ing with %s (install tesseract-ocr- for each missing pack)", + "+".join(missing), + "+".join(kept) if kept else "the requested list anyway", + ) + langs = kept or langs return TesseractCliOcrOptions( lang=langs, force_full_page_ocr=force_full_page_ocr ) diff --git a/application/parser/file/html_parser.py b/application/parser/file/html_parser.py index 37cd1a0a..d2a21a01 100644 --- a/application/parser/file/html_parser.py +++ b/application/parser/file/html_parser.py @@ -25,6 +25,10 @@ MARKDOWNIFY_OPTIONS = {"heading_style": "ATX", "newline_style": "BACKSLASH"} # Elements whose text is never document content. ``title`` is reported via # ``get_file_metadata`` instead of leaking in as a stray first line. _DROP_TAGS = ("title", "script", "style", "noscript", "template") +# Attributes whose ``data:`` URIs would otherwise land in the Markdown as +# link/image targets: one inline image is megabytes of base64 "text" for the +# chunker and embedder (a 5 MB data-URI ```` measured 7 MB of output). +_URI_ATTRIBUTES = ("src", "href", "srcset", "poster", "data") def html_to_markdown(html: Union[str, bytes]) -> str: @@ -56,10 +60,22 @@ def soup_to_markdown(soup) -> str: Returns: Markdown with runs of blank lines collapsed to one. """ + from bs4 import CData, Declaration, ProcessingInstruction from markdownify import MarkdownConverter for tag in soup.find_all(_DROP_TAGS): tag.decompose() + # ```` (every XHTML file), CDATA and stray declarations are + # not text; markdownify would emit them as the document's first line. + for node in soup.find_all(string=lambda s: isinstance(s, (ProcessingInstruction, CData, Declaration))): + node.extract() + for tag in soup.find_all(True): + for attribute in _URI_ATTRIBUTES: + value = tag.get(attribute) + if isinstance(value, list): + value = " ".join(value) + if isinstance(value, str) and "data:" in value.lower(): + del tag[attribute] markdown = MarkdownConverter(**MARKDOWNIFY_OPTIONS).convert_soup(soup) return re.sub(r"\n{3,}", "\n\n", markdown).strip() @@ -130,7 +146,15 @@ def read_markup_head(file: Path, max_bytes: int) -> bytes: f"Markup {Path(file).name} exceeds MARKUP_MAX_BYTES ({max_bytes}); " f"parsing the first {max_bytes} bytes to bound memory" ) - return _trim_torn_utf8_tail(truncate_to_line_boundary(head[:max_bytes])) + head = head[:max_bytes] + if head[:2] in (b"\xff\xfe", b"\xfe\xff"): + # UTF-16 (BOM-declared): a byte-level line cut lands between the two + # bytes of a code unit and the strict decode then fails outright, so + # keep an even byte count and let the lenient parser take the torn + # tail. (Only 2-byte units matter here; a torn surrogate pair is one + # lost character.) + return head[: len(head) - (len(head) % 2)] + return _trim_torn_utf8_tail(truncate_to_line_boundary(head)) class HTMLParser(BaseParser): diff --git a/application/parser/file/ocr_parser.py b/application/parser/file/ocr_parser.py index 331d9b27..784a1493 100644 --- a/application/parser/file/ocr_parser.py +++ b/application/parser/file/ocr_parser.py @@ -23,6 +23,7 @@ tesseract binary alone, and one that does install it keeps today's behaviour unchanged. """ import base64 +import functools import io import logging import re @@ -63,6 +64,11 @@ _MIN_RENDER_DPI, _MAX_RENDER_DPI = 72, 600 # usable ~125 dpi; anything larger renders at the scale that fits. _MAX_RENDER_PIXELS = 40_000_000 _TESSERACT_TIMEOUT_SECONDS = 300 +# tesseract reports a missing language pack on stderr and, when at least one +# other requested pack loads, exits 0 and silently OCRs with what it has — +# so the exit code alone cannot catch OCR_LANGS=eng+chi_sim without chi_sim. +_TESSERACT_LANG_ERROR_RE = re.compile(r"Error opening data file|Failed loading language") +_TESSERACT_LANG_RE = re.compile(r"^[A-Za-z0-9_/\-]+$") # docling's DeepSeek-OCR prompt minus its ``<|grounding|>`` prefix: grounding # makes the model wrap every element in ref/det tags with bounding boxes, # which docling parses back into a layout tree. Plain Markdown is what the @@ -91,7 +97,6 @@ class OcrEngine(Protocol): def ocr_image(self, image) -> str: # pragma: no cover - protocol """Return the text (or Markdown) recognised in ``image`` (a PIL image).""" - ... # --------------------------------------------------------------------------- @@ -181,15 +186,80 @@ def render_dpi() -> int: return max(_MIN_RENDER_DPI, min(_MAX_RENDER_DPI, dpi)) +def _has_alpha(image) -> bool: + return "A" in image.getbands() or (image.mode == "P" and "transparency" in image.info) + + def _png_bytes(image) -> bytes: - """Encode a PIL image as PNG, flattening modes the engines cannot take.""" - if image.mode not in ("RGB", "L"): + """Encode a PIL image as PNG, flattening modes the engines cannot take. + + Transparency is composited onto white: a plain ``convert("RGB")`` drops + the alpha channel and leaves transparent pixels black, which turns dark + text on a transparent background into an all-black page that OCRs to + nothing. + """ + from PIL import Image + + if _has_alpha(image): + rgba = image.convert("RGBA") + canvas = Image.new("RGB", rgba.size, "white") + canvas.paste(rgba, mask=rgba.getchannel("A")) + image = canvas + elif image.mode not in ("RGB", "L"): image = image.convert("RGB") buffer = io.BytesIO() image.save(buffer, format="PNG") return buffer.getvalue() +def fit_to_pixel_budget(image): + """Downscale a decoded image to ``_MAX_RENDER_PIXELS`` before OCR. + + The same budget the PDF renderer applies: a 28 KB PNG can decode to + 81 MP (Pillow's bomb guard only trips at ~178 MP), and every extra + pixel is paid again in the PNG re-encode and inside tesseract. + """ + from PIL import Image + + width, height = image.size + scale = render_scale(width, height, 72) + if scale >= 1.0: + return image + if image.mode not in ("RGB", "L", "RGBA", "LA"): + image = image.convert("RGBA" if _has_alpha(image) else "RGB") + logger.warning( + "Image is %dx%d px; downscaling to %.0f%% to stay within %d MP before OCR", + width, + height, + scale * 100, + _MAX_RENDER_PIXELS // 1_000_000, + ) + return image.resize( + (max(1, int(width * scale)), max(1, int(height * scale))), Image.Resampling.BILINEAR + ) + + +@functools.lru_cache(maxsize=1) +def tesseract_languages() -> Optional[frozenset]: + """Language packs the ``tesseract`` binary reports (``--list-langs``), or None when unknown. + + Cached for the process: packs are installed with the image, not at runtime. + """ + if shutil.which("tesseract") is None: + return None + try: + completed = subprocess.run( + ["tesseract", "--list-langs"], capture_output=True, timeout=30, check=False + ) + except (OSError, subprocess.TimeoutExpired): + return None + output = completed.stdout.decode("utf-8", "replace") + "\n" + completed.stderr.decode("utf-8", "replace") + langs = frozenset( + line.strip() for line in output.splitlines() if _TESSERACT_LANG_RE.match(line.strip()) + ) + return langs or None + + # --------------------------------------------------------------------------- # Engines # --------------------------------------------------------------------------- @@ -261,9 +331,17 @@ class TesseractEngine: raise DocumentParseError(f"tesseract timed out after {self.timeout:.0f}s on a page") from exc except OSError as exc: raise DocumentParseError(f"tesseract could not be started: {exc}") from exc + stderr = completed.stderr.decode("utf-8", "replace").strip() + if _TESSERACT_LANG_ERROR_RE.search(stderr): + # Exit code 0 here means tesseract dropped the missing pack and + # OCR'd with the rest: Chinese scans would come out as garbage + # with no signal. Fail the file and name the fix instead. + raise OcrUnavailableError( + f"tesseract could not load a language pack for OCR_LANGS={self._language_arg()!r}: " + f"{stderr[:300]}. Install the tesseract-ocr- package for every language listed." + ) if completed.returncode != 0: - detail = completed.stderr.decode("utf-8", "replace").strip()[:300] - raise DocumentParseError(f"tesseract failed (exit {completed.returncode}): {detail}") + raise DocumentParseError(f"tesseract failed (exit {completed.returncode}): {stderr[:300]}") return collapse_cjk_spaces(completed.stdout.decode("utf-8", "replace")).strip() @@ -343,13 +421,15 @@ class DeepseekOcrEngine: response = requests.post(self.url, json=self.payload(image), timeout=self.timeout) response.raise_for_status() body = response.json() + except requests.exceptions.JSONDecodeError as exc: + # Subclasses RequestException too, so it must be caught first or a + # proxy's HTML error page is reported as a connection failure. + raise DocumentParseError(f"DeepSeek-OCR endpoint {self.url} returned a non-JSON body") from exc except requests.RequestException as exc: raise DocumentParseError( f"DeepSeek-OCR request to {self.url} failed: {exc}. Check OCR_DEEPSEEK_URL " f"and that model {self.model!r} is served (e.g. `ollama pull {self.model}`)." ) from exc - except ValueError as exc: - raise DocumentParseError(f"DeepSeek-OCR endpoint {self.url} returned a non-JSON body") from exc try: content = body["choices"][0]["message"]["content"] except (KeyError, IndexError, TypeError) as exc: @@ -433,7 +513,10 @@ def _page_has_image(page) -> bool: import pypdfium2.raw as pdfium_c try: - return next(iter(page.get_objects(filter=(pdfium_c.FPDF_PAGEOBJ_IMAGE,))), None) is not None + # pypdfium2's default max_depth=2 misses images inside nested Form + # XObjects, which print drivers and InDesign produce routinely. + objects = page.get_objects(filter=(pdfium_c.FPDF_PAGEOBJ_IMAGE,), max_depth=16) + return next(iter(objects), None) is not None except Exception: # noqa: BLE001 - treat an unreadable page as image-bearing return True @@ -663,7 +746,7 @@ class NativeOcrPdfParser(BaseParser): if counts[index] >= self.min_text_chars: textpage = page.get_textpage() try: - text = textpage.get_text_range() + text = textpage.get_text_bounded() finally: textpage.close() else: @@ -723,6 +806,12 @@ class NativeOcrImageParser(BaseParser): def parse_file(self, file: Path, errors: str = "ignore") -> Union[str, List[str]]: """OCR an image file. + Only TIFF frames are pages (multi-page faxes and scans); every other + multi-frame format is an animation (GIF, WebP, APNG) whose frames + repeat one picture, so only the first is OCR'd — a 50-frame WebP + would otherwise cost 50 tesseract runs and then be rejected as an + empty multi-page document. + Raises: DocumentParseError: The image cannot be decoded, the engine failed, or a multi-frame image OCR'd to nothing. @@ -737,8 +826,9 @@ class NativeOcrImageParser(BaseParser): texts: List[str] = [] try: with Image.open(path) as image: - for frame in ImageSequence.Iterator(image): - texts.append((engine.ocr_image(frame.convert("RGB")) or "").strip()) + frames = ImageSequence.Iterator(image) if path.suffix.lower() in (".tif", ".tiff") else [image] + for frame in frames: + texts.append((engine.ocr_image(fit_to_pixel_budget(frame)) or "").strip()) except DocumentParseError: raise except Exception as exc: diff --git a/application/parser/file/tableize.py b/application/parser/file/tableize.py index 9fd9d675..7c764a1f 100644 --- a/application/parser/file/tableize.py +++ b/application/parser/file/tableize.py @@ -37,14 +37,23 @@ def _merge_currency(tokens: List[str]) -> List[str]: def _parse_row(line: str) -> Optional[Tuple[str, List[str]]]: - """``(label, values)`` when the line looks like a typographic table row, else None.""" - match = _LEADER.match(line) or _TRAILING.match(line) + """``(label, values)`` when the line looks like a typographic table row, else None. + + Without dot leaders a row needs at least two numeric columns: a single + trailing number is what "Chapter 1", "Footnote 2", "ISO 9001" and + "Version 3.0" look like, and three of those in a row are a list, not a + table. + """ + leader = _LEADER.match(line) + match = leader or _TRAILING.match(line) if not match: return None label, rest = match.group(1).strip(" ."), match.group(2) values = _merge_currency(rest.split()) if not values or not all(_NUM_TOKEN.fullmatch(v.lstrip("$")) for v in values): return None + if leader is None and len(values) < 2: + return None if not label or _NUM_TOKEN.fullmatch(label): return None return label, values diff --git a/application/parser/remote/s3_loader.py b/application/parser/remote/s3_loader.py index cc670f62..8b2bafdd 100644 --- a/application/parser/remote/s3_loader.py +++ b/application/parser/remote/s3_loader.py @@ -5,6 +5,7 @@ import tempfile import mimetypes from typing import List, Optional from application.core.url_validation import SSRFError, validate_url +from application.parser.file.constants import SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS from application.parser.remote.base import BaseRemote from application.parser.schema.base import Document @@ -17,6 +18,61 @@ except ImportError: logger = logging.getLogger(__name__) +# Plain-text suffixes the loader reads directly (no document parser). +_TEXT_EXTENSIONS = { + ".txt", + ".md", + ".markdown", + ".rst", + ".json", + ".xml", + ".yaml", + ".yml", + ".py", + ".js", + ".ts", + ".jsx", + ".tsx", + ".java", + ".c", + ".cpp", + ".h", + ".hpp", + ".cs", + ".go", + ".rs", + ".rb", + ".php", + ".swift", + ".kt", + ".scala", + ".html", + ".css", + ".scss", + ".sass", + ".less", + ".sh", + ".bash", + ".zsh", + ".fish", + ".sql", + ".r", + ".m", + ".mat", + ".ini", + ".cfg", + ".conf", + ".config", + ".env", + ".gitignore", + ".dockerignore", + ".editorconfig", + ".log", + ".csv", + ".tsv", + } + + class S3Loader(BaseRemote): """Load documents from an AWS S3 bucket.""" @@ -128,58 +184,7 @@ class S3Loader(BaseRemote): def is_text_file(self, file_path: str) -> bool: """Determine if a file is a text file based on extension.""" - text_extensions = { - ".txt", - ".md", - ".markdown", - ".rst", - ".json", - ".xml", - ".yaml", - ".yml", - ".py", - ".js", - ".ts", - ".jsx", - ".tsx", - ".java", - ".c", - ".cpp", - ".h", - ".hpp", - ".cs", - ".go", - ".rs", - ".rb", - ".php", - ".swift", - ".kt", - ".scala", - ".html", - ".css", - ".scss", - ".sass", - ".less", - ".sh", - ".bash", - ".zsh", - ".fish", - ".sql", - ".r", - ".m", - ".mat", - ".ini", - ".cfg", - ".conf", - ".config", - ".env", - ".gitignore", - ".dockerignore", - ".editorconfig", - ".log", - ".csv", - ".tsv", - } + text_extensions = _TEXT_EXTENSIONS file_lower = file_path.lower() for ext in text_extensions: @@ -197,7 +202,9 @@ class S3Loader(BaseRemote): def is_supported_document(self, file_path: str) -> bool: """Check if file is a supported document type for parsing.""" - document_extensions = { + # The upload whitelist plus the historical S3 list, so an object the + # upload API accepts is never silently skipped here. + document_extensions = (set(SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS) - _TEXT_EXTENSIONS) | { ".pdf", ".docx", ".doc", diff --git a/application/requirements-docling.txt b/application/requirements-docling.txt index 1241187b..1514483f 100644 --- a/application/requirements-docling.txt +++ b/application/requirements-docling.txt @@ -2,7 +2,7 @@ # # Needed only for: DOC_PARSER_ENGINE=docling, the docling OCR backend # (OCR_BACKEND=docling: layout-model hybrid OCR, ocrmac/rapidocr engines), -# .adoc/.vtt/.xml source parsing, and the read_document tool's `structured` +# .adoc/.vtt/.xml attachment parsing, and the read_document tool's `structured` # output. The default anydoc engine needs none of this, and OCR itself does # not either: OCR_ENABLED=true with the tesseract binary (or a DeepSeek-OCR # endpoint) runs through application/parser/file/ocr_parser.py. diff --git a/deployment/docker-compose-azure.yaml b/deployment/docker-compose-azure.yaml index 3295a12c..d991009a 100644 --- a/deployment/docker-compose-azure.yaml +++ b/deployment/docker-compose-azure.yaml @@ -10,7 +10,11 @@ services: - backend backend: - build: ../application + build: + context: ../application + args: + INSTALL_DOCLING: ${INSTALL_DOCLING:-false} + INSTALL_TESSERACT: ${INSTALL_TESSERACT:-true} env_file: - ../.env environment: @@ -31,7 +35,11 @@ services: condition: service_healthy worker: - build: ../application + build: + context: ../application + args: + INSTALL_DOCLING: ${INSTALL_DOCLING:-false} + INSTALL_TESSERACT: ${INSTALL_TESSERACT:-true} # `parsing` queue carries read_document/parse_document; required for its await to resolve. command: celery -A application.app.celery worker -l INFO -Q docsgpt,parsing env_file: diff --git a/docs/content/Deploying/Docker-Deploying.mdx b/docs/content/Deploying/Docker-Deploying.mdx index 382db5fb..8cec05e0 100644 --- a/docs/content/Deploying/Docker-Deploying.mdx +++ b/docs/content/Deploying/Docker-Deploying.mdx @@ -48,7 +48,7 @@ The fastest way to try out DocsGPT is by using the public API endpoint. This req Navigate to the root directory of the DocsGPT repository in your terminal and run: ```bash - docker compose -f deployment/docker-compose.yaml up -d + docker compose --env-file .env -f deployment/docker-compose.yaml up -d ``` The `-d` flag runs Docker Compose in detached mode (in the background). diff --git a/docs/content/Deploying/DocsGPT-Settings.mdx b/docs/content/Deploying/DocsGPT-Settings.mdx index c6f3b03c..d51d128d 100644 --- a/docs/content/Deploying/DocsGPT-Settings.mdx +++ b/docs/content/Deploying/DocsGPT-Settings.mdx @@ -222,7 +222,7 @@ for the engines and flows. | `OCR_DEEPSEEK_MODEL` | `deepseek-ocr:3b` | Model name at that endpoint. | | `OCR_DEEPSEEK_TIMEOUT` | `300` | Seconds allowed per page request to the DeepSeek endpoint, on both backends (the native backend sends pages one at a time). | | `OCR_RENDER_DPI` | `200` | Native backend: resolution at which pages without a text layer are rendered before OCR (clamped to 72-600). | -| `OCR_MIN_CHARS_PER_PAGE` | `20` | Chars-per-page floor below which an OCR'd parse is treated as an OCR dropout and fails loudly instead of indexing an empty document; `0` disables. Alias: `DOCLING_OCR_MIN_CHARS_PER_PAGE`. | +| `OCR_MIN_CHARS_PER_PAGE` | `20` | Chars-per-page floor for the OCR dropout guard: a multi-page parse whose OCR returns nothing fails loudly instead of indexing an empty document (docling retries once on full-page OCR first); output below the floor but non-empty is indexed with a warning. `0` disables. Alias: `DOCLING_OCR_MIN_CHARS_PER_PAGE`. | ## Speech-to-Text Settings diff --git a/docs/content/Guides/ocr.mdx b/docs/content/Guides/ocr.mdx index 90140242..9c0a6b44 100644 --- a/docs/content/Guides/ocr.mdx +++ b/docs/content/Guides/ocr.mdx @@ -19,12 +19,12 @@ DOC_PARSER_ENGINE=anydoc - `anydoc` (default): [firecrawl-anydoc](https://github.com/firecrawl/anydoc), a Rust converter with no ML models. It reads PDF, DOCX, PPTX, XLSX and CSV in milliseconds with ~100 MB peak memory; HTML/XHTML is converted with - `markdownify`, head-truncated at `MARKUP_MAX_BYTES`. Source uploads and the - `read_document` tool accept the same file types as before (the - `SUPPORTED_SOURCE_EXTENSIONS` whitelist is unchanged); chat attachments, - which are not whitelisted by extension, can additionally be parsed from - anydoc's other formats (DOC, PPT/PPS/POT, XLS, ODT/ODS/ODP, RTF and the - macro-enabled Office variants). It never performs OCR: + `markdownify`, head-truncated at `MARKUP_MAX_BYTES`. Because anydoc is a + core dependency, source uploads and the `read_document` tool now also + accept its other formats — DOC, PPT/PPS/POT, XLS, ODT/ODS/ODP, RTF, XHTML + and the macro-enabled Office variants — on every engine (they are in the + `SUPPORTED_SOURCE_EXTENSIONS` whitelist; chat attachments are not + whitelisted by extension and take them too). It never performs OCR: a scanned or image-only PDF is *detected* and handed to the fallback parser — the active OCR backend when OCR is on (see below), Docling when it is installed, the legacy text parsers otherwise — and if nothing can @@ -37,17 +37,17 @@ DOC_PARSER_ENGINE=anydoc this one variable; nothing else changes. With `DOC_PARSER_ENGINE=anydoc`, Docling still handles what anydoc cannot when -it is installed: `.adoc`, `.vtt` and `.xml` sources, the fallback for files -anydoc rejects, and — when it is the OCR backend — scanned PDFs and images. -Without Docling those formats use the standard parsers; OCR still works -through the native backend, and with OCR off images are only read when -`PARSE_IMAGE_REMOTE=true`. +it is installed: the fallback for files anydoc rejects, `.adoc`/`.vtt`/`.xml` +chat attachments (those suffixes are not in the source-upload whitelist), +and — when it is the OCR backend — scanned PDFs and images. Without Docling +those formats use the standard parsers; OCR still works through the native +backend, and with OCR off images are only read when `PARSE_IMAGE_REMOTE=true`. ## Installing the docling engine docling is not part of the base install, and OCR does not need it (see [OCR backends](#ocr-backends)). Add it when you want its layout-model OCR, -`.adoc`/`.vtt`/`.xml` source parsing, or `read_document`'s `structured` +`.adoc`/`.vtt`/`.xml` attachment parsing, or `read_document`'s `structured` output: ```bash @@ -63,7 +63,12 @@ docker build --build-arg INSTALL_DOCLING=true ./application `deployment/docker-compose.yaml` forwards the same switch, so setting `INSTALL_DOCLING=true` in `.env` (or the shell) bakes docling into locally built backend and worker images; `setup.sh` offers it as a follow-up to the -OCR question. Either way no code changes are needed — docling is picked up as +OCR question. Compose reads build arguments from the shell or from the +`.env` you pass with `--env-file .env` (not from the containers' `env_file`), +so build with `docker compose --env-file .env -f deployment/docker-compose.yaml build` +as `setup.sh` does. Pre-built Docker Hub images (`docker-compose-hub.yaml`) +never include docling; install it in a derived image instead. Either way no +code changes are needed — docling is picked up as the fallback engine (and, under `OCR_BACKEND=auto`, as the OCR backend) as soon as it is importable, and `DOC_PARSER_ENGINE=docling` makes it the primary parser. diff --git a/frontend/src/locale/de.json b/frontend/src/locale/de.json index b1ee147f..acad9fdc 100644 --- a/frontend/src/locale/de.json +++ b/frontend/src/locale/de.json @@ -765,7 +765,7 @@ "start": "Chat starten", "name": "Name", "choose": "Dateien auswählen", - "info": "Bitte lade .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .ppt, .odp, .zip hoch (max. 25 MB)", + "info": "Bitte lade .pdf, .txt, .rst, .csv, .xlsx, .xlsm, .xlsb, .xls, .ods, .docx, .docm, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .pptm, .ppt, .pps, .ppsx, .ppsm, .pot, .odp, .zip hoch (max. 25 MB)", "uploadedFiles": "Hochgeladene Dateien", "cancel": "Abbrechen", "train": "Trainieren", diff --git a/frontend/src/locale/en.json b/frontend/src/locale/en.json index e9c7badc..adf41dc8 100644 --- a/frontend/src/locale/en.json +++ b/frontend/src/locale/en.json @@ -770,7 +770,7 @@ "start": "Start Chatting", "name": "Name", "choose": "Choose Files", - "info": "Please upload .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .ppt, .odp, .zip limited to 25mb", + "info": "Please upload .pdf, .txt, .rst, .csv, .xlsx, .xlsm, .xlsb, .xls, .ods, .docx, .docm, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .pptm, .ppt, .pps, .ppsx, .ppsm, .pot, .odp, .zip limited to 25mb", "uploadedFiles": "Uploaded Files", "cancel": "Cancel", "train": "Train", diff --git a/frontend/src/locale/es.json b/frontend/src/locale/es.json index 88bd2a18..3fcb19ef 100644 --- a/frontend/src/locale/es.json +++ b/frontend/src/locale/es.json @@ -765,7 +765,7 @@ "start": "Comenzar a chatear", "name": "Nombre", "choose": "Seleccionar Archivos", - "info": "Por favor, sube archivos .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .ppt, .odp, .zip limitados a 25MB", + "info": "Por favor, sube archivos .pdf, .txt, .rst, .csv, .xlsx, .xlsm, .xlsb, .xls, .ods, .docx, .docm, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .pptm, .ppt, .pps, .ppsx, .ppsm, .pot, .odp, .zip limitados a 25MB", "uploadedFiles": "Archivos Subidos", "cancel": "Cancelar", "train": "Entrenar", diff --git a/frontend/src/locale/jp.json b/frontend/src/locale/jp.json index 3bbda7ee..faa486f4 100644 --- a/frontend/src/locale/jp.json +++ b/frontend/src/locale/jp.json @@ -765,7 +765,7 @@ "start": "チャットを開始する", "name": "名前", "choose": "ファイルを選択", - "info": "25MBまでの.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.xhtml、.png、.jpg、.jpeg、.epub、.json、.pptx、.ppt、.odp、.zipファイルをアップロードしてください", + "info": "25MBまでの.pdf、.txt、.rst、.csv、.xlsx、.xlsm、.xlsb、.xls、.ods、.docx、.docm、.doc、.odt、.rtf、.md、.html、.xhtml、.png、.jpg、.jpeg、.epub、.json、.pptx、.pptm、.ppt、.pps、.ppsx、.ppsm、.pot、.odp、.zipファイルをアップロードしてください", "uploadedFiles": "アップロードされたファイル", "cancel": "キャンセル", "train": "トレーニング", diff --git a/frontend/src/locale/ru.json b/frontend/src/locale/ru.json index 39203650..52ffa34d 100644 --- a/frontend/src/locale/ru.json +++ b/frontend/src/locale/ru.json @@ -765,7 +765,7 @@ "start": "Начать чат", "name": "Имя", "choose": "Выбрать файлы", - "info": "Пожалуйста, загрузите файлы .pdf, .txt, .rst, .csv, .xlsx, .xls, .ods, .docx, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .ppt, .odp, .zip размером до 25 МБ", + "info": "Пожалуйста, загрузите файлы .pdf, .txt, .rst, .csv, .xlsx, .xlsm, .xlsb, .xls, .ods, .docx, .docm, .doc, .odt, .rtf, .md, .html, .xhtml, .png, .jpg, .jpeg, .epub, .json, .pptx, .pptm, .ppt, .pps, .ppsx, .ppsm, .pot, .odp, .zip размером до 25 МБ", "uploadedFiles": "Загруженные файлы", "cancel": "Отмена", "train": "Тренировка", diff --git a/frontend/src/locale/zh-TW.json b/frontend/src/locale/zh-TW.json index e4bf57e3..0b77cb1e 100644 --- a/frontend/src/locale/zh-TW.json +++ b/frontend/src/locale/zh-TW.json @@ -765,7 +765,7 @@ "start": "開始對話", "name": "名稱", "choose": "選擇檔案", - "info": "請上傳限制為25MB的.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.xhtml、.png、.jpg、.jpeg、.epub、.json、.pptx、.ppt、.odp、.zip檔案", + "info": "請上傳限制為25MB的.pdf、.txt、.rst、.csv、.xlsx、.xlsm、.xlsb、.xls、.ods、.docx、.docm、.doc、.odt、.rtf、.md、.html、.xhtml、.png、.jpg、.jpeg、.epub、.json、.pptx、.pptm、.ppt、.pps、.ppsx、.ppsm、.pot、.odp、.zip檔案", "uploadedFiles": "已上傳檔案", "cancel": "取消", "train": "訓練", diff --git a/frontend/src/locale/zh.json b/frontend/src/locale/zh.json index e51cdd3b..7e664a4f 100644 --- a/frontend/src/locale/zh.json +++ b/frontend/src/locale/zh.json @@ -765,7 +765,7 @@ "start": "开始聊天", "name": "名称", "choose": "选择文件", - "info": "请上传限制为25MB的.pdf、.txt、.rst、.csv、.xlsx、.xls、.ods、.docx、.doc、.odt、.rtf、.md、.html、.xhtml、.png、.jpg、.jpeg、.epub、.json、.pptx、.ppt、.odp、.zip文件", + "info": "请上传限制为25MB的.pdf、.txt、.rst、.csv、.xlsx、.xlsm、.xlsb、.xls、.ods、.docx、.docm、.doc、.odt、.rtf、.md、.html、.xhtml、.png、.jpg、.jpeg、.epub、.json、.pptx、.pptm、.ppt、.pps、.ppsx、.ppsm、.pot、.odp、.zip文件", "uploadedFiles": "已上传文件", "cancel": "取消", "train": "训练", diff --git a/setup.ps1 b/setup.ps1 index 8d94a2e6..796c4a2d 100644 --- a/setup.ps1 +++ b/setup.ps1 @@ -508,10 +508,17 @@ function Configure-DocProcessing { Write-ColorText "PDF-as-image parsing enabled." -ForegroundColor "Green" } - $ocr_enabled = Read-Host "Enable OCR for document processing (Docling)? (y/N)" + $ocr_enabled = Read-Host "Enable OCR for scanned PDFs and images? (y/N)" if ($ocr_enabled -eq "y" -or $ocr_enabled -eq "Y") { - "DOCLING_OCR_ENABLED=true" | Add-Content -Path $ENV_FILE -Encoding utf8 - Write-ColorText "Docling OCR enabled." -ForegroundColor "Green" + "OCR_ENABLED=true" | Add-Content -Path $ENV_FILE -Encoding utf8 + Write-ColorText "OCR enabled (tesseract, shipped in the Docker image; set OCR_ENGINE=deepseek for a DeepSeek-OCR endpoint)." -ForegroundColor "Green" + $docling_ocr = Read-Host "Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N)" + if ($docling_ocr -eq "y" -or $docling_ocr -eq "Y") { + # Honoured by locally built images (docker compose --env-file .env build); + # pre-built Docker Hub images do not include docling. + "INSTALL_DOCLING=true" | Add-Content -Path $ENV_FILE -Encoding utf8 + Write-ColorText "Docling will be built into locally built images (pre-built Docker Hub images are unaffected)." -ForegroundColor "Green" + } } } diff --git a/setup.sh b/setup.sh index d3111c87..66c4b6a6 100755 --- a/setup.sh +++ b/setup.sh @@ -371,7 +371,7 @@ configure_doc_processing() { # Locally built images include docling via this build arg; it becomes # the OCR backend automatically (OCR_BACKEND=auto). echo "INSTALL_DOCLING=true" >> "$ENV_FILE" - echo -e "${GREEN}Docling will be built into locally built images.${NC}" + echo -e "${GREEN}Docling will be built into locally built images (docker compose --env-file .env build). Pre-built Docker Hub images do not include it.${NC}" fi fi } diff --git a/tests/parser/file/test_anydoc_parser.py b/tests/parser/file/test_anydoc_parser.py index feba529a..6f86fa07 100644 --- a/tests/parser/file/test_anydoc_parser.py +++ b/tests/parser/file/test_anydoc_parser.py @@ -308,7 +308,7 @@ def test_init_parser_imports_for_real_not_just_find_spec(monkeypatch): """A wheel whose native extension fails to load has a spec but no importable module; that must surface at init, not as a bare ImportError from parse_file mid-ingest (which load_data does not catch).""" - import application.parser.file.anydoc_parser as mod + from application.parser.file import anydoc_parser as mod monkeypatch.setattr(mod, "anydoc_available", lambda: True) monkeypatch.setitem(sys.modules, "anydoc", None) @@ -337,6 +337,7 @@ class _FakeDoclingFallback: class _Inner(DoclingParser): def __init__(self): + super().__init__(ocr_enabled=False) self._parser_config = {} self.calls = [] self.last_engine = None diff --git a/tests/parser/file/test_bulk.py b/tests/parser/file/test_bulk.py index 7e68fa93..b5a5cb8a 100644 --- a/tests/parser/file/test_bulk.py +++ b/tests/parser/file/test_bulk.py @@ -749,3 +749,16 @@ class TestGainedFormats: assert "# Title" in out assert "Body text here." in out + + +def test_gained_suffix_without_a_parser_is_rejected_not_read_as_text(tmp_path): + """Without anydoc an OLE .doc must not be indexed as decoded binary garbage.""" + from application.parser.file.base_parser import DocumentParseError + from application.parser.file.bulk import SimpleDirectoryReader + + path = tmp_path / "legacy.doc" + path.write_bytes(b"\xd0\xcf\x11\xe0\xa1\xb1\x1a\xe1" + b"\x00" * 64) + reader = SimpleDirectoryReader(input_files=[str(path)], file_extractor={".pdf": object()}) + + with pytest.raises(DocumentParseError, match="firecrawl-anydoc"): + reader.load_data() diff --git a/tests/parser/file/test_docling_parser.py b/tests/parser/file/test_docling_parser.py index 3bc6f6df..147d7d4a 100644 --- a/tests/parser/file/test_docling_parser.py +++ b/tests/parser/file/test_docling_parser.py @@ -121,19 +121,19 @@ class TestOcrEngineSelection: assert _resolve_ocr_engine("easyocr") == "auto" def test_tesseract_without_binary_degrades_to_auto(self, monkeypatch): - import application.parser.file.docling_parser as dp + from application.parser.file import docling_parser as dp monkeypatch.setattr(dp.shutil, "which", lambda name: None) assert dp._resolve_ocr_engine("tesseract") == "auto" def test_tesseract_with_binary_selected(self, monkeypatch): - import application.parser.file.docling_parser as dp + from application.parser.file import docling_parser as dp monkeypatch.setattr(dp.shutil, "which", lambda name: "/usr/bin/tesseract") assert dp._resolve_ocr_engine("tesseract") == "tesseract" def test_ocrmac_off_darwin_degrades_to_auto(self, monkeypatch): - import application.parser.file.docling_parser as dp + from application.parser.file import docling_parser as dp monkeypatch.setattr(dp.sys, "platform", "linux") assert dp._resolve_ocr_engine("ocrmac") == "auto" @@ -141,7 +141,7 @@ class TestOcrEngineSelection: def test_rapidocr_missing_degrades_to_auto(self, monkeypatch): import sys - import application.parser.file.docling_parser as dp + from application.parser.file import docling_parser as dp monkeypatch.setitem(sys.modules, "rapidocr", None) assert dp._resolve_ocr_engine("rapidocr") == "auto" @@ -167,10 +167,13 @@ class TestOcrEngineSelection: assert options.lang == ["eng", "chi_sim"] assert options.force_full_page_ocr is True - def test_build_tesseract_explicit_languages_win(self): + def test_build_tesseract_explicit_languages_win(self, monkeypatch): pytest.importorskip("docling") + import application.parser.file.ocr_parser as op from application.parser.file.docling_parser import _build_ocr_options + # Pack inventory unknown: the requested list is passed through untouched. + monkeypatch.setattr(op, "tesseract_languages", lambda: None) options = _build_ocr_options("tesseract", ["deu"], False) assert options.lang == ["deu"] @@ -591,7 +594,7 @@ class TestNonOcrParsersLeaveTextAlone: ], ) def test_ocr_is_off_by_construction(self, name): - import application.parser.file.docling_parser as dp + from application.parser.file import docling_parser as dp assert getattr(dp, name)().ocr_enabled is False @@ -1673,3 +1676,33 @@ class TestDoclingOcrPages: pdf.write_bytes(b"%PDF-1.4") with pytest.raises(DocumentParseError, match="page 1 of doc.pdf"): parser.ocr_pages(pdf, [0]) + + +@pytest.mark.unit +class TestTesseractLanguageFilter: + def test_uninstalled_packs_are_dropped_with_a_warning(self, monkeypatch, caplog): + pytest.importorskip("docling") + import application.parser.file.ocr_parser as op + from application.parser.file.docling_parser import _build_ocr_options + + monkeypatch.setattr(op, "tesseract_languages", lambda: frozenset({"eng", "osd"})) + with caplog.at_level("WARNING"): + options = _build_ocr_options("tesseract", ["eng", "chi_sim"], False) + assert options.lang == ["eng"] + assert "chi_sim" in caplog.text + + def test_all_packs_missing_falls_back_to_eng(self, monkeypatch): + pytest.importorskip("docling") + import application.parser.file.ocr_parser as op + from application.parser.file.docling_parser import _build_ocr_options + + monkeypatch.setattr(op, "tesseract_languages", lambda: frozenset({"eng"})) + assert _build_ocr_options("tesseract", ["xyz"], False).lang == ["eng"] + + def test_unknown_inventory_keeps_the_list(self, monkeypatch): + pytest.importorskip("docling") + import application.parser.file.ocr_parser as op + from application.parser.file.docling_parser import _build_ocr_options + + monkeypatch.setattr(op, "tesseract_languages", lambda: None) + assert _build_ocr_options("tesseract", ["eng", "deu"], False).lang == ["eng", "deu"] diff --git a/tests/parser/file/test_html_parser.py b/tests/parser/file/test_html_parser.py index a1b9a4fc..28bdfa05 100644 --- a/tests/parser/file/test_html_parser.py +++ b/tests/parser/file/test_html_parser.py @@ -218,7 +218,7 @@ def test_trim_torn_utf8_tail(): def test_markdown_parser_metadata_reuses_last_parse(tmp_path, rich_html_file, monkeypatch): """The metadata call after parse_file must not build the soup again.""" - import application.parser.file.html_parser as mod + from application.parser.file import html_parser as mod parser = HTMLMarkdownParser() parser.parse_file(rich_html_file) @@ -231,3 +231,48 @@ def test_markdown_parser_metadata_reuses_last_parse(tmp_path, rich_html_file, mo other.write_text("Other") parser.get_file_metadata(other) assert len(calls) == 1 + + +# --- second-pass fixes: XML prolog, data URIs, UTF-16 heads --------------------- + + +def test_xml_prolog_does_not_leak_into_markdown(tmp_path): + from application.parser.file.html_parser import HTMLMarkdownParser + + path = tmp_path / "doc.xhtml" + path.write_bytes( + b'\n\n' + b'T' + b"

Hi

para

" + ) + text = HTMLMarkdownParser().parse_file(path) + assert "xml version" not in text + assert "raw" not in text + assert text.startswith("# Hi") + + +def test_data_uris_are_stripped_from_images_and_links(): + payload = "data:image/png;base64," + "A" * 200_000 + html = ( + f'

before

chartdl' + f'x

after

' + ) + text = html_to_markdown(html) + assert "AAAA" not in text + assert "before" in text and "after" in text + assert len(text) < 200 + + +def test_utf16_head_keeps_an_even_byte_count(tmp_path, monkeypatch): + from application.core.settings import settings + from application.parser.file.html_parser import HTMLMarkdownParser, read_markup_head + + body = "".join(f"

Zeile {i} Über Größe

\n" for i in range(200)) + path = tmp_path / "wide.html" + path.write_bytes(("" + body + "").encode("utf-16")) # BOM-prefixed + head = read_markup_head(path, 3001) + assert len(head) % 2 == 0 + monkeypatch.setattr(settings, "MARKUP_MAX_BYTES", 3001) + text = HTMLMarkdownParser().parse_file(path) + assert "Zeile 0 Über Größe" in text + assert "\x00" not in text diff --git a/tests/parser/file/test_ocr_parser.py b/tests/parser/file/test_ocr_parser.py index 3f608b35..e2e3f6e2 100644 --- a/tests/parser/file/test_ocr_parser.py +++ b/tests/parser/file/test_ocr_parser.py @@ -10,6 +10,7 @@ import sys from pathlib import Path from unittest.mock import MagicMock, patch +import io import logging import pytest @@ -46,7 +47,7 @@ def _text_pdf( path: Path, pages: int = 1, text: str = "Hello text layer, plenty of characters here.", lines: int = 1 ) -> Path: """A born-digital PDF; ``lines`` paragraphs per page make anydoc classify it as text-based.""" - reportlab = pytest.importorskip("reportlab") # noqa: F841 + pytest.importorskip("reportlab") from reportlab.lib.pagesizes import letter from reportlab.pdfgen import canvas @@ -847,3 +848,119 @@ class TestScannedPageProbeImageGate: def test_pages_with_no_text_at_all_still_count(self, tmp_path): assert op.scanned_page_indices(_image_pdf(tmp_path / "scan.pdf", pages=2)) == [0, 1] + + +# --------------------------------------------------------------------------- +# Second-pass fixes: language packs, alpha, animations, image budget, errors +# --------------------------------------------------------------------------- + + +@pytest.mark.unit +class TestTesseractLanguagePacks: + def test_missing_pack_with_exit_zero_is_a_typed_error(self, monkeypatch): + """tesseract drops a missing pack, exits 0 and OCRs with the rest — that must not pass silently.""" + monkeypatch.setattr(op.shutil, "which", lambda name: "/usr/bin/tesseract") + + def fake_run(cmd, input=None, capture_output=None, timeout=None, check=None): + return subprocess.CompletedProcess( + cmd, 0, stdout=b"English only\n", + stderr=b"Error opening data file /usr/share/tessdata/chi_sim.traineddata\n", + ) + + monkeypatch.setattr(op.subprocess, "run", fake_run) + engine = op.TesseractEngine(languages=["eng", "chi_sim"]) + with pytest.raises(op.OcrUnavailableError, match="chi_sim"): + engine.ocr_image(Image.new("RGB", (10, 10), "white")) + + def test_list_langs_is_parsed(self, monkeypatch): + op.tesseract_languages.cache_clear() + monkeypatch.setattr(op.shutil, "which", lambda name: "/usr/bin/tesseract") + + def fake_run(cmd, capture_output=None, timeout=None, check=None): + return subprocess.CompletedProcess( + cmd, 0, stdout=b"List of available languages in /usr/share/tessdata/ (3):\nchi_sim\neng\nosd\n", stderr=b"" + ) + + monkeypatch.setattr(op.subprocess, "run", fake_run) + try: + assert op.tesseract_languages() == frozenset({"chi_sim", "eng", "osd"}) + finally: + op.tesseract_languages.cache_clear() + + +@pytest.mark.unit +class TestImageNormalisation: + def test_transparent_background_is_flattened_to_white(self): + rgba = Image.new("RGBA", (4, 4), (0, 0, 0, 0)) + rgba.putpixel((1, 1), (0, 0, 0, 255)) # one black "text" pixel + flat = Image.open(io.BytesIO(op._png_bytes(rgba))) + assert flat.mode == "RGB" + assert flat.getpixel((0, 0)) == (255, 255, 255) + assert flat.getpixel((1, 1)) == (0, 0, 0) + + def test_palette_transparency_is_flattened(self): + pal = Image.new("P", (2, 2), 0) + pal.info["transparency"] = 0 + flat = Image.open(io.BytesIO(op._png_bytes(pal))) + assert flat.getpixel((0, 0)) == (255, 255, 255) + + def test_oversized_image_is_downscaled_to_the_budget(self, caplog): + big = Image.new("L", (9000, 9000), 255) + with caplog.at_level(logging.WARNING, logger="application.parser.file.ocr_parser"): + small = op.fit_to_pixel_budget(big) + assert small.width * small.height <= op._MAX_RENDER_PIXELS + assert "downscaling" in caplog.text + + def test_small_image_is_untouched(self): + img = Image.new("RGB", (300, 200), "white") + assert op.fit_to_pixel_budget(img) is img + + def test_animated_gif_ocrs_only_the_first_frame(self, tmp_path): + frames = [Image.new("RGB", (20, 20), "white") for _ in range(5)] + gif = tmp_path / "anim.gif" + frames[0].save(gif, save_all=True, append_images=frames[1:], duration=50, loop=0) + engine = FakeEngine(["frame text"]) + parser = op.NativeOcrImageParser(engine=engine) + assert parser.parse_file(gif) == "frame text" + assert parser.get_file_metadata(gif)["ocr_pages"] == 1 + + def test_multi_page_tiff_still_reads_every_frame(self, tmp_path): + frames = [Image.new("L", (20, 20), 255) for _ in range(3)] + tiff = tmp_path / "fax.tiff" + frames[0].save(tiff, save_all=True, append_images=frames[1:]) + parser = op.NativeOcrImageParser(engine=FakeEngine(["a", "b", "c"])) + assert parser.parse_file(tiff) == "a\n\nb\n\nc" + + +@pytest.mark.unit +class TestDeepseekErrorShapes: + def test_non_json_body_is_reported_as_such(self, monkeypatch): + import requests + + class _Resp: + def raise_for_status(self): + return None + + def json(self): + raise requests.exceptions.JSONDecodeError("Expecting value", "", 0) + + monkeypatch.setattr(requests, "post", lambda *a, **k: _Resp()) + engine = op.DeepseekOcrEngine(url="http://proxy.local/v1/chat/completions", model="m") + with pytest.raises(DocumentParseError, match="non-JSON body"): + engine.ocr_image(Image.new("RGB", (5, 5), "white")) + + +@pytest.mark.unit +def test_delegate_parse_lets_setup_errors_through(): + """A fallback whose dependency is missing is a deployment problem, not a bad file.""" + from application.parser.file.base_parser import BaseParser, delegate_parse + + class _NeedsLib(BaseParser): + def _init_parser(self): + raise ImportError("docling is required") + + def parse_file(self, file, errors="ignore"): + return "never" + + with pytest.raises(ImportError, match="docling is required"): + delegate_parse(_NeedsLib(), Path("x.pdf"), "ignore") diff --git a/tests/parser/file/test_tableize.py b/tests/parser/file/test_tableize.py index 2599befa..88138e3e 100644 --- a/tests/parser/file/test_tableize.py +++ b/tests/parser/file/test_tableize.py @@ -72,3 +72,23 @@ def test_non_table_text_passes_through_verbatim(): """Only the trailing newline may differ (splitlines/join round-trip).""" md = "# Heading\n\nA paragraph with no numbers.\n\n- a list item\n" assert tableize(md) == md.rstrip("\n") + + +def test_single_trailing_number_lines_are_not_a_table(): + """Headings, footnotes and version lists look like 'word number' rows; leave them alone.""" + from application.parser.file.tableize import tableize + + for block in ( + "Chapter 1\nChapter 2\nChapter 3", + "Footnote 1\nFootnote 2\nFootnote 3", + "ISO 9001\nISO 14001\nISO 27001", + "Version 2.0\nVersion 3.0\nVersion 4.0", + ): + assert tableize(block) == block + + +def test_leader_rows_with_one_value_still_convert(): + from application.parser.file.tableize import tableize + + block = "Revenue ...... 1,234\nCosts ...... 567\nProfit ...... 667" + assert "| --- |" in tableize(block) diff --git a/tests/parser/remote/test_s3_loader.py b/tests/parser/remote/test_s3_loader.py index a381e07d..e2dd4c45 100644 --- a/tests/parser/remote/test_s3_loader.py +++ b/tests/parser/remote/test_s3_loader.py @@ -903,3 +903,14 @@ class TestSSRFValidation: with pytest.raises(ValueError, match="Invalid S3 endpoint_url"): s3_loader.load_data(input_data) mock_boto3.client.assert_not_called() + + +def test_is_supported_document_follows_the_upload_whitelist(): + from application.parser.file.constants import SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS + from application.parser.remote.s3_loader import S3Loader + + for suffix in SUPPORTED_SOURCE_DOCUMENT_EXTENSIONS: + key = f"bucket/file{suffix}" + assert S3Loader.is_text_file(None, key) or S3Loader.is_supported_document(None, key), suffix + assert S3Loader.is_supported_document(None, "bucket/file.doc") + assert not S3Loader.is_supported_document(None, "bucket/file.exe") diff --git a/tests/parser/test_document_reader.py b/tests/parser/test_document_reader.py index ab6bd550..f110431b 100644 --- a/tests/parser/test_document_reader.py +++ b/tests/parser/test_document_reader.py @@ -707,3 +707,49 @@ def test_structured_output_stays_docling_under_anydoc(monkeypatch): out = parse_document_bytes(b"%PDF-1.4", "doc.pdf", output="structured") assert out["output"] == "structured" assert len(calls) == 1 + + +# --- second-pass routing fixes --------------------------------------------------- + + +def test_fast_engine_covers_the_whole_legacy_map(): + """`fast` must never fall through to the configured (anydoc/docling) map.""" + for suffix, name in { + ".xhtml": "HTMLParser", + ".pptx": "PPTXParser", + ".epub": "EpubParser", + ".rst": "RstParser", + ".json": "JSONParser", + }.items(): + assert type(dr._legacy_parser_for(suffix)).__name__ == name + + +def test_tables_are_not_collected_when_the_pdf_parser_is_not_docling(monkeypatch): + """Under OCR_BACKEND=native the docling engine hands PDFs to the native OCR + parser; a vanilla DocumentConverter table pass would OCR the scan again.""" + + from application.parser.file.ocr_parser import NativeOcrPdfParser + + native = NativeOcrPdfParser() + native._parser_config = {} + monkeypatch.setattr(native, "parse_file", lambda path, errors="ignore": "native text") + + calls = [] + monkeypatch.setattr(dr, "_pick_parser", lambda *a, **k: native) + monkeypatch.setattr(dr, "_effective_engine", lambda engine: "docling") + monkeypatch.setattr(dr, "_docling_structured", lambda *a, **k: calls.append(1) or {"tables": []}) + monkeypatch.setattr(dr, "_zip_bomb_reason", lambda *a, **k: None) + + out = parse_document_bytes(b"%PDF-1.4", "scan.pdf", engine="docling", include_tables=True) + + assert out["content"] == "native text" + assert calls == [] + + +def test_binary_office_suffix_without_anydoc_is_an_error_not_text(monkeypatch): + monkeypatch.setattr(dr, "_pick_parser", lambda *a, **k: None) + monkeypatch.setattr(dr, "_zip_bomb_reason", lambda *a, **k: None) + + out = parse_document_bytes(b"\xd0\xcf\x11\xe0 OLE bytes", "legacy.doc", output="text") + + assert "firecrawl-anydoc" in out["error"]