"""Tests for the anydoc parser (the default ``DOC_PARSER_ENGINE``). Three behaviours matter and are asserted separately: * a file anydoc converts comes back as Markdown, with ``last_engine`` set; * a file it refuses (a scanned PDF, an unknown format) goes to the fallback parser when one is configured, and every failure surfaces as ``DocumentParseError`` — never as an empty document or a raw traceback; * the suffix list the parser map is built from matches what the installed anydoc actually accepts. """ import importlib.machinery import sys import types from pathlib import Path import pytest from docsgpt.parser.file.base_parser import BaseParser, DocumentParseError anydoc = pytest.importorskip("anydoc") from docsgpt.parser.file.anydoc_parser import ( # noqa: E402 — after importorskip ANYDOC_SUFFIXES, AnydocParser, anydoc_available, ) # Long enough to clear the scanned-PDF near-empty guard (_MIN_SCAN_FALLBACK_CHARS). FALLBACK_TEXT = "fallback parser text output, long enough to clear the scanned-PDF guard." class _RecordingFallback(BaseParser): """Stands in for docling / a legacy parser so delegation is observable.""" def __init__(self, result=FALLBACK_TEXT): super().__init__(parser_config={}) self.calls = [] self._result = result def _init_parser(self): return {} def parse_file(self, file, errors="ignore"): self.calls.append(file) return self._result class _ExplodingFallback(BaseParser): def _init_parser(self): return {} def parse_file(self, file, errors="ignore"): raise RuntimeError("fallback blew up") def _scanned_pdf(path, pages=2): """A PDF with pages but no text layer: what a scan looks like to a parser.""" from pypdf import PdfWriter writer = PdfWriter() for _ in range(pages): writer.add_blank_page(width=612, height=792) with open(path, "wb") as fh: writer.write(fh) return path def _fake_anydoc(monkeypatch, to_markdown): """Install a stub ``anydoc`` module whose ``to_markdown`` is ``to_markdown``.""" fake = types.ModuleType("anydoc") fake.__spec__ = importlib.machinery.ModuleSpec("anydoc", None) class ConvertError(Exception): pass class UnsupportedError(ConvertError): pass class NeedsOcrError(ConvertError): pass class ResourceLimitError(ConvertError): pass fake.ConvertError = ConvertError fake.UnsupportedError = UnsupportedError fake.NeedsOcrError = NeedsOcrError fake.ResourceLimitError = ResourceLimitError fake.to_markdown = to_markdown monkeypatch.setitem(sys.modules, "anydoc", fake) return fake # --- the suffix list is what anydoc really supports --------------------------- def test_every_listed_suffix_is_an_anydoc_format(): for suffix in ANYDOC_SUFFIXES: assert anydoc.format_from_extension(suffix) is not None, suffix def test_epub_and_html_are_not_routed_to_anydoc(): assert ".epub" not in ANYDOC_SUFFIXES assert ".html" not in ANYDOC_SUFFIXES # --- successful conversion ----------------------------------------------------- def test_csv_converts_to_markdown_table(tmp_path): path = tmp_path / "items.csv" path.write_text("name,qty\nwidget,3\ngadget,5\n") parser = AnydocParser() parser.init_parser() out = parser.parse_file(path) assert isinstance(out, str) assert "widget" in out and "gadget" in out assert "|" in out # rendered as a GFM table, not flat text assert parser.last_engine == "anydoc" def test_xlsx_converts_with_cell_text(tmp_path): openpyxl = pytest.importorskip("openpyxl") path = tmp_path / "book.xlsx" wb = openpyxl.Workbook() ws = wb.active ws.append(["city", "population"]) ws.append(["Ljubljana", 295000]) wb.save(path) out = AnydocParser().parse_file(path) assert "Ljubljana" in out assert "295000" in out def test_accepts_path_and_str(tmp_path): path = tmp_path / "a.csv" path.write_text("x,y\n1,2\n") assert AnydocParser().parse_file(str(path)) == AnydocParser().parse_file(path) # --- refusal routes to the fallback -------------------------------------------- def test_scanned_pdf_delegates_to_fallback(tmp_path): path = _scanned_pdf(tmp_path / "scan.pdf") fallback = _RecordingFallback() parser = AnydocParser(fallback_parser=fallback) out = parser.parse_file(path) assert out == FALLBACK_TEXT assert fallback.calls == [path] assert parser.last_engine == "_RecordingFallback" def test_fallback_last_engine_is_reported_when_it_has_one(tmp_path): path = _scanned_pdf(tmp_path / "scan.pdf") fallback = _RecordingFallback() fallback.last_engine = "pypdfium2" parser = AnydocParser(fallback_parser=fallback) parser.parse_file(path) assert parser.last_engine == "pypdfium2" def test_scanned_pdf_without_fallback_is_document_parse_error(tmp_path): path = _scanned_pdf(tmp_path / "scan.pdf") parser = AnydocParser() with pytest.raises(DocumentParseError, match="scan.pdf"): parser.parse_file(path) assert parser.last_engine is None def test_unknown_format_without_fallback_is_document_parse_error(tmp_path): path = tmp_path / "blob.xyz" path.write_bytes(b"hello") with pytest.raises(DocumentParseError, match="blob.xyz"): AnydocParser().parse_file(path) def test_fallback_failure_is_document_parse_error(tmp_path): """A fallback that raises must not leak a bare exception (load_data skips only DocumentParseError).""" path = _scanned_pdf(tmp_path / "scan.pdf") with pytest.raises(DocumentParseError): AnydocParser(fallback_parser=_ExplodingFallback()).parse_file(path) def test_fallback_document_parse_error_passes_through(tmp_path): class _Refusing(BaseParser): def _init_parser(self): return {} def parse_file(self, file, errors="ignore"): raise DocumentParseError("scan needs OCR") path = _scanned_pdf(tmp_path / "scan.pdf") with pytest.raises(DocumentParseError, match="scan needs OCR"): AnydocParser(fallback_parser=_Refusing()).parse_file(path) def test_missing_file_is_document_parse_error(tmp_path): with pytest.raises(DocumentParseError, match="could not be read"): AnydocParser().parse_file(tmp_path / "missing.docx") def test_empty_output_delegates(tmp_path, monkeypatch): """Whitespace-only output is treated as a failed conversion, not stored.""" _fake_anydoc(monkeypatch, lambda path: " \n\n ") fallback = _RecordingFallback("real text") path = tmp_path / "x.docx" path.write_bytes(b"irrelevant") out = AnydocParser(fallback_parser=fallback).parse_file(path) assert out == "real text" assert fallback.calls == [path] def test_typed_convert_error_delegates(tmp_path, monkeypatch): fake = _fake_anydoc(monkeypatch, None) def _refuse(path): raise fake.UnsupportedError("PDF has no extractable text; OCR is required") fake.to_markdown = _refuse fallback = _RecordingFallback() path = tmp_path / "x.pdf" path.write_bytes(b"%PDF-1.4") assert AnydocParser(fallback_parser=fallback).parse_file(path) == FALLBACK_TEXT def test_typed_convert_error_message_reaches_the_user(tmp_path, monkeypatch): fake = _fake_anydoc(monkeypatch, None) def _refuse(path): raise fake.UnsupportedError("PDF has no extractable text; OCR is required") fake.to_markdown = _refuse path = tmp_path / "x.pdf" path.write_bytes(b"%PDF-1.4") with pytest.raises(DocumentParseError, match="OCR is required"): AnydocParser().parse_file(path) def test_resource_limit_is_terminal_even_with_fallback(tmp_path, monkeypatch): """anydoc refusing a file as too expensive must not hand it to a heavier engine that would then spend exactly what was refused.""" fake = _fake_anydoc(monkeypatch, None) def _refuse(path): raise fake.ResourceLimitError( "resource limit exceeded (max_entry_bytes): word/document.xml declares 400400167 bytes" ) fake.to_markdown = _refuse fallback = _RecordingFallback() parser = AnydocParser(fallback_parser=fallback) path = tmp_path / "bomb.docx" path.write_bytes(b"PK") with pytest.raises(DocumentParseError, match="max_entry_bytes"): parser.parse_file(path) assert fallback.calls == [] assert parser.last_engine is None def test_unexpected_exception_is_document_parse_error(tmp_path, monkeypatch): def _crash(path): raise RuntimeError("panic") _fake_anydoc(monkeypatch, _crash) path = tmp_path / "x.docx" path.write_bytes(b"irrelevant") with pytest.raises(DocumentParseError, match="x.docx"): AnydocParser().parse_file(path) # --- init / availability -------------------------------------------------------- def test_init_parser_records_fallback(): parser = AnydocParser(fallback_parser=_RecordingFallback()) parser.init_parser() assert parser.parser_config == {"fallback_parser": "_RecordingFallback"} bare = AnydocParser() bare.init_parser() assert bare.parser_config == {"fallback_parser": None} def test_init_parser_without_anydoc_raises_import_error(monkeypatch): monkeypatch.setitem(sys.modules, "anydoc", None) assert not anydoc_available() with pytest.raises(ImportError, match="firecrawl-anydoc"): AnydocParser().init_parser() def test_init_parser_imports_for_real_not_just_find_spec(monkeypatch): """A wheel whose native extension fails to load has a spec but no importable module; that must surface at init, not as a bare ImportError from parse_file mid-ingest (which load_data does not catch).""" from docsgpt.parser.file import anydoc_parser as mod monkeypatch.setattr(mod, "anydoc_available", lambda: True) monkeypatch.setitem(sys.modules, "anydoc", None) with pytest.raises(ImportError, match="firecrawl-anydoc"): AnydocParser().init_parser() # --- PDF trust check + tableize wiring (PR 3) ----------------------------------- FIXTURES = Path(__file__).parent / "fixtures" CID_PDF = FIXTURES / "nda_en_zh_cid_font.pdf" # Long enough to clear the near-empty guard a trust-check re-parse must pass. _REROUTE_TEXT = "docling reroute text, long enough to clear the near-empty reroute guard" # What anydoc's CID-font silent drop looked like: the English column extracted, # the Chinese one gone. anydoc 0.2.4 refuses this PDF outright # (``test_cid_font_pdf_is_refused_as_needing_ocr``), so the trust check is # exercised against CID_PDF's real bytes with the dropped output stubbed in — # the check's own inputs, independent of which anydoc is installed. _CID_DROPPED_TEXT = ( "# NON-DISCLOSURE AGREEMENT\n\nThis Agreement is entered into between the " "parties on the date set out below. Confidential Information shall not be " "disclosed to any third party." ) def _anydoc_dropping_cjk(monkeypatch): """Stub anydoc into the silent CID-font drop the trust check exists for.""" return _fake_anydoc(monkeypatch, lambda path: _CID_DROPPED_TEXT) class _FakeDoclingFallback: """Registered as a DoclingParser subclass so ``_is_docling_backed`` is True.""" def __new__(cls): from docsgpt.parser.file.docling_parser import DoclingParser class _Inner(DoclingParser): def __init__(self): super().__init__(ocr_enabled=False) self._parser_config = {} self.calls = [] self.last_engine = None def parse_file(self, file, errors="ignore"): self.calls.append(file) return _REROUTE_TEXT return _Inner() def test_cid_font_pdf_is_refused_as_needing_ocr(): """anydoc 0.2.4 detects the unmappable CID font and raises NeedsOcrError rather than dropping the Chinese column silently; that must route as "needs OCR", which is what makes the scan guard fire.""" fallback = _RecordingFallback(result=" ") fallback.ocr_enabled = False parser = AnydocParser(fallback_parser=fallback) with pytest.raises(DocumentParseError, match="OCR_ENABLED=true"): parser.parse_file(CID_PDF) def test_trust_flagged_pdf_reroutes_to_docling_fallback(monkeypatch): _anydoc_dropping_cjk(monkeypatch) fallback = _FakeDoclingFallback() parser = AnydocParser(fallback_parser=fallback) out = parser.parse_file(CID_PDF) assert out == _REROUTE_TEXT assert fallback.calls == [CID_PDF] assert parser.last_engine == "_Inner" assert parser.get_file_metadata(CID_PDF) == {} # rerouted, nothing to warn about def test_trust_flagged_pdf_without_docling_keeps_output_and_warns(monkeypatch): _anydoc_dropping_cjk(monkeypatch) parser = AnydocParser(fallback_parser=_RecordingFallback()) # not docling-backed out = parser.parse_file(CID_PDF) assert "NON-DISCLOSURE" in out.upper() # anydoc's own output kept meta = parser.get_file_metadata(CID_PDF) assert "parse_warnings" in meta assert any("ToUnicode" in w for w in meta["parse_warnings"]) assert any("CJK" in w for w in meta["parse_warnings"]) assert parser.last_engine == "anydoc" def test_trust_check_disabled_stamps_nothing(monkeypatch): from docsgpt.parser.file import anydoc_parser as ap monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False) _anydoc_dropping_cjk(monkeypatch) parser = AnydocParser() parser.parse_file(CID_PDF) assert parser.get_file_metadata(CID_PDF) == {} def test_trust_reroute_failure_keeps_anydoc_output(monkeypatch): _anydoc_dropping_cjk(monkeypatch) fallback = _FakeDoclingFallback() def _boom(file, errors="ignore"): raise RuntimeError("docling exploded") fallback.parse_file = _boom parser = AnydocParser(fallback_parser=fallback) out = parser.parse_file(CID_PDF) assert "docling" not in out assert "parse_warnings" in parser.get_file_metadata(CID_PDF) assert parser.last_engine == "anydoc" def test_trust_reroute_near_empty_keeps_anydoc_output(monkeypatch): """A docling pipeline dropout ('' / '') returns without raising; adopting it would swap anydoc's real text for an empty document.""" _anydoc_dropping_cjk(monkeypatch) fallback = _FakeDoclingFallback() fallback.parse_file = lambda file, errors="ignore": "" parser = AnydocParser(fallback_parser=fallback) out = parser.parse_file(CID_PDF) assert "" not in out assert "parse_warnings" in parser.get_file_metadata(CID_PDF) assert parser.last_engine == "anydoc" def test_warnings_reset_between_files(monkeypatch, tmp_path): _anydoc_dropping_cjk(monkeypatch) parser = AnydocParser() parser.parse_file(CID_PDF) assert parser.get_file_metadata(CID_PDF) != {} clean = tmp_path / "clean.csv" clean.write_text("a,b\n1,2\n") parser.parse_file(clean) assert parser.get_file_metadata(clean) == {} assert parser.get_file_metadata(CID_PDF) == {} def test_trust_check_errors_never_fail_the_parse(monkeypatch, tmp_path): def _explode(path, markdown): raise RuntimeError("scanner bug") import docsgpt.parser.file.pdf_trust as pt monkeypatch.setattr(pt, "verify_pdf_file", _explode) _fake_anydoc(monkeypatch, lambda path: "# converted fine") path = tmp_path / "x.pdf" path.write_bytes(b"%PDF-1.4 irrelevant") assert AnydocParser().parse_file(path) == "# converted fine" def test_tableize_applied_when_enabled(monkeypatch, tmp_path): from docsgpt.parser.file import anydoc_parser as ap monkeypatch.setattr(ap.settings, "ANYDOC_TABLEIZE", True) monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False) _fake_anydoc( monkeypatch, lambda path: "Cash ..... 1,234 900\nDebt ..... 2,000 1,500\nEquity ..... 900 800", ) path = tmp_path / "x.pdf" path.write_bytes(b"%PDF-1.4 irrelevant") out = AnydocParser().parse_file(path) assert "| Cash | 1,234 | 900 |" in out def test_tableize_disabled_by_default(monkeypatch, tmp_path): from docsgpt.parser.file import anydoc_parser as ap monkeypatch.setattr(ap.settings, "PDF_TRUST_CHECK", False) flat = "Cash ..... 1,234 900\nDebt ..... 2,000 1,500\nEquity ..... 900 800" _fake_anydoc(monkeypatch, lambda path: flat) path = tmp_path / "x.pdf" path.write_bytes(b"%PDF-1.4 irrelevant") assert AnydocParser().parse_file(path) == flat def test_tableize_never_touches_docling_reroute(monkeypatch): from docsgpt.parser.file import anydoc_parser as ap monkeypatch.setattr(ap.settings, "ANYDOC_TABLEIZE", True) _anydoc_dropping_cjk(monkeypatch) fallback = _FakeDoclingFallback() parser = AnydocParser(fallback_parser=fallback) out = parser.parse_file(CID_PDF) assert out == _REROUTE_TEXT # verbatim, not post-processed # --- scanned-PDF near-empty guard (the OCR_ENGINE seam's loud-failure side) ------ def test_scanned_pdf_with_near_empty_fallback_fails_loudly_when_ocr_off(tmp_path): """A scan whose fallback (OCR off) extracts almost nothing must fail with an actionable message, not be stored as an empty document.""" from docsgpt.parser.file.base_parser import NoTextLayerError path = _scanned_pdf(tmp_path / "scan.pdf") fallback = _RecordingFallback(result=" ") fallback.ocr_enabled = False parser = AnydocParser(fallback_parser=fallback) # Still a DocumentParseError for source ingestion, but typed, so the # attachment worker can keep the file for models that read it natively. with pytest.raises(NoTextLayerError, match="OCR_ENABLED"): parser.parse_file(path) assert parser.last_engine is None def test_scanned_pdf_with_near_empty_fallback_and_ocr_on_reports_it(tmp_path): from docsgpt.parser.file.base_parser import NoTextLayerError path = _scanned_pdf(tmp_path / "scan.pdf") fallback = _RecordingFallback(result="x") fallback.ocr_enabled = True with pytest.raises(NoTextLayerError, match="even with OCR enabled"): AnydocParser(fallback_parser=fallback).parse_file(path) def test_scanned_pdf_with_substantial_fallback_output_passes(tmp_path): """docling extracting a text layer anydoc refused (the HK-bill case) must keep working — the guard only fires on near-empty results.""" path = _scanned_pdf(tmp_path / "scan.pdf") out = AnydocParser(fallback_parser=_RecordingFallback()).parse_file(path) assert out == FALLBACK_TEXT def test_non_scan_refusals_do_not_trigger_the_guard(tmp_path, monkeypatch): """Only the OCR-required refusal implies 'content exists but needs OCR'; a malformed file with a short fallback result stays a successful parse.""" fake = _fake_anydoc(monkeypatch, None) class MalformedError(fake.ConvertError): pass fake.MalformedError = MalformedError def _refuse(path): raise MalformedError("structurally unusable") fake.to_markdown = _refuse path = tmp_path / "x.docx" path.write_bytes(b"irrelevant") out = AnydocParser(fallback_parser=_RecordingFallback(result="tiny")).parse_file(path) assert out == "tiny" # --- spreadsheets: a resource limit delegates, other formats stay terminal ---- def test_resource_limit_on_a_spreadsheet_delegates_to_the_tabular_fallback(tmp_path, monkeypatch): """anydoc refuses a 265k-row sheet on fixed limits that openpyxl/pandas read fine, so for tabular suffixes the limit is a reason to fall back.""" fake = _fake_anydoc(monkeypatch, None) def _refuse(path): raise fake.ResourceLimitError("resource limit exceeded (max_xml_nodes): 2000000") fake.to_markdown = _refuse fallback = _RecordingFallback() parser = AnydocParser(fallback_parser=fallback) path = tmp_path / "big.xlsx" path.write_bytes(b"PK") assert parser.parse_file(path) == FALLBACK_TEXT assert fallback.calls == [path] assert parser.last_engine == "_RecordingFallback" def test_resource_limit_on_a_spreadsheet_without_fallback_is_terminal(tmp_path, monkeypatch): fake = _fake_anydoc(monkeypatch, None) def _refuse(path): raise fake.ResourceLimitError("resource limit exceeded (max_entry_bytes)") fake.to_markdown = _refuse path = tmp_path / "big.xlsx" path.write_bytes(b"PK") with pytest.raises(DocumentParseError, match="max_entry_bytes"): AnydocParser().parse_file(path) # --- mixed documents: one page's OCR failure costs that page only ------------ class _PerPageOcrFallback(BaseParser): ocr_enabled = True def __init__(self, failing_index): super().__init__() self.failing_index = failing_index self.requests = [] def _init_parser(self): return {} def parse_file(self, file, errors="ignore"): return "unused" def ocr_pages(self, file, indices): self.requests.append(list(indices)) if self.failing_index in indices: raise DocumentParseError("tesseract failed on this page") return {index: f"OCR TEXT PAGE {index + 1}" for index in indices} def test_one_failing_page_does_not_drop_the_other_pages_ocr(tmp_path, monkeypatch): pytest.importorskip("pypdfium2") pytest.importorskip("pypdf") _fake_anydoc(monkeypatch, lambda path: "Text pages read by anydoc.") fallback = _PerPageOcrFallback(failing_index=0) parser = AnydocParser(fallback_parser=fallback) path = _scanned_pdf(tmp_path / "mixed.pdf", pages=3) # every page probes as scanned text = parser.parse_file(path) assert fallback.requests == [[0], [1], [2]] assert "OCR TEXT PAGE 1" not in text assert "OCR TEXT PAGE 2" in text and "OCR TEXT PAGE 3" in text assert text.startswith("Text pages read by anydoc.") # (The font-less blank-page fixture also earns a trust-check warning.) assert parser.get_file_metadata(path)["ocr_pages"] == 2