Files
DocsGPT/tests/worker/test_attachment_worker.py
T
2026-08-04 15:07:54 +01:00

253 lines
9.4 KiB
Python

"""Smoke test for ``application.worker.attachment_worker``.
The happy path parses an uploaded file and inserts a row into
``attachments``. We mock the parser boundary (``StorageCreator.get_storage``
returns a storage whose ``process_file`` produces a pre-built Document)
but let the PG insert run against the ephemeral ``pg_conn`` so we can
assert one concrete row is visible after the task returns.
"""
from __future__ import annotations
from pathlib import Path
from unittest.mock import MagicMock
import pytest
from application.parser.schema.base import Document
from application.storage.db.repositories.attachments import AttachmentsRepository
@pytest.mark.unit
class TestAttachmentWorker:
def test_inserts_row_in_attachments(
self, pg_conn, patch_worker_db, task_self, monkeypatch
):
from application import worker
fake_doc = Document(
text="hello world",
extra_info={"transcript_language": "en"},
)
fake_storage = MagicMock(name="storage")
fake_storage.process_file.return_value = fake_doc
monkeypatch.setattr(
worker.StorageCreator, "get_storage", lambda: fake_storage
)
# Stub the parser selection so the docling import path isn't taken.
monkeypatch.setattr(
worker, "get_default_file_extractor", lambda ocr_enabled=False: {}
)
file_info = {
"filename": "notes.txt",
"attachment_id": "507f1f77bcf86cd799439011",
"path": "uploads/user1/notes.txt",
"metadata": {"source": "chat"},
}
result = worker.attachment_worker(task_self, file_info, "user1")
assert result["filename"] == "notes.txt"
assert result["token_count"] > 0
# Parser metadata (``transcript_*``) should have been merged in.
assert result["metadata"]["transcript_language"] == "en"
assert result["metadata"]["source"] == "chat"
# Row should be resolvable by the caller-visible handle stored in
# ``legacy_mongo_id``.
row = AttachmentsRepository(pg_conn).get_by_legacy_id(
file_info["attachment_id"], "user1"
)
assert row is not None, "attachment_worker should insert a row"
assert row["filename"] == "notes.txt"
assert row["upload_path"] == "uploads/user1/notes.txt"
assert row["content"] == "hello world"
assert row["user_id"] == "user1"
@pytest.mark.unit
class TestBoundedAttachmentCopy:
"""``_bounded_attachment_copy`` head-truncates oversized line-oriented
text attachments to a temp copy before parsing.
The parsed content is capped at ~250k chars downstream anyway, so bytes
beyond the cap only cost parse time and memory. Local storage hands the
canonical stored file to the processor, so truncation must never happen
in place.
"""
def _write(self, tmp_path, name: str, data: bytes):
path = tmp_path / name
path.write_bytes(data)
return path
def test_oversized_csv_is_copied_and_truncated(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
monkeypatch.setattr(settings, "ATTACHMENT_TEXT_MAX_BYTES", 1024)
original = self._write(
tmp_path, "big.csv", b"".join(b"%d,%d\n" % (i, i) for i in range(1000))
)
original_size = original.stat().st_size
parse_path, is_temp = worker._bounded_attachment_copy(str(original))
assert is_temp is True
assert parse_path != str(original)
assert parse_path.endswith(".csv")
copied = Path(parse_path).read_bytes()
assert 0 < len(copied) <= 1024
assert copied.endswith(b"\n"), "must cut on a line boundary"
# The stored original must be untouched.
assert original.stat().st_size == original_size
Path(parse_path).unlink()
def test_small_file_returned_as_is(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
monkeypatch.setattr(settings, "ATTACHMENT_TEXT_MAX_BYTES", 1024)
original = self._write(tmp_path, "small.csv", b"a,b\n1,2\n")
parse_path, is_temp = worker._bounded_attachment_copy(str(original))
assert parse_path == str(original)
assert is_temp is False
def test_non_text_suffix_is_never_truncated(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
monkeypatch.setattr(settings, "ATTACHMENT_TEXT_MAX_BYTES", 64)
original = self._write(tmp_path, "doc.pdf", b"%PDF-1.7 " + b"x" * 500)
parse_path, is_temp = worker._bounded_attachment_copy(str(original))
assert parse_path == str(original)
assert is_temp is False
def test_cap_zero_disables_truncation(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
monkeypatch.setattr(settings, "ATTACHMENT_TEXT_MAX_BYTES", 0)
original = self._write(tmp_path, "big.csv", b"1,2\n" * 1000)
parse_path, is_temp = worker._bounded_attachment_copy(str(original))
assert parse_path == str(original)
assert is_temp is False
def test_single_line_without_newline_falls_back_to_hard_cut(
self, tmp_path, monkeypatch
):
from application import worker
from application.core.settings import settings
monkeypatch.setattr(settings, "ATTACHMENT_TEXT_MAX_BYTES", 256)
original = self._write(tmp_path, "oneline.txt", b"x" * 5000)
parse_path, is_temp = worker._bounded_attachment_copy(str(original))
assert is_temp is True
data = Path(parse_path).read_bytes()
assert len(data) == 256
Path(parse_path).unlink()
def test_leading_newline_does_not_collapse_the_copy(self, tmp_path, monkeypatch):
"""A window whose only newline sits at byte 0 must keep its content.
``rfind`` returns 0 for this shape, so cutting at that boundary would
write a one-byte copy and throw the attachment away — the partial
final line is the better trade.
"""
from application import worker
from application.core.settings import settings
monkeypatch.setattr(settings, "ATTACHMENT_TEXT_MAX_BYTES", 256)
original = self._write(tmp_path, "leading.log", b"\n" + b"x" * 5000)
parse_path, is_temp = worker._bounded_attachment_copy(str(original))
assert is_temp is True
data = Path(parse_path).read_bytes()
assert len(data) == 256
Path(parse_path).unlink()
@pytest.mark.unit
class TestAttachmentZipBombGuard:
"""``_reject_attachment_zip_bomb`` brings the ingest-path zip-bomb guard to
the attachment path (which previously had none): a zip-container attachment
that declares too many entries / too much inner data is rejected before any
parser touches it, with a non-retryable error."""
def _make_xlsx(self, path: Path, rows: int = 20):
from openpyxl import Workbook
wb = Workbook(write_only=True)
ws = wb.create_sheet()
for i in range(rows):
ws.append([i, i * 2, i * 3])
wb.save(str(path))
def test_rejects_when_inner_size_exceeds_cap(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
path = tmp_path / "book.xlsx"
self._make_xlsx(path)
monkeypatch.setattr(settings, "DOCUMENT_MAX_DECOMPRESSED_BYTES", 100)
with pytest.raises(worker.AttachmentRejectedError):
worker._reject_attachment_zip_bomb(str(path))
def test_rejects_when_too_many_entries(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
path = tmp_path / "book.xlsx"
self._make_xlsx(path)
monkeypatch.setattr(settings, "DOCUMENT_MAX_ARCHIVE_ENTRIES", 1)
with pytest.raises(worker.AttachmentRejectedError):
worker._reject_attachment_zip_bomb(str(path))
def test_allows_reasonable_archive(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
path = tmp_path / "book.xlsx"
self._make_xlsx(path)
monkeypatch.setattr(settings, "DOCUMENT_MAX_DECOMPRESSED_BYTES", 300 * 1024 * 1024)
monkeypatch.setattr(settings, "DOCUMENT_MAX_ARCHIVE_ENTRIES", 10000)
# Must not raise.
worker._reject_attachment_zip_bomb(str(path))
def test_non_container_suffix_is_ignored(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
path = tmp_path / "notes.txt"
path.write_bytes(b"x" * 5000)
monkeypatch.setattr(settings, "DOCUMENT_MAX_DECOMPRESSED_BYTES", 1)
# Text files are not zip containers — never inspected, never rejected.
worker._reject_attachment_zip_bomb(str(path))
def test_corrupt_zip_is_left_to_the_parser(self, tmp_path, monkeypatch):
from application import worker
from application.core.settings import settings
path = tmp_path / "broken.xlsx"
path.write_bytes(b"not a real zip")
monkeypatch.setattr(settings, "DOCUMENT_MAX_DECOMPRESSED_BYTES", 1)
# BadZipFile → return quietly; the format parser surfaces a clean error.
worker._reject_attachment_zip_bomb(str(path))