Files
Alex 574f96341e refactor: rename the application package to docsgpt
The backend import package is now docsgpt, the name it will carry on PyPI;
application was far too generic to install into anyone's site-packages.
git mv plus a mechanical rewrite of every import, dotted string and path
reference: 734 Python files, the compose files, Dockerfile, workflows, docs,
setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage
config, .gitignore. Behaviour is unchanged.

Kept for one release:
- A top-level application package whose meta-path finder resolves
  application.x.y to the already-imported docsgpt.x.y object, so old imports
  and entry points (celery -A application.app.celery,
  uvicorn application.asgi:asgi_app) keep working with a FutureWarning.
- Celery registers every application.* task name as an alias of its
  docsgpt.* task on start-up, so messages queued by the previous release still
  run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries
  the previous release wrote are left unread instead of firing twice.

The backend image builds from the repository root (docker build -f
docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore
allow-lists docsgpt/ and application/ and keeps caches, local data, .env
files, the sample index files and the Dockerfile out. Compose and the image
workflows point at the new context.
2026-09-07 10:20:43 +01:00

108 lines
3.6 KiB
Python

import pytest
from docsgpt.parser.schema.schema import BaseDocument
class ConcreteDoc(BaseDocument):
@classmethod
def get_type(cls) -> str:
return "test"
@pytest.mark.unit
class TestBaseDocument:
def test_get_text(self):
doc = ConcreteDoc(text="hello")
assert doc.get_text() == "hello"
def test_get_text_raises_when_none(self):
doc = ConcreteDoc()
with pytest.raises(ValueError, match="text field not set"):
doc.get_text()
def test_get_doc_id(self):
doc = ConcreteDoc(text="x", doc_id="doc1")
assert doc.get_doc_id() == "doc1"
def test_get_doc_id_raises_when_none(self):
doc = ConcreteDoc(text="x")
with pytest.raises(ValueError, match="doc_id not set"):
doc.get_doc_id()
def test_is_doc_id_none(self):
doc = ConcreteDoc(text="x")
assert doc.is_doc_id_none is True
def test_is_doc_id_not_none(self):
doc = ConcreteDoc(text="x", doc_id="y")
assert doc.is_doc_id_none is False
def test_get_embedding(self):
doc = ConcreteDoc(text="x", embedding=[1.0, 2.0])
assert doc.get_embedding() == [1.0, 2.0]
def test_get_embedding_raises_when_none(self):
doc = ConcreteDoc(text="x")
with pytest.raises(ValueError, match="embedding not set"):
doc.get_embedding()
def test_extra_info_str(self):
doc = ConcreteDoc(text="x", extra_info={"key": "value", "num": 42})
result = doc.extra_info_str
assert "key: value" in result
assert "num: 42" in result
def test_extra_info_str_none(self):
doc = ConcreteDoc(text="x")
assert doc.extra_info_str is None
# =====================================================================
# Coverage gap tests for docsgpt/parser/schema/base.py (lines 19, 27, 34)
# =====================================================================
@pytest.mark.unit
class TestDocumentBase:
def test_document_post_init_raises_on_none_text(self):
"""Cover line 19: Document.__post_init__ raises ValueError for None text."""
from docsgpt.parser.schema.base import Document
with pytest.raises(ValueError, match="text field not set"):
Document(text=None)
def test_document_to_vector_format(self):
"""Cover line 27: Document.to_vector_format converts correctly."""
from docsgpt.parser.schema.base import Document
doc = Document(text="hello world", extra_info={"source": "test"})
lc_doc = doc.to_vector_format()
assert lc_doc.page_content == "hello world"
assert lc_doc.metadata == {"source": "test"}
def test_document_to_vector_format_no_extra_info(self):
"""Cover: to_vector_format with no extra_info uses empty dict."""
from docsgpt.parser.schema.base import Document
doc = Document(text="hello")
lc_doc = doc.to_vector_format()
assert lc_doc.metadata == {}
def test_document_from_vector_format(self):
"""Cover line 34: Document.from_vector_format creates Document."""
from docsgpt.parser.schema.base import Document
from docsgpt.vectorstore.document_class import Document as LCDocument
lc_doc = LCDocument(page_content="test content", metadata={"key": "val"})
doc = Document.from_vector_format(lc_doc)
assert doc.text == "test content"
assert doc.extra_info == {"key": "val"}
def test_document_get_type(self):
"""Cover line 24: Document.get_type returns 'Document'."""
from docsgpt.parser.schema.base import Document
assert Document.get_type() == "Document"