mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 20:13:04 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
61 lines
2.0 KiB
Python
61 lines
2.0 KiB
Python
"""Tests for tool-result sanitization at the executor fan-out point.
|
|
|
|
``result_full`` fans out to the conversation row, ``tool_call_attempts``,
|
|
and (via the executor's return value) the LLM copy — sanitizing at the
|
|
source protects every lane at once.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from docsgpt.agents.tool_executor import (
|
|
RESULT_FULL_MAX_CHARS,
|
|
bound_result_full,
|
|
sanitize_tool_result,
|
|
)
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestSanitizeToolResult:
|
|
def test_strips_nul_bytes(self):
|
|
assert sanitize_tool_result("a\x00b\x00c") == "abc"
|
|
|
|
def test_strips_control_chars_keeps_whitespace(self):
|
|
assert (
|
|
sanitize_tool_result("line1\nline2\ttab\rret\x01\x08\x0b\x1f\x7fend")
|
|
== "line1\nline2\ttab\rret" + "end"
|
|
)
|
|
|
|
def test_clean_string_returned_unchanged(self):
|
|
s = "perfectly normal résumé text\nwith newlines"
|
|
assert sanitize_tool_result(s) is s
|
|
|
|
def test_recurses_into_dicts_including_keys(self):
|
|
got = sanitize_tool_result({"k\x00ey": {"inner": "v\x00al"}})
|
|
assert got == {"key": {"inner": "val"}}
|
|
|
|
def test_recurses_into_lists_and_tuples(self):
|
|
assert sanitize_tool_result(["a\x00", ("b\x00",)]) == ["a", ("b",)]
|
|
|
|
def test_non_string_scalars_pass_through(self):
|
|
for value in (42, 3.14, True, None, b"\x00raw"):
|
|
assert sanitize_tool_result(value) == value
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestBoundResultFull:
|
|
def test_short_text_unchanged(self):
|
|
assert bound_result_full("short") == "short"
|
|
|
|
def test_text_at_limit_unchanged(self):
|
|
s = "x" * RESULT_FULL_MAX_CHARS
|
|
assert bound_result_full(s) == s
|
|
|
|
def test_oversized_text_truncated_with_marker(self):
|
|
s = "x" * (RESULT_FULL_MAX_CHARS + 1000)
|
|
got = bound_result_full(s)
|
|
assert len(got) < len(s)
|
|
assert "truncated" in got
|
|
# The marker names the original size so the audit copy is honest
|
|
# about what was dropped.
|
|
assert str(len(s)) in got
|