mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-03 22:13:01 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
95 lines
2.9 KiB
Python
95 lines
2.9 KiB
Python
"""Tests for the dot-leader/whitespace table reconstruction (``ANYDOC_TABLEIZE``)."""
|
|
from docsgpt.parser.file.tableize import tableize
|
|
|
|
DOT_LEADER = """Revenues
|
|
Insurance premiums ............ 83,431 77,731
|
|
Sales and service revenues ........ 24,660 23,406
|
|
Freight rail transportation ......... 22,341 23,852
|
|
Total revenues 130,432 124,989
|
|
"""
|
|
|
|
|
|
def test_dot_leader_run_becomes_table():
|
|
out = tableize(DOT_LEADER)
|
|
assert "| Insurance premiums | 83,431 | 77,731 |" in out
|
|
assert "| Total revenues | 130,432 | 124,989 |" in out
|
|
assert "| --- |" in out
|
|
assert "Revenues" in out # heading line untouched
|
|
|
|
|
|
def test_whitespace_only_rows_convert_too():
|
|
md = "Alpha 1 2\nBeta 3 4\nGamma 5 6\n"
|
|
out = tableize(md)
|
|
assert "| Alpha | 1 | 2 |" in out
|
|
|
|
|
|
def test_currency_symbol_merges_with_number():
|
|
md = "Cash $ 1,234 900\nDebt $ 2,000 1,500\nEquity $ 900 800\n"
|
|
out = tableize(md)
|
|
assert "| Cash | $1,234 | 900 |" in out
|
|
|
|
|
|
def test_parenthesised_negatives_and_dash():
|
|
md = "Losses (30) —\nGains (12) —\nNet (42) —\n"
|
|
out = tableize(md)
|
|
assert "| Losses | (30) | — |" in out
|
|
|
|
|
|
def test_short_run_is_left_alone():
|
|
md = "Alpha 1 2\nBeta 3 4\n"
|
|
assert "|" not in tableize(md)
|
|
|
|
|
|
def test_mixed_widths_are_left_alone():
|
|
md = "Alpha 1 2\nBeta 3\nGamma 5 6\n"
|
|
assert "|" not in tableize(md)
|
|
|
|
|
|
def test_prose_with_numbers_is_left_alone():
|
|
md = "The division shipped 47 releases in 2023.\nIt hired 12 engineers this year alone.\nRevenue grew by a factor of 3 since 2019.\n"
|
|
assert "|" not in tableize(md)
|
|
|
|
|
|
def test_min_rows_boundary():
|
|
two = "A 1 2\nB 3 4\n"
|
|
three = two + "C 5 6\n"
|
|
assert "|" not in tableize(two, min_rows=3)
|
|
assert "|" in tableize(three, min_rows=3)
|
|
|
|
|
|
def test_pipe_in_label_is_escaped():
|
|
md = "Assets | current 1,234 900\nDebt 2,000 1,500\nEquity 900 800\n"
|
|
out = tableize(md)
|
|
assert "| Assets \\| current | 1,234 | 900 |" in out
|
|
|
|
|
|
def test_dollar_prefixed_word_is_not_a_value():
|
|
md = "Alpha $TBD 2\nBeta $TBD 4\nGamma $TBD 6\n"
|
|
assert "|" not in tableize(md)
|
|
|
|
|
|
def test_non_table_text_passes_through_verbatim():
|
|
"""Only the trailing newline may differ (splitlines/join round-trip)."""
|
|
md = "# Heading\n\nA paragraph with no numbers.\n\n- a list item\n"
|
|
assert tableize(md) == md.rstrip("\n")
|
|
|
|
|
|
def test_single_trailing_number_lines_are_not_a_table():
|
|
"""Headings, footnotes and version lists look like 'word number' rows; leave them alone."""
|
|
from docsgpt.parser.file.tableize import tableize
|
|
|
|
for block in (
|
|
"Chapter 1\nChapter 2\nChapter 3",
|
|
"Footnote 1\nFootnote 2\nFootnote 3",
|
|
"ISO 9001\nISO 14001\nISO 27001",
|
|
"Version 2.0\nVersion 3.0\nVersion 4.0",
|
|
):
|
|
assert tableize(block) == block
|
|
|
|
|
|
def test_leader_rows_with_one_value_still_convert():
|
|
from docsgpt.parser.file.tableize import tableize
|
|
|
|
block = "Revenue ...... 1,234\nCosts ...... 567\nProfit ...... 667"
|
|
assert "| --- |" in tableize(block)
|