mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-05 04:13:25 +00:00
Follow-up review pass over the embeddings branch. - Fold an oversized header back into the body, and drop header duplication when it would leave under a quarter of the chunk budget. A header at or over max_tokens collapsed the body budget to one token, so a document became one chunk per body token, each still over the cap: a 95 KB file produced 20k chunks of 2563 tokens against a 1250 cap. Also clamp max_tokens to at least 1, as the strategy chunkers already do. - Emit a header-only document as its own chunk. With no body piece to attach it to, splitting returned nothing and the document was dropped from the index with no error and no log line. - Skip add_custom_model for a repository FastEmbed already ships. It rejects a name it knows, so configuring any of its ~30 built-ins (MiniLM, bge, e5, gte, ...) failed every embed call and every query. - Decide "the user chose this model" by comparing against the field default rather than model_fields_set, which is true for anything read from .env. Every setup script has always written EMBEDDINGS_NAME, so an upgraded remote-embeddings install inherited mpnet's 384-token window and silently clipped ~80% off every chunk. - Cut tiktoken splits at character offsets instead of decoding each token window. A multi-byte character straddling a boundary decoded to U+FFFD on both sides, destroying one character at roughly one boundary in five on CJK text -- including at the default max_tokens of 2000. - Let the re-embed script open a FAISS index whose width does not match the configured model. That mismatch is the main reason to run it, and the error recommending the script was raised by the script itself, so the advice failed on every source. - Re-embed graph_nodes.name_embedding when GraphRAG is enabled. Those vectors seed every traversal and share the chunk vectors' width, so a same-width model swap left the graph retrieving from the old space with nothing to report it. - Prefetch the models before copying the application source, so editing any file no longer re-downloads ~780 MB of artifacts on every build. - Mirror the setup.sh embedding menu into setup.ps1: granite default, legacy mpnet as an explicit option, and both engine flows updated. Windows users were otherwise stranded on mpnet with no granite path. - Drop the unused EmbeddingsWrapper.tokenizer property.
376 lines
13 KiB
Python
376 lines
13 KiB
Python
"""Comprehensive tests for application/parser/chunking.py
|
|
|
|
Covers: Chunker (init, separate_header_and_body, split_document,
|
|
classic_chunk, chunk), edge cases, token counting.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from application.parser.chunking import Chunker
|
|
from application.parser.schema.base import Document
|
|
|
|
|
|
# =====================================================================
|
|
# Chunker - Init
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestChunkerInit:
|
|
|
|
def test_default_init(self):
|
|
chunker = Chunker()
|
|
assert chunker.chunking_strategy == "classic_chunk"
|
|
assert chunker.max_tokens == 2000
|
|
assert chunker.min_tokens == 150
|
|
assert chunker.duplicate_headers is False
|
|
|
|
def test_custom_init(self):
|
|
chunker = Chunker(
|
|
chunking_strategy="classic_chunk",
|
|
max_tokens=1000,
|
|
min_tokens=50,
|
|
duplicate_headers=True,
|
|
)
|
|
assert chunker.max_tokens == 1000
|
|
assert chunker.min_tokens == 50
|
|
assert chunker.duplicate_headers is True
|
|
|
|
def test_unknown_strategy_construction_no_longer_raises(self):
|
|
# Strategy dispatch/whitelist moved to ChunkerCreator; the Chunker
|
|
# constructor itself no longer rejects an unknown strategy string.
|
|
chunker = Chunker(chunking_strategy="unknown_strategy")
|
|
assert chunker.chunking_strategy == "unknown_strategy"
|
|
|
|
|
|
# =====================================================================
|
|
# Separate Header and Body
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestSeparateHeaderAndBody:
|
|
|
|
def test_with_header(self):
|
|
chunker = Chunker()
|
|
text = "line1\nline2\nline3\nbody content here"
|
|
header, body = chunker.separate_header_and_body(text)
|
|
assert "line1" in header
|
|
assert "line2" in header
|
|
assert "line3" in header
|
|
assert "body content here" in body
|
|
|
|
def test_without_header(self):
|
|
chunker = Chunker()
|
|
text = "short"
|
|
header, body = chunker.separate_header_and_body(text)
|
|
assert header == ""
|
|
assert body == "short"
|
|
|
|
def test_empty_text(self):
|
|
chunker = Chunker()
|
|
header, body = chunker.separate_header_and_body("")
|
|
assert header == ""
|
|
assert body == ""
|
|
|
|
def test_exactly_three_lines(self):
|
|
chunker = Chunker()
|
|
text = "line1\nline2\nline3\n"
|
|
header, body = chunker.separate_header_and_body(text)
|
|
assert header == "line1\nline2\nline3\n"
|
|
assert body == ""
|
|
|
|
|
|
# =====================================================================
|
|
# Split Document
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestSplitDocument:
|
|
|
|
def test_split_large_document(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=5)
|
|
long_text = "word " * 200
|
|
doc = Document(text=long_text, doc_id="doc1")
|
|
|
|
result = chunker.split_document(doc)
|
|
assert len(result) > 1
|
|
for split_doc in result:
|
|
assert split_doc.doc_id.startswith("doc1-")
|
|
assert split_doc.extra_info is not None
|
|
assert "token_count" in split_doc.extra_info
|
|
|
|
def test_split_preserves_header_on_first(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=5, duplicate_headers=False)
|
|
text = "h1\nh2\nh3\n" + "word " * 200
|
|
doc = Document(text=text, doc_id="doc1")
|
|
|
|
result = chunker.split_document(doc)
|
|
assert len(result) > 1
|
|
assert "h1" in result[0].text
|
|
# Only the first: this is what duplicate_headers=False means.
|
|
assert all("h1" not in chunk.text for chunk in result[1:])
|
|
|
|
def test_split_duplicates_header(self):
|
|
"""Every chunk carries the header when the flag is set.
|
|
|
|
Asserting only the first chunk passed even while the flag did nothing:
|
|
the implementation cleared the header after the first iteration, so
|
|
duplicate_headers was unreachable.
|
|
"""
|
|
chunker = Chunker(max_tokens=50, min_tokens=5, duplicate_headers=True)
|
|
text = "h1\nh2\nh3\n" + "word " * 200
|
|
doc = Document(text=text, doc_id="doc1")
|
|
|
|
result = chunker.split_document(doc)
|
|
assert len(result) > 1
|
|
assert all("h1" in chunk.text for chunk in result)
|
|
|
|
def test_oversized_header_does_not_multiply_chunks(self):
|
|
"""A header past the budget used to leave a one-token body budget.
|
|
|
|
``max(1, max_tokens - header_tokens)`` bottomed out at 1, so the body
|
|
was cut into one chunk per token -- each still over ``max_tokens``,
|
|
since the whole header was prepended to it.
|
|
"""
|
|
chunker = Chunker(max_tokens=50, min_tokens=5, duplicate_headers=True)
|
|
header = ("verylongheaderword " * 40) + "h2\nh3\n"
|
|
body = "word " * 100
|
|
doc = Document(text=f"{header}\n{body}", doc_id="doc1")
|
|
|
|
result = chunker.split_document(doc)
|
|
|
|
body_tokens = chunker.counter.count(body)
|
|
assert len(result) <= body_tokens // 10, "chunk count must track the budget"
|
|
for chunk in result:
|
|
assert chunk.extra_info["token_count"] <= chunker.max_tokens
|
|
assert "".join(chunk.text for chunk in result) == doc.text
|
|
|
|
def test_header_taking_most_of_the_budget_is_not_duplicated(self):
|
|
"""Duplication is dropped when it would leave almost no room for body.
|
|
|
|
Below a quarter of the budget every chunk is mostly repeated header,
|
|
which multiplies the chunk count without adding retrievable text.
|
|
"""
|
|
chunker = Chunker(max_tokens=50, min_tokens=5, duplicate_headers=True)
|
|
header = ("headerword " * 20) + "h2\nh3\n"
|
|
doc = Document(text=f"{header}\n" + "word " * 200, doc_id="doc1")
|
|
|
|
result = chunker.split_document(doc)
|
|
|
|
assert len(result) > 1
|
|
assert "headerword" in result[0].text
|
|
assert all("headerword" not in chunk.text for chunk in result[1:])
|
|
|
|
def test_header_only_document_is_not_dropped(self):
|
|
"""The loop only ever emitted the header attached to a body piece.
|
|
|
|
With no body there was no piece, so an entire document disappeared
|
|
from the index with no error and no log line.
|
|
"""
|
|
chunker = Chunker(max_tokens=50, min_tokens=1, duplicate_headers=False)
|
|
text = "h1\nh2\nh3\n"
|
|
doc = Document(text=text, doc_id="doc1")
|
|
|
|
result = chunker.split_document(doc)
|
|
|
|
assert len(result) == 1
|
|
assert result[0].text == text
|
|
|
|
def test_header_only_document_keeps_its_text_through_chunk(self):
|
|
"""The reachable shape: three lines that alone exceed the budget."""
|
|
chunker = Chunker(max_tokens=5, min_tokens=1, duplicate_headers=False)
|
|
text = "h1\nh2\nh3\n"
|
|
|
|
result = chunker.chunk([Document(text=text, doc_id="doc1")])
|
|
|
|
assert result, "the document must not vanish"
|
|
assert "".join(chunk.text for chunk in result) == text
|
|
|
|
def test_zero_max_tokens_does_not_split_per_token(self):
|
|
chunker = Chunker(max_tokens=0, min_tokens=1)
|
|
assert chunker.max_tokens == 1
|
|
|
|
def test_split_preserves_embedding(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=5)
|
|
doc = Document(
|
|
text="word " * 200,
|
|
doc_id="doc1",
|
|
embedding=[0.1, 0.2],
|
|
)
|
|
|
|
result = chunker.split_document(doc)
|
|
for split_doc in result:
|
|
assert split_doc.embedding == [0.1, 0.2]
|
|
|
|
def test_split_preserves_extra_info(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=5)
|
|
doc = Document(
|
|
text="word " * 200,
|
|
doc_id="doc1",
|
|
extra_info={"source": "test"},
|
|
)
|
|
|
|
result = chunker.split_document(doc)
|
|
for split_doc in result:
|
|
assert split_doc.extra_info["source"] == "test"
|
|
assert "token_count" in split_doc.extra_info
|
|
|
|
|
|
# =====================================================================
|
|
# Classic Chunk
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestClassicChunk:
|
|
|
|
def test_small_doc_passes_through(self):
|
|
chunker = Chunker(max_tokens=2000, min_tokens=1)
|
|
doc = Document(text="Short text", doc_id="d1")
|
|
|
|
result = chunker.classic_chunk([doc])
|
|
assert len(result) == 1
|
|
assert result[0].extra_info is not None
|
|
assert "token_count" in result[0].extra_info
|
|
|
|
def test_large_doc_gets_split(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=5)
|
|
doc = Document(text="word " * 200, doc_id="d1")
|
|
|
|
result = chunker.classic_chunk([doc])
|
|
assert len(result) > 1
|
|
|
|
def test_medium_doc_within_range(self):
|
|
chunker = Chunker(max_tokens=2000, min_tokens=5)
|
|
doc = Document(text="Hello " * 50, doc_id="d1")
|
|
|
|
result = chunker.classic_chunk([doc])
|
|
assert len(result) == 1
|
|
|
|
def test_multiple_docs(self):
|
|
chunker = Chunker(max_tokens=2000, min_tokens=1)
|
|
docs = [
|
|
Document(text="Doc 1 content", doc_id="d1"),
|
|
Document(text="Doc 2 content", doc_id="d2"),
|
|
]
|
|
|
|
result = chunker.classic_chunk(docs)
|
|
assert len(result) == 2
|
|
|
|
def test_empty_docs_list(self):
|
|
chunker = Chunker()
|
|
result = chunker.classic_chunk([])
|
|
assert result == []
|
|
|
|
def test_very_small_doc_below_min(self):
|
|
chunker = Chunker(max_tokens=2000, min_tokens=500)
|
|
doc = Document(text="tiny", doc_id="d1")
|
|
|
|
result = chunker.classic_chunk([doc])
|
|
assert len(result) == 1
|
|
assert result[0].extra_info["token_count"] < 500
|
|
|
|
def test_existing_extra_info_preserved(self):
|
|
chunker = Chunker(max_tokens=2000, min_tokens=1)
|
|
doc = Document(
|
|
text="Hello world",
|
|
doc_id="d1",
|
|
extra_info={"source": "test"},
|
|
)
|
|
|
|
result = chunker.classic_chunk([doc])
|
|
assert result[0].extra_info["source"] == "test"
|
|
assert "token_count" in result[0].extra_info
|
|
|
|
def test_none_extra_info_initialized(self):
|
|
chunker = Chunker(max_tokens=2000, min_tokens=1)
|
|
doc = Document(text="Hello", doc_id="d1", extra_info=None)
|
|
|
|
result = chunker.classic_chunk([doc])
|
|
assert result[0].extra_info is not None
|
|
assert "token_count" in result[0].extra_info
|
|
|
|
|
|
# =====================================================================
|
|
# Chunk (dispatcher)
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestChunkDispatcher:
|
|
|
|
def test_dispatch_classic_chunk(self):
|
|
chunker = Chunker(chunking_strategy="classic_chunk")
|
|
doc = Document(text="content", doc_id="d1")
|
|
|
|
result = chunker.chunk([doc])
|
|
assert len(result) == 1
|
|
|
|
def test_chunk_runs_classic_regardless_of_strategy_attr(self):
|
|
# Chunk() now always runs the classic implementation; strategy
|
|
# selection happens at the ChunkerCreator level, not here.
|
|
chunker = Chunker()
|
|
chunker.chunking_strategy = "nonexistent"
|
|
|
|
result = chunker.chunk([Document(text="x", doc_id="d")])
|
|
assert len(result) == 1
|
|
|
|
|
|
# =====================================================================
|
|
# Integration-like test
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestChunkerIntegration:
|
|
|
|
def test_mixed_document_sizes(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=5)
|
|
docs = [
|
|
Document(text="small text", doc_id="small"),
|
|
Document(text="word " * 200, doc_id="large"),
|
|
Document(text="medium " * 20, doc_id="medium"),
|
|
]
|
|
|
|
result = chunker.chunk(docs)
|
|
# Small and medium should pass through, large should be split
|
|
assert len(result) >= 3
|
|
doc_ids = [d.doc_id for d in result]
|
|
assert "small" in doc_ids
|
|
|
|
def test_all_chunks_have_token_counts(self):
|
|
chunker = Chunker(max_tokens=50, min_tokens=1)
|
|
docs = [
|
|
Document(text="word " * 200, doc_id="big"),
|
|
Document(text="tiny", doc_id="small"),
|
|
]
|
|
|
|
result = chunker.chunk(docs)
|
|
for doc in result:
|
|
assert doc.extra_info is not None
|
|
assert "token_count" in doc.extra_info
|
|
assert doc.extra_info["token_count"] > 0
|
|
|
|
|
|
# =====================================================================
|
|
# Special-token text must chunk as ordinary text, not raise
|
|
# =====================================================================
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestSpecialTokenText:
|
|
|
|
def test_chunking_document_containing_special_token_markers(self):
|
|
# A document ABOUT LLMs contains literal <|endoftext|>; plain
|
|
# ``encode()`` raises ValueError on it and destroyed the ingest.
|
|
chunker = Chunker(chunking_strategy="classic_chunk", max_tokens=50, min_tokens=0)
|
|
doc = Document(
|
|
text="The <|endoftext|> marker separates documents. " * 30,
|
|
doc_id="d1",
|
|
)
|
|
chunks = chunker.chunk([doc])
|
|
assert chunks
|
|
assert all("<|endoftext|>" in c.text for c in chunks[:1])
|