mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 10:13:06 +00:00
Graph retrieval tied plain vector search at best and never beat it. Measured
across five corpora, the bottleneck was seeding, not the graph: the walk
started from nodes whose embeddings were computed from bare entity names, and
a whole question shares almost nothing with a name like "Quill".
Extraction now embeds each node from "name (type): description" and each
relationship as the fact it asserts ("Alder streams_to Quill: ..."), stored on
a new nullable graph_edges.fact_embedding column that ensure_vector_schema adds
in place. Entity names are canonicalised (case, punctuation, word breaks and a
cautious plural) so "VECTOR_STORE" and "vector stores" land on one node. Extraction calls run
concurrently (GRAPHRAG_EXTRACTION_WORKERS, default 8) while embedding and graph
writes stay serial on the task thread, so ordering and idempotency are
unchanged; that measured 8.4x faster with identical output.
Retrieval gains per-source options, stored under retrieval.graph and read live
at query time:
- seed_strategy: start from matching entities (default) or matching
relationships, which can reach an entity the question never names;
- passage_nodes (on): walk the source's passages alongside entities, with
PageRank damping 0.5 instead of 0.85;
- blend_vector (on): fuse the graph ranking with the source's vector ranking
by reciprocal rank.
The defaults are the measured-best configuration. Through GraphRAGRetriever,
the new seeding moved recall@4 from 0.41 to 0.68 on a multi-hop corpus and
from 0.50 to 1.00 on the docs corpus, and regressed none of the corpora
measured. Existing graphs keep name-only embeddings until rebuilt.
48 lines
1.8 KiB
Python
48 lines
1.8 KiB
Python
"""Retrieval strategy and GraphRAG."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Literal, Optional
|
|
|
|
from pydantic import Field, field_validator
|
|
|
|
from docsgpt.core.settings._shared import SettingsGroup, normalize_choice
|
|
|
|
|
|
class RetrievalSettings(SettingsGroup):
|
|
"""Which vector store answers searches and how retrieval fans out across sources."""
|
|
|
|
VECTOR_STORE: Literal["faiss", "elasticsearch", "mongodb", "qdrant", "milvus", "pgvector"] = Field(
|
|
default="faiss", description="Vector store backend."
|
|
)
|
|
RETRIEVAL_MAX_PARALLEL_SOURCES: int = Field(
|
|
default=4,
|
|
ge=1,
|
|
description="Concurrent per-source searches in one retrieval; the query is embedded once and shared.",
|
|
)
|
|
PER_SOURCE_RETRIEVAL_ENABLED: bool = Field(
|
|
default=True,
|
|
description="Kill-switch for per-source retrieval dispatch; False collapses to a single retriever.",
|
|
)
|
|
GRAPHRAG_ENABLED: bool = Field(default=False, description="Gates graph-aware ingestion and retrieval.")
|
|
GRAPHRAG_EXTRACTION_MODEL: Optional[str] = Field(
|
|
default=None, description="Model for ingest-time graph extraction; unset reuses LLM_PROVIDER/LLM_NAME."
|
|
)
|
|
GRAPHRAG_MAX_CHUNKS_FOR_EXTRACTION: int = Field(
|
|
default=2000, ge=0, description="Hard cap on chunks extracted per source (cost control); 0 extracts nothing."
|
|
)
|
|
GRAPHRAG_EXTRACTION_WORKERS: int = Field(
|
|
default=8,
|
|
ge=1,
|
|
le=32,
|
|
description=(
|
|
"Concurrent extraction calls during ingest. Model calls run in parallel while "
|
|
"graph writes stay serial, so ordering and idempotency are unchanged; 1 is fully serial."
|
|
),
|
|
)
|
|
|
|
@field_validator("VECTOR_STORE", mode="before")
|
|
@classmethod
|
|
def _normalize_vector_store(cls, v):
|
|
return normalize_choice(v)
|