mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 00:13:14 +00:00
Query-side: composes ClassicRAG (rephrase/budget/fallback). Per source: pgvector entity-name seed -> bounded subgraph -> networkx personalized PageRank (IDF hub down-weight) -> rank graph_node_chunks -> chunk text -> shared token budget; no LLM at query time. Citations derived from chunk metadata (matches ClassicRAG). Falls back to ClassicRAG per-source when no graph rows / unavailable / error (retrieval never breaks). get_chunk_texts queries the co-located documents table by configured table/column names (parameterized). Registered 'graphrag'. Unit G5.
26 lines
884 B
Python
26 lines
884 B
Python
from application.retriever.classic_rag import ClassicRAG
|
|
from application.retriever.graph_rag import GraphRAGRetriever
|
|
from application.retriever.hybrid_rag import HybridRetriever
|
|
|
|
|
|
class RetrieverCreator:
|
|
retrievers = {
|
|
"classic": ClassicRAG,
|
|
"default": ClassicRAG,
|
|
"hybrid": HybridRetriever,
|
|
"graphrag": GraphRAGRetriever,
|
|
}
|
|
|
|
@classmethod
|
|
def create_retriever(cls, type, *args, **kwargs):
|
|
retriever_type = (type or "default").lower()
|
|
retiever_class = cls.retrievers.get(retriever_type)
|
|
if not retiever_class:
|
|
raise ValueError(f"No retievers class found for type {type}")
|
|
return retiever_class(*args, **kwargs)
|
|
|
|
@classmethod
|
|
def register(cls, key, retriever_class):
|
|
"""Register ``retriever_class`` under ``key`` (idempotent)."""
|
|
cls.retrievers[key] = retriever_class
|