mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 12:13:05 +00:00
Rewrite the default/creative/strict presets (classic + agentic) into
structured sections: grounding and cite-by-title guidance, insufficient-
context behavior, current date, respond-in-user-language, scoped mermaid
usage, an untrusted-content guardrail, and a conditional XML-tagged
document context block. A memory directory listing is injected at render
time via the template prefetch mechanism so the model starts oriented
without burning a tool call.
Fixes along the way:
- Agentic preset swap was dead code: _get_prompt_content cached the
classic preset before create_agent's swap check ran, so agentic and
research agents always got the classic preset. The swap now happens
inside _get_prompt_content.
- Jinja autoescape corrupted document content in custom prompts
(< -> <); prompts are not HTML, autoescape is now off.
- Literal {summaries} leaked into the prompt when no docs were
retrieved; the placeholder is now stripped.
- Agentic/research prompts referenced tool names from a dropped naming
scheme (search_internal, reason_think); they now reference the real
names (search, reason).
- The strict preset told the model to "be very creative and use your
imagination" right after "never make up information".
- extract_tool_usages recorded intermediate attribute chains as
bare-tool usages, which meant "run all actions" at prefetch; only
maximal chains are recorded now.
- Headless runs retrieved docs but never rendered them into the
prompt; the prompt is now rendered like the streaming path.
- Default tools were unreachable by name in prompt templates
(prefetch results were keyed by synthetic id only); defaults now
claim the name key unless an explicit row shadows it.
Tool layer: memory/notes/todo actions are namespaced (memory_view,
note_overwrite, todo_create, ...) with legacy unprefixed names still
accepted via prefix stripping; duplicate action names across tools are
disambiguated with the owning tool's name instead of numeric suffixes;
thin tool descriptions rewritten (brave, duckduckgo, telegram, ntfy,
cryptoprice, read_webpage, internal_search, think).
Docs are now wrapped per chunk in <document index>/<source>/<content>
tags for citation-by-title support.
119 lines
4.2 KiB
Python
119 lines
4.2 KiB
Python
import logging
|
|
from typing import Any, Dict, Optional
|
|
|
|
from application.templates.namespaces import NamespaceManager
|
|
|
|
from application.templates.template_engine import TemplateEngine, TemplateRenderError
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
def format_docs_for_prompt(docs: Optional[list]) -> Optional[str]:
|
|
"""Format retrieved chunks as XML-tagged documents for prompt injection.
|
|
|
|
Each chunk is wrapped in a ``<document index="n">`` block with a
|
|
``<source>`` subtag (when a filename/title is known) so the model can
|
|
tell chunks apart and cite them by name.
|
|
"""
|
|
if not docs:
|
|
return None
|
|
parts = []
|
|
for i, doc in enumerate(docs, start=1):
|
|
source = doc.get("filename") or doc.get("title") or doc.get("source")
|
|
lines = [f'<document index="{i}">']
|
|
if source:
|
|
lines.append(f"<source>{source}</source>")
|
|
lines.append(f"<content>\n{doc.get('text', '')}\n</content>")
|
|
lines.append("</document>")
|
|
parts.append("\n".join(lines))
|
|
return "\n\n".join(parts)
|
|
|
|
|
|
class PromptRenderer:
|
|
"""Service for rendering prompts with dynamic context using namespaces"""
|
|
|
|
def __init__(self):
|
|
self.template_engine = TemplateEngine()
|
|
self.namespace_manager = NamespaceManager()
|
|
|
|
def render_prompt(
|
|
self,
|
|
prompt_content: str,
|
|
user_id: Optional[str] = None,
|
|
request_id: Optional[str] = None,
|
|
passthrough_data: Optional[Dict[str, Any]] = None,
|
|
docs: Optional[list] = None,
|
|
docs_together: Optional[str] = None,
|
|
tools_data: Optional[Dict[str, Any]] = None,
|
|
**kwargs,
|
|
) -> str:
|
|
"""
|
|
Render prompt with full context from all namespaces.
|
|
|
|
Args:
|
|
prompt_content: Raw prompt template string
|
|
user_id: Current user identifier
|
|
request_id: Unique request identifier
|
|
passthrough_data: Parameters from web request
|
|
docs: RAG retrieved documents
|
|
docs_together: Concatenated document content
|
|
tools_data: Pre-fetched tool results organized by tool name
|
|
**kwargs: Additional parameters for namespace builders
|
|
|
|
Returns:
|
|
Rendered prompt string with all variables substituted
|
|
|
|
Raises:
|
|
TemplateRenderError: If template rendering fails
|
|
"""
|
|
if not prompt_content:
|
|
return ""
|
|
|
|
uses_template = self._uses_template_syntax(prompt_content)
|
|
|
|
if not uses_template:
|
|
return self._apply_legacy_substitutions(prompt_content, docs_together)
|
|
|
|
try:
|
|
context = self.namespace_manager.build_context(
|
|
user_id=user_id,
|
|
request_id=request_id,
|
|
passthrough_data=passthrough_data,
|
|
docs=docs,
|
|
docs_together=docs_together,
|
|
tools_data=tools_data,
|
|
**kwargs,
|
|
)
|
|
|
|
return self.template_engine.render(prompt_content, context)
|
|
except TemplateRenderError:
|
|
raise
|
|
except Exception as e:
|
|
error_msg = f"Prompt rendering failed: {str(e)}"
|
|
logger.error(error_msg)
|
|
raise TemplateRenderError(error_msg) from e
|
|
|
|
def _uses_template_syntax(self, prompt_content: str) -> bool:
|
|
"""Check if prompt uses Jinja2 template syntax"""
|
|
return "{{" in prompt_content and "}}" in prompt_content
|
|
|
|
def _apply_legacy_substitutions(
|
|
self, prompt_content: str, docs_together: Optional[str] = None
|
|
) -> str:
|
|
"""
|
|
Apply backward-compatible substitutions for old prompt format.
|
|
|
|
Handles the legacy {summaries} placeholder. When no documents were
|
|
retrieved the placeholder is removed so the model never sees the
|
|
raw template artifact.
|
|
"""
|
|
return prompt_content.replace("{summaries}", docs_together or "")
|
|
|
|
def validate_template(self, prompt_content: str) -> bool:
|
|
"""Validate prompt template syntax"""
|
|
return self.template_engine.validate_template(prompt_content)
|
|
|
|
def extract_variables(self, prompt_content: str) -> set[str]:
|
|
"""Extract all variable names from prompt template"""
|
|
return self.template_engine.extract_variables(prompt_content)
|