Files
DocsGPT/tests/test_prompt_presets.py
T
Alex 636d4d35d1 feat(prompts): revamp preset prompts, tool naming, and prompt templating
Rewrite the default/creative/strict presets (classic + agentic) into
structured sections: grounding and cite-by-title guidance, insufficient-
context behavior, current date, respond-in-user-language, scoped mermaid
usage, an untrusted-content guardrail, and a conditional XML-tagged
document context block. A memory directory listing is injected at render
time via the template prefetch mechanism so the model starts oriented
without burning a tool call.

Fixes along the way:
- Agentic preset swap was dead code: _get_prompt_content cached the
  classic preset before create_agent's swap check ran, so agentic and
  research agents always got the classic preset. The swap now happens
  inside _get_prompt_content.
- Jinja autoescape corrupted document content in custom prompts
  (< -> &lt;); prompts are not HTML, autoescape is now off.
- Literal {summaries} leaked into the prompt when no docs were
  retrieved; the placeholder is now stripped.
- Agentic/research prompts referenced tool names from a dropped naming
  scheme (search_internal, reason_think); they now reference the real
  names (search, reason).
- The strict preset told the model to "be very creative and use your
  imagination" right after "never make up information".
- extract_tool_usages recorded intermediate attribute chains as
  bare-tool usages, which meant "run all actions" at prefetch; only
  maximal chains are recorded now.
- Headless runs retrieved docs but never rendered them into the
  prompt; the prompt is now rendered like the streaming path.
- Default tools were unreachable by name in prompt templates
  (prefetch results were keyed by synthetic id only); defaults now
  claim the name key unless an explicit row shadows it.

Tool layer: memory/notes/todo actions are namespaced (memory_view,
note_overwrite, todo_create, ...) with legacy unprefixed names still
accepted via prefix stripping; duplicate action names across tools are
disambiguated with the owning tool's name instead of numeric suffixes;
thin tool descriptions rewritten (brave, duckduckgo, telegram, ntfy,
cryptoprice, read_webpage, internal_search, think).

Docs are now wrapped per chunk in <document index>/<source>/<content>
tags for citation-by-title support.
2026-06-11 11:18:46 +01:00

139 lines
4.8 KiB
Python

"""Preset prompt templates: rendering, structure, and tool-name accuracy.
Guards against regressions like the strict preset carrying creative
language, prompts referencing tool names that don't exist, and the
``{summaries}`` placeholder leaking into the model-visible prompt.
"""
from datetime import datetime, timezone
from pathlib import Path
import pytest
from application.api.answer.services.prompt_renderer import (
PromptRenderer,
format_docs_for_prompt,
)
PROMPTS_DIR = Path(__file__).resolve().parents[1] / "application" / "prompts"
CLASSIC_PRESETS = [
"chat_combine_default.txt",
"chat_combine_creative.txt",
"chat_combine_strict.txt",
]
AGENTIC_PRESETS = [
"agentic/default.txt",
"agentic/creative.txt",
"agentic/strict.txt",
]
DOCS = [
{"text": "The refund window is 30 days.", "filename": "policy.pdf"},
{"text": "Contact support@acme.test", "title": "handbook"},
]
def _read(preset: str) -> str:
return (PROMPTS_DIR / preset).read_text()
@pytest.mark.unit
class TestClassicPresets:
@pytest.mark.parametrize("preset", CLASSIC_PRESETS)
def test_renders_with_docs(self, preset):
renderer = PromptRenderer()
docs_together = format_docs_for_prompt(DOCS)
result = renderer.render_prompt(
_read(preset), docs=DOCS, docs_together=docs_together
)
assert "The refund window is 30 days." in result
assert "policy.pdf" in result
assert "<documents>" in result
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
assert today in result
for artifact in ("{summaries}", "{{", "{%"):
assert artifact not in result
@pytest.mark.parametrize("preset", CLASSIC_PRESETS)
def test_renders_without_docs(self, preset):
renderer = PromptRenderer()
result = renderer.render_prompt(_read(preset))
assert "<documents>" not in result
assert "No document context was retrieved" in result
for artifact in ("{summaries}", "{{", "{%"):
assert artifact not in result
def test_strict_has_no_creative_language(self):
content = _read("chat_combine_strict.txt").lower()
assert "imagination" not in content
assert "creative" not in content
@pytest.mark.unit
class TestAgenticPresets:
@pytest.mark.parametrize("preset", AGENTIC_PRESETS)
def test_renders_clean(self, preset):
renderer = PromptRenderer()
result = renderer.render_prompt(_read(preset))
for artifact in ("{summaries}", "{{", "{%"):
assert artifact not in result
@pytest.mark.parametrize("preset", AGENTIC_PRESETS + ["research/step.txt"])
def test_references_real_tool_names(self, preset):
# LLM-visible action names are ``search`` / ``list_files`` /
# ``reason`` (see ToolExecutor.prepare_tools_for_llm); the old
# ``{action}_{tool}`` names must not reappear.
content = _read(preset)
assert "search_internal" not in content
assert "reason_think" not in content
def test_agentic_strict_has_no_creative_language(self):
content = _read("agentic/strict.txt").lower()
assert "imagination" not in content
assert "be creative" not in content
@pytest.mark.unit
class TestMemorySection:
MEMORY_TOOLS_DATA = {
"memory": {"memory_view": "Directory: /\n- preferences.md\n- projects/"}
}
@pytest.mark.parametrize("preset", CLASSIC_PRESETS + AGENTIC_PRESETS)
def test_renders_when_memory_prefetched(self, preset):
renderer = PromptRenderer()
result = renderer.render_prompt(
_read(preset), tools_data=self.MEMORY_TOOLS_DATA
)
assert "## Memory" in result
assert "- preferences.md" in result
assert "<memory_directory>" in result
@pytest.mark.parametrize("preset", CLASSIC_PRESETS + AGENTIC_PRESETS)
def test_absent_without_memory_data(self, preset):
renderer = PromptRenderer()
result = renderer.render_prompt(_read(preset))
assert "## Memory" not in result
assert "<memory_directory>" not in result
@pytest.mark.unit
class TestFormatDocsForPrompt:
def test_wraps_each_doc_with_index_and_source(self):
out = format_docs_for_prompt(DOCS)
assert '<document index="1">' in out
assert "<source>policy.pdf</source>" in out
assert '<document index="2">' in out
assert "<source>handbook</source>" in out
assert "The refund window is 30 days." in out
def test_doc_without_source_omits_tag(self):
out = format_docs_for_prompt([{"text": "anonymous chunk"}])
assert "<source>" not in out
assert "anonymous chunk" in out
def test_empty_returns_none(self):
assert format_docs_for_prompt([]) is None
assert format_docs_for_prompt(None) is None