mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 12:13:05 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
396 lines
14 KiB
Python
396 lines
14 KiB
Python
import logging
|
|
import re
|
|
import uuid
|
|
from abc import ABC, abstractmethod
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any, Dict, Optional, Tuple
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_PLATFORM_PARTIAL_PATH = (
|
|
Path(__file__).resolve().parents[1] / "prompts" / "partials" / "platform_capabilities.txt"
|
|
)
|
|
_platform_partial_content: Optional[str] = None
|
|
|
|
|
|
class NamespaceBuilder(ABC):
|
|
"""Base class for building template context namespaces"""
|
|
|
|
@abstractmethod
|
|
def build(self, **kwargs) -> Dict[str, Any]:
|
|
"""Build namespace context dictionary"""
|
|
pass
|
|
|
|
@property
|
|
@abstractmethod
|
|
def namespace_name(self) -> str:
|
|
"""Name of this namespace for template access"""
|
|
pass
|
|
|
|
|
|
class SystemNamespace(NamespaceBuilder):
|
|
"""System metadata namespace: {{ system.* }}"""
|
|
|
|
@property
|
|
def namespace_name(self) -> str:
|
|
return "system"
|
|
|
|
def build(
|
|
self,
|
|
request_id: Optional[str] = None,
|
|
user_id: Optional[str] = None,
|
|
persona: Optional[str] = None,
|
|
**kwargs,
|
|
) -> Dict[str, Any]:
|
|
"""
|
|
Build system context with metadata.
|
|
|
|
Args:
|
|
request_id: Unique request identifier
|
|
user_id: Current user identifier
|
|
|
|
Returns:
|
|
Dictionary with system variables
|
|
"""
|
|
now = datetime.now(timezone.utc)
|
|
api_base_url, platform = self._platform_info()
|
|
|
|
return {
|
|
"date": now.strftime("%Y-%m-%d"),
|
|
"time": now.strftime("%H:%M:%S"),
|
|
"timestamp": now.isoformat(),
|
|
"request_id": request_id or str(uuid.uuid4()),
|
|
"user_id": user_id,
|
|
"api_base_url": api_base_url,
|
|
"platform": platform,
|
|
# Operator-authored persona, injected as a VALUE: braces inside it
|
|
# are inert, so a custom prompt can never break the skeleton or
|
|
# reach the template sandbox.
|
|
"persona": persona,
|
|
}
|
|
|
|
@staticmethod
|
|
def _platform_info() -> Tuple[Optional[str], str]:
|
|
"""Return (public API base URL, rendered platform-capabilities block).
|
|
|
|
The block is rendered from ``prompts/partials/platform_capabilities.txt``
|
|
so the preset prompts can reference a single source via
|
|
``{{ system.platform }}``; failures degrade to an empty block rather
|
|
than breaking prompt rendering.
|
|
|
|
Concrete endpoint URLs are emitted only when ``PUBLIC_API_BASE_URL``
|
|
is set. ``API_URL`` is deliberately not a fallback: stock compose
|
|
deployments point it at an internal hostname (http://backend:7091)
|
|
that would otherwise be advertised to end users.
|
|
"""
|
|
from docsgpt.core.settings import settings
|
|
from docsgpt.templates.template_engine import TemplateEngine
|
|
|
|
global _platform_partial_content
|
|
base = settings.PUBLIC_API_BASE_URL
|
|
api_base_url = (
|
|
(base.strip().rstrip("/") or None)
|
|
if isinstance(base, str) and base.strip()
|
|
else None
|
|
)
|
|
try:
|
|
if _platform_partial_content is None:
|
|
_platform_partial_content = _PLATFORM_PARTIAL_PATH.read_text(encoding="utf-8")
|
|
platform = TemplateEngine().render(
|
|
_platform_partial_content, {"api_base_url": api_base_url}
|
|
).strip()
|
|
except Exception as e:
|
|
logger.warning(
|
|
f"Failed to build platform capabilities block: {e}", exc_info=True
|
|
)
|
|
platform = ""
|
|
return api_base_url, platform
|
|
|
|
|
|
class PassthroughNamespace(NamespaceBuilder):
|
|
"""Request parameters namespace: {{ passthrough.* }}"""
|
|
|
|
@property
|
|
def namespace_name(self) -> str:
|
|
return "passthrough"
|
|
|
|
def build(
|
|
self, passthrough_data: Optional[Dict[str, Any]] = None, **kwargs
|
|
) -> Dict[str, Any]:
|
|
"""
|
|
Build passthrough context from request parameters.
|
|
|
|
Args:
|
|
passthrough_data: Dictionary of parameters from web request
|
|
|
|
Returns:
|
|
Dictionary with passthrough variables
|
|
"""
|
|
if not passthrough_data:
|
|
return {}
|
|
safe_data = {}
|
|
for key, value in passthrough_data.items():
|
|
if isinstance(value, (str, int, float, bool, type(None))):
|
|
safe_data[key] = value
|
|
else:
|
|
logger.warning(
|
|
f"Skipping non-serializable passthrough value for key '{key}': {type(value)}"
|
|
)
|
|
return safe_data
|
|
|
|
|
|
class SourceNamespace(NamespaceBuilder):
|
|
"""RAG source documents namespace: {{ source.* }}"""
|
|
|
|
@property
|
|
def namespace_name(self) -> str:
|
|
return "source"
|
|
|
|
def build(
|
|
self, docs: Optional[list] = None, docs_together: Optional[str] = None, **kwargs
|
|
) -> Dict[str, Any]:
|
|
"""
|
|
Build source context from RAG retrieval results.
|
|
|
|
Args:
|
|
docs: List of retrieved documents
|
|
docs_together: Concatenated document content (for backward compatibility)
|
|
|
|
Returns:
|
|
Dictionary with source variables
|
|
"""
|
|
context = {}
|
|
|
|
if docs:
|
|
context["documents"] = docs
|
|
context["count"] = len(docs)
|
|
if docs_together:
|
|
context["docs_together"] = docs_together # Add docs_together for custom templates
|
|
context["content"] = docs_together
|
|
context["summaries"] = docs_together
|
|
return context
|
|
|
|
|
|
class ToolsNamespace(NamespaceBuilder):
|
|
"""Pre-executed tools namespace: {{ tools.* }}"""
|
|
|
|
@property
|
|
def namespace_name(self) -> str:
|
|
return "tools"
|
|
|
|
def build(
|
|
self,
|
|
tools_data: Optional[Dict[str, Any]] = None,
|
|
enabled_tools: Optional[Any] = None,
|
|
**kwargs,
|
|
) -> Dict[str, Any]:
|
|
"""
|
|
Build tools context: pre-executed tool results plus the enabled-tool set.
|
|
|
|
Args:
|
|
tools_data: Dictionary of pre-fetched tool results organized by tool name
|
|
e.g., {"memory": {"notes": "content", "tasks": "list"}}
|
|
enabled_tools: Names of the tools enabled for this turn. When provided,
|
|
exposed as the reserved ``tools.enabled`` list for gating
|
|
tool-specific prompt sections. Left absent (not just empty)
|
|
when the caller did not compute it, so a gate can fail open.
|
|
|
|
Returns:
|
|
Dictionary with tool results by tool name, plus ``enabled`` when known
|
|
"""
|
|
safe_data: Dict[str, Any] = {}
|
|
for tool_name, tool_result in (tools_data or {}).items():
|
|
if isinstance(tool_result, (str, dict, list, int, float, bool, type(None))):
|
|
safe_data[tool_name] = tool_result
|
|
else:
|
|
logger.warning(
|
|
f"Skipping non-serializable tool result for '{tool_name}': {type(tool_result)}"
|
|
)
|
|
if enabled_tools is not None:
|
|
safe_data["enabled"] = sorted({str(t) for t in enabled_tools if t})
|
|
return safe_data
|
|
|
|
|
|
# Artifact-reference metadata is the only thing that may enter template context;
|
|
# bytes (and anything non-primitive) are never exposed through this namespace.
|
|
_ARTIFACT_METADATA_KEYS = (
|
|
"artifact_id",
|
|
"version",
|
|
"mime_type",
|
|
"filename",
|
|
"size",
|
|
"kind",
|
|
"title",
|
|
)
|
|
|
|
|
|
def _artifact_view(ref: Any) -> Optional[Dict[str, Any]]:
|
|
"""Project an artifact reference to a serializable metadata view, or None if it isn't one."""
|
|
if not isinstance(ref, dict) or not ref.get("artifact_id"):
|
|
return None
|
|
view: Dict[str, Any] = {}
|
|
for key in _ARTIFACT_METADATA_KEYS:
|
|
value = ref.get(key)
|
|
if isinstance(value, (str, int, float, bool, type(None))):
|
|
view[key] = value
|
|
# ``{{ artifacts.<name>.id }}`` is the documented accessor; mirror artifact_id to id.
|
|
view["id"] = view.get("artifact_id")
|
|
return view
|
|
|
|
|
|
class ArtifactsNamespace(NamespaceBuilder):
|
|
"""Artifact references namespace: {{ artifacts.<name>.id }} / .mime_type / .filename + artifact(id)."""
|
|
|
|
@property
|
|
def namespace_name(self) -> str:
|
|
return "artifacts"
|
|
|
|
def build(
|
|
self,
|
|
artifacts_data: Optional[Dict[str, Any]] = None,
|
|
artifact_parent: Optional[Dict[str, Any]] = None,
|
|
**kwargs,
|
|
) -> Dict[str, Any]:
|
|
"""Build the artifacts namespace from named references plus a parent-scoped ``artifact(id)`` lookup.
|
|
|
|
Args:
|
|
artifacts_data: Map of output-variable name -> artifact reference dict.
|
|
artifact_parent: Parent scope ({"conversation_id"|"workflow_run_id"}) for ``artifact(id)``.
|
|
|
|
Returns:
|
|
A dict of named metadata views plus an ``artifact`` callable; never any bytes.
|
|
"""
|
|
context: Dict[str, Any] = {}
|
|
for name, ref in (artifacts_data or {}).items():
|
|
view = _artifact_view(ref)
|
|
if view is not None:
|
|
context[str(name)] = view
|
|
context["artifact"] = self._make_lookup(artifact_parent or {})
|
|
return context
|
|
|
|
def _make_lookup(self, parent: Dict[str, Any]):
|
|
"""Return an ``artifact(id)`` helper that resolves metadata scoped to the run/conversation parent."""
|
|
conversation_id = parent.get("conversation_id")
|
|
workflow_run_id = parent.get("workflow_run_id")
|
|
|
|
def artifact(artifact_id: Any) -> Dict[str, Any]:
|
|
"""Resolve one artifact's metadata by id, scoped to this parent (never cross-tenant)."""
|
|
if not artifact_id or (conversation_id is None and workflow_run_id is None):
|
|
return {}
|
|
try:
|
|
from docsgpt.storage.db.repositories.artifacts import (
|
|
ArtifactsRepository,
|
|
)
|
|
from docsgpt.storage.db.session import db_readonly
|
|
|
|
with db_readonly() as conn:
|
|
repo = ArtifactsRepository(conn)
|
|
row = repo.get_artifact_in_parent(
|
|
str(artifact_id),
|
|
conversation_id=conversation_id,
|
|
workflow_run_id=workflow_run_id,
|
|
)
|
|
if row is None:
|
|
return {}
|
|
version = repo.get_version(str(artifact_id), row.get("current_version", 1))
|
|
except Exception:
|
|
logger.exception("Failed to resolve artifact %s in namespace", artifact_id)
|
|
return {}
|
|
view = {
|
|
"artifact_id": str(row["id"]),
|
|
"id": str(row["id"]),
|
|
"version": row.get("current_version"),
|
|
"kind": row.get("kind"),
|
|
"title": row.get("title"),
|
|
}
|
|
if version is not None:
|
|
for key in ("mime_type", "filename", "size"):
|
|
value = version.get(key)
|
|
if isinstance(value, (str, int, float, bool, type(None))):
|
|
view[key] = value
|
|
return view
|
|
|
|
return artifact
|
|
|
|
|
|
class AttachmentsNamespace(NamespaceBuilder):
|
|
"""Attached-file metadata namespace: {{ attachments.files }} (name / mime_type / size)."""
|
|
|
|
@property
|
|
def namespace_name(self) -> str:
|
|
return "attachments"
|
|
|
|
def build(self, attachments: Optional[list] = None, **kwargs) -> Dict[str, Any]:
|
|
"""Expose metadata for files attached to this message.
|
|
|
|
Only the filename, MIME type, and size enter the prompt; file bytes and
|
|
any extracted ``content`` are never surfaced through this namespace. Field
|
|
names mirror the ``artifacts`` namespace (``filename`` / ``mime_type`` / ``size``).
|
|
"""
|
|
if not attachments:
|
|
return {}
|
|
files = []
|
|
for att in attachments:
|
|
if not isinstance(att, dict):
|
|
continue
|
|
filename = att.get("filename") or att.get("name")
|
|
if not filename:
|
|
continue
|
|
# Filenames are user-controlled and rendered unescaped into the prompt;
|
|
# strip control chars/newlines and cap length so a crafted name cannot
|
|
# inject fake markdown sections regardless of the upstream upload path.
|
|
clean = re.sub(r"[\x00-\x1f\x7f]", " ", str(filename))[:255].strip()
|
|
if not clean:
|
|
continue
|
|
entry: Dict[str, Any] = {
|
|
"filename": clean,
|
|
"mime_type": att.get("mime_type") or "application/octet-stream",
|
|
}
|
|
size = att.get("size")
|
|
if isinstance(size, int) and size > 0:
|
|
entry["size"] = size
|
|
files.append(entry)
|
|
return {"files": files} if files else {}
|
|
|
|
|
|
class NamespaceManager:
|
|
"""Manages all namespace builders and context assembly"""
|
|
|
|
def __init__(self):
|
|
self._builders = {
|
|
"system": SystemNamespace(),
|
|
"passthrough": PassthroughNamespace(),
|
|
"source": SourceNamespace(),
|
|
"tools": ToolsNamespace(),
|
|
"artifacts": ArtifactsNamespace(),
|
|
"attachments": AttachmentsNamespace(),
|
|
}
|
|
|
|
def build_context(self, **kwargs) -> Dict[str, Any]:
|
|
"""
|
|
Build complete template context from all namespaces.
|
|
|
|
Args:
|
|
**kwargs: Parameters to pass to namespace builders
|
|
|
|
Returns:
|
|
Complete context dictionary for template rendering
|
|
"""
|
|
context = {}
|
|
|
|
for namespace_name, builder in self._builders.items():
|
|
try:
|
|
namespace_context = builder.build(**kwargs)
|
|
# Always include namespace, even if empty, to prevent undefined errors
|
|
context[namespace_name] = namespace_context if namespace_context else {}
|
|
except Exception as e:
|
|
logger.error(f"Failed to build {namespace_name} namespace: {str(e)}")
|
|
# Include empty namespace on error to prevent template failures
|
|
context[namespace_name] = {}
|
|
return context
|
|
|
|
def get_builder(self, namespace_name: str) -> Optional[NamespaceBuilder]:
|
|
"""Get specific namespace builder"""
|
|
return self._builders.get(namespace_name)
|