diff --git a/AGENTS.md b/AGENTS.md index 90356fc4..6aea7be9 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -198,7 +198,7 @@ vale . - Parsers live in `docsgpt/parser/` and handle different document formats in the ingestion stage. - Agents and tools are in `docsgpt/agents/` and `docsgpt/agents/tools/`. - Celery setup/config lives in `docsgpt/celery_init.py` and `docsgpt/celeryconfig.py`. -- Settings and env vars are managed via Pydantic in `docsgpt/core/settings.py`. +- Settings and env vars are managed via Pydantic in `docsgpt/core/settings/` (one module per domain, composed into `Settings`). Every field needs a `description`; regenerate the docs reference with `python -m docsgpt.core.settings.reference --write`. ### Frontend diff --git a/docs/content/Guides/Architecture.mdx b/docs/content/Guides/Architecture.mdx index a65f3dcb..a25d34e6 100644 --- a/docs/content/Guides/Architecture.mdx +++ b/docs/content/Guides/Architecture.mdx @@ -251,4 +251,4 @@ The main extension points in [arc53/DocsGPT](https://github.com/arc53/DocsGPT) a | Parsing and workers | [`docsgpt/parser/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/parser), [`docsgpt/worker.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/worker.py), [`docsgpt/api/user/tasks.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/api/user/tasks.py) | | Models and vector stores | [`docsgpt/llm/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/llm), [`docsgpt/vectorstore/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/vectorstore) | | Storage and events | [`docsgpt/storage/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/storage), [`docsgpt/streaming/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/streaming), [`docsgpt/events/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/events) | -| Configuration, UI, and deployment | [`docsgpt/core/settings.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/core/settings.py), [`frontend/`](https://github.com/arc53/DocsGPT/tree/main/frontend), [`deployment/`](https://github.com/arc53/DocsGPT/tree/main/deployment) | +| Configuration, UI, and deployment | [`docsgpt/core/settings/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/core/settings), [`frontend/`](https://github.com/arc53/DocsGPT/tree/main/frontend), [`deployment/`](https://github.com/arc53/DocsGPT/tree/main/deployment) | diff --git a/docs/content/Guides/compression.md b/docs/content/Guides/compression.md index 14b90c62..53bcea57 100644 --- a/docs/content/Guides/compression.md +++ b/docs/content/Guides/compression.md @@ -19,7 +19,7 @@ The compression system operates on a "summarize and truncate" principle: ## Configuration -You can configure the compression behavior in your `.env` file or `docsgpt/core/settings.py`: +You can configure the compression behavior in your `.env` file or `docsgpt/core/settings/agents.py`: | Setting | Default | Description | | :--- | :--- | :--- | diff --git a/docs/content/quickstart.mdx b/docs/content/quickstart.mdx index 3b134fe6..a1702b3e 100644 --- a/docs/content/quickstart.mdx +++ b/docs/content/quickstart.mdx @@ -124,6 +124,6 @@ To work from the source tree, for example to build the images yourself, use `set ## Advanced Configuration -For more advanced customization of DocsGPT settings, such as configuring vector stores, embedding models, and other parameters, please refer to the [DocsGPT Settings documentation](/Deploying/DocsGPT-Settings). This guide explains how to modify the `.env` file or `settings.py` for deeper configuration. +For more advanced customization of DocsGPT settings, such as configuring vector stores, embedding models, and other parameters, please refer to the [DocsGPT Settings documentation](/Deploying/DocsGPT-Settings). This guide explains how to configure DocsGPT through the `.env` file, and links to the full settings reference. Enjoy using DocsGPT! diff --git a/docs/runbooks/sse-notifications.md b/docs/runbooks/sse-notifications.md index 1d100c43..6afaf225 100644 --- a/docs/runbooks/sse-notifications.md +++ b/docs/runbooks/sse-notifications.md @@ -320,7 +320,7 @@ redis-cli -n 2 DEL user::stream ## Settings reference -Everything in `docsgpt/core/settings.py`: +Everything in `docsgpt/core/settings/events.py`: | Setting | Default | Purpose | | --------------------------------------------- | ------- | --------------------------------------------- | diff --git a/docsgpt/core/db_uri.py b/docsgpt/core/db_uri.py index 99e93bc3..cb875b4e 100644 --- a/docsgpt/core/db_uri.py +++ b/docsgpt/core/db_uri.py @@ -15,7 +15,7 @@ have to know which driver a given field feeds. Each normalizer also silently upgrades the legacy ``postgresql+psycopg2://`` prefix since psycopg2 is no longer in the project. -This module is deliberately separate from ``docsgpt/core/settings.py`` +This module is deliberately separate from ``docsgpt/core/settings`` so the Settings class stays focused on field declarations, and the URI-rewriting logic can be unit-tested without triggering ``.env`` file loading from importing Settings. diff --git a/docsgpt/core/settings.py b/docsgpt/core/settings.py deleted file mode 100644 index 8bf864d0..00000000 --- a/docsgpt/core/settings.py +++ /dev/null @@ -1,596 +0,0 @@ -import os -from typing import Optional - -from pydantic import AliasChoices, Field, field_validator -from pydantic_settings import BaseSettings, SettingsConfigDict - -from docsgpt.core.db_uri import ( - normalize_pgvector_connection_string, - normalize_postgres_uri, -) -from docsgpt.core.paths import env_file, home_dir - -# Runtime data home (DOCSGPT_HOME, the checkout, or cwd); see docsgpt.core.paths. -current_dir = str(home_dir()) - - -class Settings(BaseSettings): - model_config = SettingsConfigDict(extra="ignore") - - AUTH_TYPE: Optional[str] = None # simple_jwt, session_jwt, oidc, or None - - # OIDC SSO (AUTH_TYPE=oidc) — any OpenID Connect IdP with discovery (Authentik, Keycloak, ...) - OIDC_ISSUER: Optional[str] = None # e.g. https://auth.example.com/application/o/docsgpt/ - OIDC_CLIENT_ID: Optional[str] = None - OIDC_CLIENT_SECRET: Optional[str] = None # optional; PKCE is always used - OIDC_SCOPES: str = "openid profile email" - OIDC_USER_ID_CLAIM: str = "sub" # ID-token claim mapped to the DocsGPT user id - OIDC_FRONTEND_URL: Optional[str] = None # browser-facing app origin, e.g. http://localhost:5173 - OIDC_REDIRECT_URI: Optional[str] = None # override; default /api/auth/oidc/callback - OIDC_SESSION_LIFETIME_SECONDS: int = 28800 # minted session JWT lifetime (8h) - OIDC_PROVIDER_NAME: Optional[str] = None # sign-in button label, e.g. "Acme SSO" - OIDC_ALLOWED_GROUPS: Optional[str] = None # comma-separated allowlist; unset = any authenticated user - OIDC_GROUPS_CLAIM: str = "groups" # ID-token/userinfo claim carrying group membership - OIDC_ADMIN_GROUPS: Optional[str] = None # comma-separated groups granted admin; unset = no OIDC admin mapping - - # RBAC: persisted admin grants live in user_roles (AUTH_TYPE=oidc only). This is the - # only non-DB admin path, for AUTH_TYPE=None self-host. MUST stay False if networked. - LOCAL_MODE_ADMIN: bool = False - - # SCIM 2.0 provisioning (IdP-driven user create/deactivate at /scim/v2) - SCIM_ENABLED: bool = False - SCIM_TOKEN: Optional[str] = None # bearer token for IdP SCIM clients (required when enabled) - - LLM_PROVIDER: str = "docsgpt" - LLM_NAME: Optional[str] = None # if LLM_PROVIDER is openai, LLM_NAME can be gpt-4 or gpt-3.5-turbo - # Legacy model on purpose: an install that never pinned this has vectors from it, and - # granite is the same width so a swap would fail silently. New installs get granite from - # .env-template; existing ones switch by setting this and running docsgpt.scripts.reembed. - EMBEDDINGS_NAME: str = "huggingface_sentence-transformers/all-mpnet-base-v2" - EMBEDDINGS_BASE_URL: Optional[str] = None # Remote embeddings API URL (OpenAI-compatible) - EMBEDDINGS_KEY: Optional[str] = None # api key for embeddings (if using openai, just copy API_KEY) - EMBEDDINGS_MAX_INPUT_TOKENS: Optional[int] = None # truncate each remote embed input to N tokens (overflow lost) - EMBEDDINGS_BATCH_SIZE: int = 32 # chunks per store transaction / remote embed request - # Documents per local ONNX forward pass. Each pass pads to its longest input, and that - # waste grows with the square of chunk length: at 1250 tokens, 32 peaked at 6.6 GB, 1 at 2.9 GB. - EMBEDDINGS_MODEL_BATCH_SIZE: int = 1 - # Intra-op threads for the local ONNX runner; None = every core. It scales sub-linearly, - # so several single-threaded workers beat one many-threaded process on the same cores. - EMBEDDINGS_THREADS: Optional[int] = None - # Embedding models and their tokenizers. Persistent by default: FastEmbed's own default is the temp dir. - EMBEDDINGS_CACHE_DIR: Optional[str] = Field(default_factory=lambda: str(home_dir() / "models")) - # Pooling ("cls"/"mean") and L2 normalisation. Read from the model's own repository; - # set these only for a repository that declares neither, or to override what it declares. - EMBEDDINGS_POOLING: Optional[str] = None - EMBEDDINGS_NORMALIZE: Optional[bool] = None - # Embed on the worker so the API holds no model (~890 MB), at one broker round trip per - # query. Ignored when EMBEDDINGS_BASE_URL is set, which is the better answer for production. - EMBEDDINGS_DELEGATE_TO_WORKER: bool = True - EMBEDDINGS_QUEUE: str = "embeddings" # queue the embed task is routed to - EMBEDDINGS_DELEGATE_TIMEOUT: int = 60 # seconds to wait for the worker - GITHUB_INGEST_MAX_FILE_BYTES: int = 1048576 # skip repo blobs larger than this (0 = no cap) - GITHUB_INGEST_MAX_WORKERS: int = 8 # parallel file fetches per GitHub repo ingest - # Operator-supplied model YAMLs, loaded after the built-in catalog; later wins on - # duplicate model id. See docsgpt/core/models/README.md. - MODELS_CONFIG_DIR: Optional[str] = None - - CELERY_BROKER_URL: str = "redis://localhost:6379/0" - CELERY_RESULT_BACKEND: str = "redis://localhost:6379/1" - # Prefetch=1 caps SIGKILL loss to one task. Visibility timeout must exceed the longest - # legitimate task runtime but stay short enough that SIGKILLed tasks redeliver promptly. - CELERY_WORKER_PREFETCH_MULTIPLIER: int = 1 - CELERY_VISIBILITY_TIMEOUT: int = 3600 - # Recycle a prefork child past this resident size in KB; backstops docling/torch heap growth. - # Checked between tasks, so it does not bound the peak within one. 0 disables. - CELERY_WORKER_MAX_MEMORY_PER_CHILD: int = 4194304 - CELERY_WORKER_MAX_TASKS_PER_CHILD: int = 0 # recycle after N tasks; 0 disables - # Only consulted when VECTOR_STORE=mongodb or when running scripts/db/backfill.py; user data lives in Postgres. - MONGO_URI: Optional[str] = None - # User-data Postgres DB. - POSTGRES_URI: Optional[str] = None - # On startup, apply pending Alembic migrations. Disable if you manage schema out-of-band. - AUTO_MIGRATE: bool = True - # On startup, create the target Postgres database if missing (needs CREATEDB privilege). - AUTO_CREATE_DB: bool = True - # On startup, create the pgvector/graph tables and verify the embedding dimension. No Alembic - # migration covers the vector DB (it may be a separate cluster); set False to manage it yourself. - AUTO_VECTOR_SCHEMA: bool = True - LLM_PATH: str = os.path.join(current_dir, "models/docsgpt-7b-f16.gguf") - DEFAULT_MAX_HISTORY: int = 150 - DEFAULT_LLM_TOKEN_LIMIT: int = 128000 # Fallback when model not found in registry - RESERVED_TOKENS: dict = { - "system_prompt": 500, - "current_query": 500, - "safety_buffer": 1000, - } - DEFAULT_AGENT_LIMITS: dict = { - "token_limit": 50000, - "request_limit": 500, - } - UPLOAD_FOLDER: str = "inputs" - # Serve the web UI shipped in the package (docsgpt/static) from the API process. - SERVE_UI: bool = True - # Request cap is applied by Flask before multipart parsing; the per-file cap also while copying. - UPLOAD_MAX_REQUEST_BYTES: int = Field(default=256 * 1024 * 1024, gt=0) - UPLOAD_MAX_FILE_BYTES: int = Field(default=100 * 1024 * 1024, gt=0) - PARSE_SPEC_MAX_BYTES: int = Field(default=10 * 1024 * 1024, gt=0) - # ZIP limits apply cumulatively across nested archives in one extraction. - UPLOAD_MAX_ARCHIVE_BYTES: int = Field(default=250 * 1024 * 1024, gt=0) - UPLOAD_MAX_ARCHIVE_FILES: int = Field(default=10_000, gt=0) - UPLOAD_MAX_ARCHIVE_RATIO: int = Field(default=1000, gt=0) - UPLOAD_MAX_ARCHIVE_DEPTH: int = Field(default=3, ge=0) - PARSE_PDF_AS_IMAGE: bool = False - PARSE_IMAGE_REMOTE: bool = False - # Document parser for source ingestion, chat attachments and the - # read_document tool. "anydoc" (default): firecrawl-anydoc, a Rust - # converter with no ML models — milliseconds per file, ~100 MB peak RSS. - # "docling": the layout/table-model pipeline (optional install; needed - # for read_document's structured output and the docling OCR backend). - # Files anydoc cannot convert (scanned PDFs, malformed input) fall back to - # docling when it is installed, otherwise to the native OCR parsers (OCR - # on) or the legacy parsers. Rollback to the previous behaviour is this - # one variable. - DOC_PARSER_ENGINE: str = "anydoc" - # OCR for scanned PDFs and images. OCR_ENABLED covers source ingestion, - # OCR_ATTACHMENTS_ENABLED chat attachments. Which stack performs it is - # OCR_BACKEND; which engine, OCR_ENGINE. The DOCLING_OCR_* names are the - # pre-2026-09 spellings and stay accepted as aliases. - OCR_ENABLED: bool = Field( - default=False, validation_alias=AliasChoices("OCR_ENABLED", "DOCLING_OCR_ENABLED") - ) - OCR_ATTACHMENTS_ENABLED: bool = Field( - default=False, - validation_alias=AliasChoices("OCR_ATTACHMENTS_ENABLED", "DOCLING_OCR_ATTACHMENTS_ENABLED"), - ) - # Which stack runs OCR when it is on: - # auto — docling when installed, otherwise native. - # docling — the layout-model pipeline (hybrid region OCR, reading order, - # table structure); needs the optional docling extra. - # native — pypdfium2/Pillow page rendering straight into tesseract or a - # DeepSeek-OCR endpoint (docsgpt/parser/file/ocr_parser.py). - # No ML models in the worker; tables come out as text lines - # under tesseract. - OCR_BACKEND: str = "auto" - # Pages docling's threaded pipeline buffers in flight; the library - # default (100) drives worker RSS to ~3 GB on a mid-size PDF. - DOCLING_PIPELINE_QUEUE_MAX_SIZE: int = 2 - DOCLING_COMPILE_TORCH_MODELS: bool = False - DOCLING_TABULAR_MAX_BYTES: int = 2_000_000 - DOCLING_MARKUP_MAX_BYTES: int = 8_000_000 - # HTML/XHTML larger than this (bytes) are head-truncated before the - # markdownify parser runs (the anydoc engine's HTML path). The tree that - # path builds costs ~50x the input — 30 MB of HTML measured at 1.6 GB RSS — - # and the upload cap is 100 MB, so the gate is what keeps one upload from - # taking the ingest worker down. 0 disables it. - MARKUP_MAX_BYTES: int = 8_000_000 - # Trust-check anydoc's PDF output (docsgpt/parser/file/pdf_trust.py): - # flag composite (Type0) fonts without a ToUnicode map, and CJK-declaring - # PDFs whose extracted text has almost no CJK — the two classes where - # anydoc drops text silently. A flagged file re-parses on the docling - # fallback when docling is installed; otherwise the anydoc output is kept - # and the document gets extra_info["parse_warnings"]. ~30 ms per scanned MB. - PDF_TRUST_CHECK: bool = True - # Rewrite dot-leader / whitespace-aligned table runs in anydoc's PDF - # markdown into GFM tables (docsgpt/parser/file/tableize.py). Off by - # default: it rewrites content on a heuristic (>=3 uniform label+numbers - # lines) validated only on a small corpus so far. - ANYDOC_TABLEIZE: bool = False - # OCR engine used when OCR is on (OCR_ENABLED / OCR_ATTACHMENTS_ENABLED). - # Benched 2026-08 on EN/ZH/table/degraded scans (docs/Guides/ocr has the - # menu): - # tesseract — recommended: best classic-engine accuracy (perfect EN word - # recall, 0.000 bilingual CER, 100% table cells), ~35 MB, CPU-only. - # Needs the system binary + language packs: an optional install like - # every OCR dependency (build with INSTALL_TESSERACT=true, or apt/brew - # install tesseract-ocr for a local run). Both backends. - # deepseek — DeepSeek-OCR against an Ollama/vLLM endpoint - # (OCR_DEEPSEEK_*). Best table/CJK quality; the worker stays light - # (no layout models) but each page costs seconds on the model server. - # Both backends. - # auto — docling's pick: ocrmac on macOS (excellent), rapidocr on Linux - # (silently shreds some long text lines — avoid as a server default). - # ocrmac | rapidocr — force one of those. - # auto/ocrmac/rapidocr exist only inside docling; the native backend runs - # tesseract for them. An engine that is not installed degrades (docling: - # to "auto") with a warning instead of failing the parse. - OCR_ENGINE: str = "tesseract" - # Tesseract language packs, "+"-separated (e.g. "eng+chi_sim+deu"). Other - # engines keep their own defaults — their language codes differ. - OCR_LANGS: str = "eng" - OCR_DEEPSEEK_URL: str = "http://localhost:11434/v1/chat/completions" - OCR_DEEPSEEK_MODEL: str = "deepseek-ocr:3b" - # Seconds allowed per page request to the DeepSeek endpoint, on both - # backends (native sends pages one at a time; docling's VLM pipeline - # keeps its own concurrency). A 3B model on a laptop needs minutes; a - # vLLM GPU deployment, seconds. - OCR_DEEPSEEK_TIMEOUT: float = 300.0 - # Native backend only: resolution at which pages without a text layer are - # rendered before OCR. 200 suits tesseract; clamped to 72-600. - OCR_RENDER_DPI: int = 200 - # Chars-per-page floor below which an OCR'd PDF/image parse is treated as an OCR - # dropout (long-running docling workers were observed returning zero characters for - # every scanned page after a long scanned PDF, with no error) rather than as content. - # docling retries once on a fresh full-page-OCR converter; both backends then fail - # loudly instead of indexing an empty document. 0 disables the guard. - OCR_MIN_CHARS_PER_PAGE: int = Field( - default=20, validation_alias=AliasChoices("OCR_MIN_CHARS_PER_PAGE", "DOCLING_OCR_MIN_CHARS_PER_PAGE") - ) - # Read PDF *attachments* via their embedded text layer (pypdfium2) instead - # of docling, falling back to docling when there is no text layer to read. - # Attachments go into a prompt, so docling's structural markdown earns far - # less than the tens of seconds per file it costs; source ingestion is - # unaffected and always uses docling, because chunking and retrieval do - # depend on that structure. - ATTACHMENT_PDF_TEXT_FAST_PATH: bool = True - # Median chars per sampled page below which a PDF is treated as a scan and handed to docling. - # Measured on real uploads: scans at 0-17 chars/page, text-layer documents at 433-6834. - ATTACHMENT_PDF_TEXT_MIN_MEDIAN_CHARS: int = 32 - ATTACHMENT_TEXT_MAX_BYTES: int = 5_000_000 - AGENT_IMAGE_MAX_BYTES: int = 5_000_000 - AGENT_IMAGE_MAX_PIXELS: int = 16_777_216 - VECTOR_STORE: str = "faiss" # "faiss" or "elasticsearch" or "qdrant" or "milvus" or "lancedb" or "pgvector" - # Retriever keys an agent may use; must match RetrieverCreator.retrievers registry keys, - # NOT the legacy ``classic_rag`` label which never matched the registry. - RETRIEVERS_ENABLED: list = ["classic", "default"] - # Concurrent per-source searches in one retrieval; the query is embedded once and shared. - RETRIEVAL_MAX_PARALLEL_SOURCES: int = 4 - # Kill-switch for per-source retrieval dispatch; False collapses to a single retriever. - PER_SOURCE_RETRIEVAL_ENABLED: bool = True - GRAPHRAG_ENABLED: bool = False # gates graph-aware ingestion/retrieval - # Model for ingest-time graph extraction; None reuses LLM_PROVIDER/LLM_NAME. - GRAPHRAG_EXTRACTION_MODEL: Optional[str] = None - # Hard cap on chunks extracted per source (cost control). - GRAPHRAG_MAX_CHUNKS_FOR_EXTRACTION: int = 2000 - AGENT_NAME: str = "classic" - FALLBACK_LLM_PROVIDER: Optional[str] = None # provider for fallback llm - FALLBACK_LLM_NAME: Optional[str] = None # model name for fallback llm - FALLBACK_LLM_API_KEY: Optional[str] = None # api key for fallback llm - - # Google Drive integration - GOOGLE_CLIENT_ID: Optional[str] = None # Replace with your actual Google OAuth client ID - GOOGLE_CLIENT_SECRET: Optional[str] = None # Replace with your actual Google OAuth client secret - CONNECTOR_REDIRECT_BASE_URI: Optional[str] = ( - "http://127.0.0.1:7091/api/connectors/callback" ##add redirect url as it is to your provider's console(gcp) - ) - # Comma-separated frontend origins allowed to receive connector OAuth results, e.g. https://docsgpt.example.com. - # The callback origin and OIDC_FRONTEND_URL are always allowed; a loopback callback also allows localhost:5173. - CONNECTOR_ALLOWED_ORIGINS: Optional[str] = None - - # Microsoft Entra ID (Azure AD) integration - MICROSOFT_CLIENT_ID: Optional[str] = None # Azure AD Application (client) ID - MICROSOFT_CLIENT_SECRET: Optional[str] = None # Azure AD Application client secret - MICROSOFT_TENANT_ID: Optional[str] = "common" # Azure AD Tenant ID (or 'common' for multi-tenant) - MICROSOFT_AUTHORITY: Optional[str] = None # e.g., "https://login.microsoftonline.com/{tenant_id}" - - # Confluence Cloud integration - CONFLUENCE_CLIENT_ID: Optional[str] = None - CONFLUENCE_CLIENT_SECRET: Optional[str] = None - - # GitHub source - GITHUB_ACCESS_TOKEN: Optional[str] = None # PAT token with read repo access - - # LLM Cache - CACHE_REDIS_URL: str = "redis://localhost:6379/2" - - API_URL: str = "http://localhost:7091" # backend url for celery worker - - # Public base URL for user-facing endpoint references in prompts - PUBLIC_API_BASE_URL: Optional[str] = None - MCP_OAUTH_REDIRECT_URI: Optional[str] = None # public callback URL for MCP OAuth - INTERNAL_KEY: Optional[str] = None # internal api key for worker-to-backend auth - - API_KEY: Optional[str] = None # LLM api key (used by LLM_PROVIDER) - - # Provider-specific API keys (for multi-model support) - OPENAI_API_KEY: Optional[str] = None - ANTHROPIC_API_KEY: Optional[str] = None - GOOGLE_API_KEY: Optional[str] = None - GROQ_API_KEY: Optional[str] = None - HUGGINGFACE_API_KEY: Optional[str] = None - OPEN_ROUTER_API_KEY: Optional[str] = None - NOVITA_API_KEY: Optional[str] = None - - OPENAI_API_BASE: Optional[str] = None # azure openai api base url - OPENAI_API_VERSION: Optional[str] = None # azure openai api version - AZURE_DEPLOYMENT_NAME: Optional[str] = None # azure deployment name for answering - AZURE_EMBEDDINGS_DEPLOYMENT_NAME: Optional[str] = None # azure deployment name for embeddings - OPENAI_BASE_URL: Optional[str] = None # openai base url for open ai compatable models - - # elasticsearch - ELASTIC_CLOUD_ID: Optional[str] = None # cloud id for elasticsearch - ELASTIC_USERNAME: Optional[str] = None # username for elasticsearch - ELASTIC_PASSWORD: Optional[str] = None # password for elasticsearch - ELASTIC_URL: Optional[str] = None # url for elasticsearch - ELASTIC_INDEX: Optional[str] = "docsgpt" # index name for elasticsearch - - # Legacy AWS credentials from the retired SageMaker provider. Still read as a deprecated - # fallback by S3 storage; do not use for new deployments. - SAGEMAKER_REGION: Optional[str] = None - SAGEMAKER_ACCESS_KEY: Optional[str] = None - SAGEMAKER_SECRET_KEY: Optional[str] = None - - # Qdrant vectorstore config - QDRANT_COLLECTION_NAME: Optional[str] = "docsgpt" - QDRANT_LOCATION: Optional[str] = None - QDRANT_URL: Optional[str] = None - QDRANT_PORT: Optional[int] = 6333 - QDRANT_GRPC_PORT: int = 6334 - QDRANT_PREFER_GRPC: bool = False - QDRANT_HTTPS: Optional[bool] = None - QDRANT_API_KEY: Optional[str] = None - QDRANT_PREFIX: Optional[str] = None - QDRANT_TIMEOUT: Optional[float] = None - QDRANT_HOST: Optional[str] = None - QDRANT_PATH: Optional[str] = None - QDRANT_DISTANCE_FUNC: str = "Cosine" - - # PGVector config. postgres://, postgresql:// and postgresql+psycopg:// are all accepted - # and normalized internally for psycopg.connect(). - PGVECTOR_CONNECTION_STRING: Optional[str] = None - PGVECTOR_POOL_MAX_SIZE: int = 8 # per-process pool; 0 = one direct connection per store - # IVFFlat probes; None derives sqrt(lists) from the index. Higher = better recall, more scan. - PGVECTOR_IVFFLAT_PROBES: Optional[int] = None - # Milvus vectorstore config - MILVUS_COLLECTION_NAME: Optional[str] = "docsgpt" - # milvus-lite (embedded) database file, under the data home like the other local stores - MILVUS_URI: Optional[str] = Field(default_factory=lambda: str(home_dir() / "milvus_local.db")) - MILVUS_TOKEN: Optional[str] = "" - - # LanceDB vectorstore config - LANCEDB_PATH: str = Field(default_factory=lambda: str(home_dir() / "data" / "lancedb")) # LanceDB local data - LANCEDB_TABLE_NAME: Optional[str] = "docsgpts" # Name of the table to use for storing vectors - - FLASK_DEBUG_MODE: bool = False - STORAGE_TYPE: str = "local" # local or s3 - - # S3-compatible object storage (STORAGE_TYPE=s3): AWS S3, MinIO, R2, B2, Spaces, ... - # For non-AWS, set S3_ENDPOINT_URL and usually S3_PATH_STYLE=true. - S3_BUCKET_NAME: str = "docsgpt-test-bucket" - S3_ENDPOINT_URL: Optional[str] = None # custom endpoint for S3-compatible services; omit for AWS - S3_ACCESS_KEY_ID: Optional[str] = None - S3_SECRET_ACCESS_KEY: Optional[str] = None - S3_REGION: Optional[str] = None # AWS region; use "auto" for Cloudflare R2 - S3_PATH_STYLE: bool = False # path-style addressing (required by most non-AWS services) - - # Anonymous startup version check for security issues. - VERSION_CHECK: bool = True - URL_STRATEGY: str = "backend" # backend or s3 - - JWT_SECRET_KEY: str = "" - - # Encryption settings - ENCRYPTION_SECRET_KEY: str = "default-docsgpt-encryption-key" - - TTS_PROVIDER: str = "google_tts" # google_tts, elevenlabs, or none to switch text-to-speech off - ELEVENLABS_API_KEY: Optional[str] = None - STT_PROVIDER: str = "openai" # openai, faster_whisper, or none to switch speech-to-text off - OPENAI_STT_MODEL: str = "gpt-4o-mini-transcribe" - STT_LANGUAGE: Optional[str] = None - STT_MAX_FILE_SIZE_MB: int = 50 - STT_ENABLE_TIMESTAMPS: bool = False - STT_ENABLE_DIARIZATION: bool = False - - # Tool pre-fetch settings - ENABLE_TOOL_PREFETCH: bool = True - - # True persists Responses API calls server-side so previous_response_id can chain turns. - # False keeps them stateless, carrying reasoning across the tool loop as encrypted items. - OPENAI_RESPONSES_STORE: bool = False - # Cross-turn ``previous_response_id`` chaining (store mode only). The - # chained transcript lives on the provider and is invisible to every - # local guard, so it is bounded: a turn starts from the local history - # when the previous turn's reported prompt already reached the budget - # (default: the model's context window) or when the conversation was - # compressed after that turn was produced. - OPENAI_RESPONSES_CHAIN_ACROSS_TURNS: bool = True - OPENAI_RESPONSES_CHAIN_BUDGET_TOKENS: Optional[int] = None - # ``truncation: "auto"`` lets the provider drop the oldest input items - # instead of failing every request once a chain exceeds the model's window. - OPENAI_RESPONSES_TRUNCATION_AUTO: bool = False - # Prompt-cache hints on the Responses API: route a user's calls to the - # same cache shard (opaque per-user key), and request extended retention - # where offered. - OPENAI_PROMPT_CACHE_KEY: bool = True - OPENAI_PROMPT_CACHE_RETENTION: Optional[str] = None - OPENAI_REASONING_SUMMARY: str = "auto" - - # Lets OpenAI-compatible clients identify a logical chat by session header, which - # chat-completions itself has no field for. - V1_SESSION_TTL_SECONDS: int = 24 * 60 * 60 - # Optional cheaper model for conversation titles; unset reuses the answer model. - TITLE_MODEL_ID: Optional[str] = None - - # Config-free tools on by default in agentless chats. ``scheduler`` is dual-registered in - # BUILTIN_AGENT_TOOLS so one synthetic id resolves via defaults or the agent picker. - # Add "code_executor" and "artifact_generator" once a sandbox runner is configured — both - # execute through it and would fail on every call without one. - DEFAULT_CHAT_TOOLS: list = [ - "memory", - "read_webpage", - "scheduler", - ] - - # Conversation Compression Settings - ENABLE_CONVERSATION_COMPRESSION: bool = True - COMPRESSION_THRESHOLD_PERCENTAGE: float = 0.8 # Trigger at 80% of context - COMPRESSION_MODEL_OVERRIDE: Optional[str] = None # Use different model for compression - COMPRESSION_PROMPT_VERSION: str = "v1.0" # Track prompt iterations - COMPRESSION_MAX_HISTORY_POINTS: int = 3 # Keep only last N compression points to prevent DB bloat - # Per-field cap on the verbatim tail kept after a compression point (0 disables). - COMPRESSION_RECENT_FIELD_MAX_TOKENS: int = 8000 - # Cap on one tool result entering the LLM context (0 disables); journal/DB keep it whole. - TOOL_RESULT_MAX_TOKENS: int = 20000 - - # Agent Guardrails - GUARDRAILS_ENABLED: bool = True # master switch; False disables every stage - # Allowlist of GuardrailCreator.checks keys; empty means every registered check. - GUARDRAILS_CHECKS_ENABLED: list = [] - # A GuardrailsConfig fragment every agent inherits and cannot weaken; agents may add - # controls or make an action stricter, never looser. "enabled" is required — without it - # the floor parses but applies to nothing. Example: - # {"enabled": true, "mode": "scan_all", - # "controls": [{"check": "secrets", "stage": "output", "action": "redact"}]} - GUARDRAILS_FLOOR: dict = {} - # Judge model for the topic/policy checks; None reuses the request's model. - GUARDRAILS_JUDGE_MODEL: Optional[str] = None - # Persist scanned text alongside guardrail_events. Off by default: pre-redaction text is - # exactly the material a PII control exists to keep out of storage. - GUARDRAILS_STORE_SCANNED_TEXT: bool = False - GUARDRAILS_EVENTS_RETENTION_DAYS: int = Field(default=30, ge=1) - - # Internal SSE push channel (notifications + durable replay journal). - # False makes /api/events emit "push_disabled" and return; clients fall back to polling. - ENABLE_SSE_PUSH: bool = True - # Per-user durable backlog cap in entries; ~24h of replay at typical rates. - EVENTS_STREAM_MAXLEN: int = 1000 - # Bounds uvicorn's shutdown drain (uvicorn_worker doesn't forward --graceful-timeout). - # Keep below the gunicorn --timeout (180) watchdog. Used by BoundedDrainUvicornWorker. - GRACEFUL_SHUTDOWN_TIMEOUT_SECONDS: int = 30 - WSGI_THREADPOOL_WORKERS: int = 96 - SSE_KEEPALIVE_SECONDS: int = Field(default=15, ge=1) - # Simultaneous SSE connections per user; each holds a pooled async Redis connection for - # its lifetime. 8 covers multi-tab use without one user starving the pool. 0 disables. - SSE_MAX_CONCURRENT_PER_USER: int = 8 - # Pool size of the async Redis client behind the event-loop routes, per process. Every - # open notification tab, chat reconnect and device session holds one connection, so this - # caps concurrent streams per worker (redis-py's own default is 100). Keep the total - # across workers below the Redis server's maxclients (10000 by default). - ASYNC_REDIS_MAX_CONNECTIONS: int = Field(default=2000, ge=1) - # Backlog entries XRANGE returns per /api/events snapshot. Bounds what one replay moves - # from Redis to the wire: a client looping Last-Event-ID reconnects enumerates at most - # this many per round-trip, and the budget below bounds total throughput. - EVENTS_REPLAY_MAX_PER_REQUEST: int = 200 - EVENTS_REPLAY_MAX_AGE_HOURS: int = 48 - # Sliding-window cap on snapshot replays per user; exhausting it returns 429 with the - # cursor pinned so the client backs off until the window rolls over. - EVENTS_REPLAY_BUDGET_REQUESTS_PER_WINDOW: int = 30 - EVENTS_REPLAY_BUDGET_WINDOW_SECONDS: int = 60 - - # Retention for the message_events journal, enforced by the cleanup_message_events beat - # task. Replay only needs streams a client could still be tailing. - MESSAGE_EVENTS_RETENTION_DAYS: int = 14 - - # Remote Device feature. - REMOTE_DEVICE_SESSION_IDLE_SECONDS: int = 60 - REMOTE_DEVICE_REQUIRE_SIGNATURE: bool = False - REMOTE_DEVICE_PAIRING_TTL_SECONDS: int = 600 - # Redis broker tunables, routing invocations cross-process so a scheduled run reaches the - # web-held device session. The queue TTL must exceed the max drain deadline (605s) so a - # command for a briefly-offline device isn't evicted before its own drain gives up. - REMOTE_DEVICE_CMD_QUEUE_TTL_SECONDS: int = 900 - REMOTE_DEVICE_INVOCATION_TTL_SECONDS: int = 900 - REMOTE_DEVICE_OUTPUT_STREAM_MAXLEN: int = 10_000 - - # Scheduler (see scheduler.md). - SCHEDULE_DISPATCHER_INTERVAL: int = 30 - SCHEDULE_MIN_INTERVAL: int = 900 - SCHEDULE_MAX_PER_USER: int = 50 - SCHEDULE_RUN_TIMEOUT: int = 600 - SCHEDULE_MISFIRE_GRACE: int = 60 - SCHEDULE_AUTOPAUSE_FAILURES: int = 3 - SCHEDULE_ONCE_MAX_HORIZON: int = 31_536_000 - SCHEDULE_RUN_OUTPUT_RETENTION_DAYS: int = 90 - - # Code-execution sandbox. The app is a CLIENT of an always-on runner; defaults are safe so - # app import never fails when the sandbox is unconfigured. - SANDBOX_BACKEND: str = "jupyter" # "jupyter" (self-host) | "daytona" (Daytona Cloud) - # URL of the Jupyter Kernel Gateway runner (the docsgpt-sandbox service). - SANDBOX_GATEWAY_URL: str = "http://localhost:8888" - SANDBOX_GATEWAY_AUTH_TOKEN: Optional[str] = None # gateway auth token, if set - # Kernelspec per session. The env-scrubbing "docsgpt-python" spec keeps kernel code from - # reading the gateway token or operator secrets from os.environ; the stock "python3" spec - # inherits the gateway env verbatim and must not be used with untrusted code. - SANDBOX_KERNEL_NAME: str = "docsgpt-python" - SANDBOX_MAX_TTL: int = 1200 # hard cap (s) on agent-selectable keep-alive TTL - # Concurrent live sessions per process, backend-agnostic; at the cap an LRU-idle session is - # evicted. 0 or negative disables the cap. - SANDBOX_MAX_SESSIONS: int = 32 - SANDBOX_EXEC_TIMEOUT: int = 60 # default wall-clock cap (s) per exec call - SANDBOX_HTTP_TIMEOUT: int = 10 # fixed cap (s) for REST control calls (create/delete/alive/interrupt) - SANDBOX_MAX_OUTPUT_BYTES: int = 8 * 1024 * 1024 # cap on buffered stdout+stderr per exec - SANDBOX_MAX_FILE_BYTES: int = 10 * 1024 * 1024 # cap on get_file size routed through stdout - SANDBOX_MAX_INPUT_BYTES: int = 25 * 1024 * 1024 # cap on an input document staged into a sandbox session - # ``read_document`` parsing on a dedicated Celery ``parsing`` queue (backend parser). - DOCUMENT_PARSE_QUEUE: str = "parsing" # queue the parse_document task is routed to - DOCUMENT_PARSE_TIMEOUT: int = 120 # seconds the tool awaits the enqueued parse before degrading - # The base timeout is a FLOOR: the window grows with document size, because OCR cost scales - # with pages. Without this a large scan is silently dropped at the base window. - DOCUMENT_PARSE_TIMEOUT_PER_MB: int = 60 # extra seconds of parse window per MiB of input - DOCUMENT_PARSE_TIMEOUT_MAX: int = 900 # absolute ceiling on the size-scaled parse window - DOCUMENT_PARSE_MAX_BYTES: int = 0 # cap on a parsed document's bytes (0 = reuse SANDBOX_MAX_INPUT_BYTES) - DOCUMENT_MAX_DECOMPRESSED_BYTES: int = 300 * 1024 * 1024 - DOCUMENT_MAX_ARCHIVE_ENTRIES: int = 10000 - # Files per node passed natively to the LLM; past the cap they are extracted to text or - # dropped, to bound context and cost. Re-uses SANDBOX_MAX_INPUT_BYTES per file. - WORKFLOW_NODE_NATIVE_MAX_FILES: int = 5 - # Documents per node extracted via the parsing worker. Each issues a separate blocking - # parse; past the cap they are skipped with a truncation note. - WORKFLOW_NODE_EXTRACT_MAX_FILES: int = 5 - # Wall clock one node may spend on blocking parses, shared across all of them. Without it a - # node could serialize WORKFLOW_NODE_EXTRACT_MAX_FILES full windows on a web threadpool slot. - WORKFLOW_NODE_EXTRACT_BUDGET_SECONDS: int = 900 - # A run row is pre-created as ``running``; a disconnect or crash can strand it there. The - # beat reaper fails runs still ``running`` past this. Generous so a long run is never cut off. - WORKFLOW_RUN_STALE_SECONDS: int = 3600 - # Runner container caps, consumed by the docsgpt-sandbox compose service, not the app. - # These cgroup limits are part of the untrusted-code security boundary. - SANDBOX_MEMORY: str = "1g" # docker mem_limit for the runner container - SANDBOX_CPUS: str = "1.0" # docker cpu quota for the runner container - # Daytona Cloud backend (SANDBOX_BACKEND="daytona"). All knobs are optional so app import - # never fails when the backend is unused. - DAYTONA_API_KEY: Optional[str] = None # Daytona Cloud API key (secret) - DAYTONA_API_URL: Optional[str] = None # override Daytona API base URL, if self-targeting - DAYTONA_TARGET: Optional[str] = None # Daytona region/target, e.g. "us" - DAYTONA_SNAPSHOT: Optional[str] = None # image for new sandboxes; render libs via scripts/build_daytona_snapshot.py - DAYTONA_LANGUAGE: str = "python" # default runtime language for created sandboxes - DAYTONA_AUTO_STOP_INTERVAL: int = 15 # minutes idle before Daytona auto-stops a sandbox (0 disables) - DAYTONA_AUTO_DELETE_INTERVAL: int = 60 # minutes after stop before Daytona auto-deletes (-1 disables) - DAYTONA_MAX_SANDBOXES: int = 50 # cap on concurrent live Daytona sandboxes (cost-DoS guard) - # Per-user artifact quotas, enforced at persistence time. 0 or negative disables a quota. - ARTIFACT_MAX_BYTES: int = 50 * 1024 * 1024 # cap on a single stored artifact version's bytes - ARTIFACT_MAX_COUNT_PER_USER: int = 5000 # cap on artifacts a user may own - ARTIFACT_MAX_TOTAL_BYTES_PER_USER: int = 5 * 1024 * 1024 * 1024 # cap on a user's total stored bytes - - @field_validator("POSTGRES_URI", mode="before") - @classmethod - def _normalize_postgres_uri_validator(cls, v): - return normalize_postgres_uri(v) - - @field_validator("PGVECTOR_CONNECTION_STRING", mode="before") - @classmethod - def _normalize_pgvector_connection_string_validator(cls, v): - return normalize_pgvector_connection_string(v) - - @field_validator( - "API_KEY", - "OPENAI_API_KEY", - "ANTHROPIC_API_KEY", - "GOOGLE_API_KEY", - "GROQ_API_KEY", - "HUGGINGFACE_API_KEY", - "NOVITA_API_KEY", - "EMBEDDINGS_KEY", - "FALLBACK_LLM_API_KEY", - "QDRANT_API_KEY", - "ELEVENLABS_API_KEY", - "INTERNAL_KEY", - mode="before", - ) - @classmethod - def normalize_api_key(cls, v: Optional[str]) -> Optional[str]: - """ - Normalize API keys: convert 'None', 'none', empty strings, - and whitespace-only strings to actual None. - Handles Pydantic loading 'None' from .env as string "None". - """ - if v is None: - return None - if not isinstance(v, str): - return v - stripped = v.strip() - if stripped == "" or stripped.lower() == "none": - return None - return stripped - - -settings = Settings(_env_file=env_file(), _env_file_encoding="utf-8") diff --git a/docsgpt/core/settings/__init__.py b/docsgpt/core/settings/__init__.py new file mode 100644 index 00000000..2af13cea --- /dev/null +++ b/docsgpt/core/settings/__init__.py @@ -0,0 +1,77 @@ +"""Application settings. + +``settings`` is the process-wide instance, loaded from the environment and the +``.env`` file in the data home (see ``docsgpt.core.paths``). Every setting is a +flat attribute, ``settings.NAME``, matching the environment variable of the +same name. + +The definitions are split by domain into the modules of this package; each +module owns one ``SettingsGroup`` and ``Settings`` composes them all. Add a new +setting to the group it belongs to (or add a group and list it in +``SETTINGS_GROUPS``), with a ``description`` -- the settings reference in the +docs is generated from these definitions. +""" + +from __future__ import annotations + +from typing import Optional + +from docsgpt.core.paths import env_file, home_dir +from docsgpt.core.settings._shared import SettingsGroup, normalize_secret +from docsgpt.core.settings.agents import AgentSettings +from docsgpt.core.settings.auth import AuthSettings +from docsgpt.core.settings.connectors import ConnectorSettings +from docsgpt.core.settings.database import DatabaseSettings +from docsgpt.core.settings.embeddings import EmbeddingsSettings +from docsgpt.core.settings.events import EventsSettings +from docsgpt.core.settings.guardrails import GuardrailSettings +from docsgpt.core.settings.ingestion import IngestionSettings +from docsgpt.core.settings.llm import LLMSettings +from docsgpt.core.settings.ocr import OCRSettings +from docsgpt.core.settings.retrieval import RetrievalSettings +from docsgpt.core.settings.sandbox import SandboxSettings +from docsgpt.core.settings.scheduler import SchedulerSettings +from docsgpt.core.settings.server import ServerSettings +from docsgpt.core.settings.speech import SpeechSettings +from docsgpt.core.settings.storage import StorageSettings +from docsgpt.core.settings.vectorstores import VectorStoreSettings +from docsgpt.core.settings.workers import WorkerSettings + +#: Every settings group, in the order the generated reference lists them. +SETTINGS_GROUPS: tuple[tuple[str, type[SettingsGroup]], ...] = ( + ("Authentication", AuthSettings), + ("LLM providers", LLMSettings), + ("Embeddings", EmbeddingsSettings), + ("Retrieval", RetrievalSettings), + ("Vector stores", VectorStoreSettings), + ("User-data database", DatabaseSettings), + ("Workers", WorkerSettings), + ("Ingestion and parsing", IngestionSettings), + ("OCR", OCRSettings), + ("File storage", StorageSettings), + ("Connectors", ConnectorSettings), + ("Server", ServerSettings), + ("Events and devices", EventsSettings), + ("Agents", AgentSettings), + ("Guardrails", GuardrailSettings), + ("Scheduler", SchedulerSettings), + ("Sandbox", SandboxSettings), + ("Speech", SpeechSettings), +) + +# Runtime data home (DOCSGPT_HOME, the checkout, or cwd); see docsgpt.core.paths. +current_dir = str(home_dir()) + + +class Settings(*(group for _, group in SETTINGS_GROUPS)): + """All settings, composed from the per-domain groups in this package.""" + + @classmethod + def normalize_api_key(cls, v: Optional[str]) -> Optional[str]: + """Normalize a secret the way the per-field validators do; kept for callers that reuse it.""" + return normalize_secret(v) + + +settings = Settings(_env_file=env_file(), _env_file_encoding="utf-8") + +__all__ = ["SETTINGS_GROUPS", "Settings", "SettingsGroup", "current_dir", "settings"] diff --git a/docsgpt/core/settings/_shared.py b/docsgpt/core/settings/_shared.py new file mode 100644 index 00000000..4eb5c48d --- /dev/null +++ b/docsgpt/core/settings/_shared.py @@ -0,0 +1,37 @@ +"""Building blocks shared by the settings groups. + +Every group in this package is a :class:`SettingsGroup`: a ``BaseSettings`` +subclass that owns one domain's fields. ``docsgpt.core.settings.Settings`` +inherits from all of them, so the composed class keeps the flat +``settings.NAME`` attributes the rest of the codebase reads while each +domain's definitions live in their own module. +""" + +from __future__ import annotations + +from typing import Optional + +from pydantic_settings import BaseSettings, SettingsConfigDict + + +class SettingsGroup(BaseSettings): + """Base for one domain's settings; groups are composed into ``Settings``.""" + + model_config = SettingsConfigDict(extra="ignore") + + +def normalize_secret(value: Optional[str]) -> Optional[str]: + """Map the ways an unset secret reaches us from ``.env`` to ``None``. + + ``.env`` files carry ``KEY=None`` and ``KEY=`` for "not set", and pydantic + would otherwise keep those as the strings ``"None"`` and ``""``. Whitespace + around a real value is stripped. + """ + if value is None: + return None + if not isinstance(value, str): + return value + stripped = value.strip() + if stripped == "" or stripped.lower() == "none": + return None + return stripped diff --git a/docsgpt/core/settings/agents.py b/docsgpt/core/settings/agents.py new file mode 100644 index 00000000..fdee1cc3 --- /dev/null +++ b/docsgpt/core/settings/agents.py @@ -0,0 +1,91 @@ +"""Agent runtime: default tools, context management, workflows and artifacts.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class AgentSettings(SettingsGroup): + """What an agent may do per turn and how its context is kept within budget.""" + + AGENT_NAME: str = Field(default="classic", description="Default agent type for agentless chats.") + DEFAULT_MAX_HISTORY: int = Field(default=150, description="Default number of history messages kept.") + DEFAULT_AGENT_LIMITS: dict = Field( + default={"token_limit": 50000, "request_limit": 500}, + description="Per-agent default quotas: tokens and requests.", + ) + DEFAULT_CHAT_TOOLS: list = Field( + default=["memory", "read_webpage", "scheduler"], + description=( + "Config-free tools on by default in agentless chats. scheduler is dual-registered in " + "BUILTIN_AGENT_TOOLS so one synthetic id resolves via defaults or the agent picker. Add " + "code_executor and artifact_generator once a sandbox runner is configured; both execute through " + "it and would fail on every call without one." + ), + ) + ENABLE_TOOL_PREFETCH: bool = Field(default=True, description="Pre-fetch retrieval before the agent's first turn.") + TOOL_RESULT_MAX_TOKENS: int = Field( + default=20000, + description="Cap on one tool result entering the LLM context (0 disables); journal and DB keep it whole.", + ) + + # Conversation compression. + ENABLE_CONVERSATION_COMPRESSION: bool = Field( + default=True, description="Compress long conversations once they approach the context window." + ) + COMPRESSION_THRESHOLD_PERCENTAGE: float = Field( + default=0.8, description="Fraction of the context window at which compression triggers." + ) + COMPRESSION_MODEL_OVERRIDE: Optional[str] = Field( + default=None, description="Use a different model for compression; unset reuses the answer model." + ) + COMPRESSION_PROMPT_VERSION: str = Field(default="v1.0", description="Tracks compression prompt iterations.") + COMPRESSION_MAX_HISTORY_POINTS: int = Field( + default=3, description="Keep only the last N compression points to prevent DB bloat." + ) + COMPRESSION_RECENT_FIELD_MAX_TOKENS: int = Field( + default=8000, description="Per-field cap on the verbatim tail kept after a compression point (0 disables)." + ) + + # Workflows. + WORKFLOW_NODE_NATIVE_MAX_FILES: int = Field( + default=5, + description=( + "Files per node passed natively to the LLM; past the cap they are extracted to text or dropped, to " + "bound context and cost. Re-uses SANDBOX_MAX_INPUT_BYTES per file." + ), + ) + WORKFLOW_NODE_EXTRACT_MAX_FILES: int = Field( + default=5, + description=( + "Documents per node extracted via the parsing worker. Each issues a separate blocking parse; past " + "the cap they are skipped with a truncation note." + ), + ) + WORKFLOW_NODE_EXTRACT_BUDGET_SECONDS: int = Field( + default=900, + description=( + "Wall clock one node may spend on blocking parses, shared across all of them. Without it a node " + "could serialize WORKFLOW_NODE_EXTRACT_MAX_FILES full windows on a web threadpool slot." + ), + ) + WORKFLOW_RUN_STALE_SECONDS: int = Field( + default=3600, + description=( + "A run row is pre-created as running; a disconnect or crash can strand it there. The beat reaper " + "fails runs still running past this. Generous so a long run is never cut off." + ), + ) + + # Per-user artifact quotas, enforced at persistence time. 0 or negative disables a quota. + ARTIFACT_MAX_BYTES: int = Field( + default=50 * 1024 * 1024, description="Cap on a single stored artifact version's bytes (0 disables)." + ) + ARTIFACT_MAX_COUNT_PER_USER: int = Field(default=5000, description="Cap on artifacts a user may own (0 disables).") + ARTIFACT_MAX_TOTAL_BYTES_PER_USER: int = Field( + default=5 * 1024 * 1024 * 1024, description="Cap on a user's total stored artifact bytes (0 disables)." + ) diff --git a/docsgpt/core/settings/auth.py b/docsgpt/core/settings/auth.py new file mode 100644 index 00000000..d0763934 --- /dev/null +++ b/docsgpt/core/settings/auth.py @@ -0,0 +1,87 @@ +"""Authentication, SSO and provisioning.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field, field_validator + +from docsgpt.core.settings._shared import SettingsGroup, normalize_secret + + +class AuthSettings(SettingsGroup): + """How users authenticate: none, a shared token, per-session JWTs, or OIDC SSO.""" + + AUTH_TYPE: Optional[str] = Field( + default=None, + description="Authentication mode: simple_jwt, session_jwt, oidc, or unset for no authentication.", + ) + JWT_SECRET_KEY: str = Field( + default="", + description=( + "Signing key for session tokens and other signed capabilities. Required on every replica in " + "production; local development may fall back to a key generated on disk." + ), + ) + ENCRYPTION_SECRET_KEY: str = Field( + default="default-docsgpt-encryption-key", + description="Key used to encrypt stored credentials such as tool and connector secrets.", + ) + INTERNAL_KEY: Optional[str] = Field( + default=None, description="Internal API key for worker-to-backend authentication." + ) + + # OIDC SSO (AUTH_TYPE=oidc): any OpenID Connect IdP with discovery (Authentik, Keycloak, ...). + OIDC_ISSUER: Optional[str] = Field( + default=None, + description="OIDC issuer URL with discovery, e.g. https://auth.example.com/application/o/docsgpt/.", + ) + OIDC_CLIENT_ID: Optional[str] = Field(default=None, description="OIDC client id.") + OIDC_CLIENT_SECRET: Optional[str] = Field( + default=None, description="OIDC client secret. Optional; PKCE is always used." + ) + OIDC_SCOPES: str = Field(default="openid profile email", description="Scopes requested from the IdP.") + OIDC_USER_ID_CLAIM: str = Field( + default="sub", description="ID-token claim mapped to the DocsGPT user id." + ) + OIDC_FRONTEND_URL: Optional[str] = Field( + default=None, description="Browser-facing app origin, e.g. http://localhost:5173." + ) + OIDC_REDIRECT_URI: Optional[str] = Field( + default=None, description="Override for the callback URL; default is /api/auth/oidc/callback." + ) + OIDC_SESSION_LIFETIME_SECONDS: int = Field( + default=28800, description="Lifetime of the minted session JWT in seconds (8h)." + ) + OIDC_PROVIDER_NAME: Optional[str] = Field( + default=None, description='Sign-in button label, e.g. "Acme SSO".' + ) + OIDC_ALLOWED_GROUPS: Optional[str] = Field( + default=None, description="Comma-separated group allowlist; unset admits any authenticated user." + ) + OIDC_GROUPS_CLAIM: str = Field( + default="groups", description="ID-token/userinfo claim carrying group membership." + ) + OIDC_ADMIN_GROUPS: Optional[str] = Field( + default=None, description="Comma-separated groups granted admin; unset means no OIDC admin mapping." + ) + + LOCAL_MODE_ADMIN: bool = Field( + default=False, + description=( + "Grant admin without a database role. Persisted admin grants live in user_roles (AUTH_TYPE=oidc " + "only); this is the only non-DB admin path, for AUTH_TYPE=None self-host. MUST stay False if " + "networked." + ), + ) + + # SCIM 2.0 provisioning (IdP-driven user create/deactivate at /scim/v2). + SCIM_ENABLED: bool = Field(default=False, description="Enable SCIM 2.0 provisioning at /scim/v2.") + SCIM_TOKEN: Optional[str] = Field( + default=None, description="Bearer token for IdP SCIM clients (required when SCIM is enabled)." + ) + + @field_validator("INTERNAL_KEY", mode="before") + @classmethod + def _normalize_auth_secrets(cls, v): + return normalize_secret(v) diff --git a/docsgpt/core/settings/connectors.py b/docsgpt/core/settings/connectors.py new file mode 100644 index 00000000..303c00b6 --- /dev/null +++ b/docsgpt/core/settings/connectors.py @@ -0,0 +1,50 @@ +"""OAuth credentials for external source connectors.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class ConnectorSettings(SettingsGroup): + """Client credentials and callback URLs for Google Drive, Microsoft, Confluence, GitHub and MCP.""" + + # Google Drive integration. + GOOGLE_CLIENT_ID: Optional[str] = Field(default=None, description="Google OAuth client id.") + GOOGLE_CLIENT_SECRET: Optional[str] = Field(default=None, description="Google OAuth client secret.") + CONNECTOR_REDIRECT_BASE_URI: Optional[str] = Field( + default="http://127.0.0.1:7091/api/connectors/callback", + description="OAuth callback URL; register it as-is in your provider's console (e.g. GCP).", + ) + CONNECTOR_ALLOWED_ORIGINS: Optional[str] = Field( + default=None, + description=( + "Comma-separated frontend origins allowed to receive connector OAuth results, e.g. " + "https://docsgpt.example.com. The callback origin and OIDC_FRONTEND_URL are always allowed; a " + "loopback callback also allows localhost:5173." + ), + ) + + # Microsoft Entra ID (Azure AD) integration. + MICROSOFT_CLIENT_ID: Optional[str] = Field(default=None, description="Azure AD application (client) id.") + MICROSOFT_CLIENT_SECRET: Optional[str] = Field(default=None, description="Azure AD application client secret.") + MICROSOFT_TENANT_ID: Optional[str] = Field( + default="common", description="Azure AD tenant id, or 'common' for multi-tenant." + ) + MICROSOFT_AUTHORITY: Optional[str] = Field( + default=None, description='Authority URL override, e.g. "https://login.microsoftonline.com/{tenant_id}".' + ) + + # Confluence Cloud integration. + CONFLUENCE_CLIENT_ID: Optional[str] = Field(default=None, description="Confluence Cloud OAuth client id.") + CONFLUENCE_CLIENT_SECRET: Optional[str] = Field(default=None, description="Confluence Cloud OAuth client secret.") + + # GitHub source. + GITHUB_ACCESS_TOKEN: Optional[str] = Field(default=None, description="GitHub PAT with read access to repositories.") + + MCP_OAUTH_REDIRECT_URI: Optional[str] = Field( + default=None, description="Public callback URL for MCP OAuth; unset derives it from CONNECTOR_REDIRECT_BASE_URI." + ) diff --git a/docsgpt/core/settings/database.py b/docsgpt/core/settings/database.py new file mode 100644 index 00000000..e88e03a9 --- /dev/null +++ b/docsgpt/core/settings/database.py @@ -0,0 +1,36 @@ +"""User-data Postgres and schema management at startup.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field, field_validator + +from docsgpt.core.db_uri import normalize_postgres_uri +from docsgpt.core.settings._shared import SettingsGroup + + +class DatabaseSettings(SettingsGroup): + """The Postgres database holding users, conversations and sources, and what startup may do to it.""" + + POSTGRES_URI: Optional[str] = Field(default=None, description="User-data Postgres connection URI.") + AUTO_MIGRATE: bool = Field( + default=True, + description="On startup, apply pending Alembic migrations. Disable if you manage schema out-of-band.", + ) + AUTO_CREATE_DB: bool = Field( + default=True, + description="On startup, create the target Postgres database if missing (needs CREATEDB privilege).", + ) + AUTO_VECTOR_SCHEMA: bool = Field( + default=True, + description=( + "On startup, create the pgvector/graph tables and verify the embedding dimension. No Alembic " + "migration covers the vector DB (it may be a separate cluster); set False to manage it yourself." + ), + ) + + @field_validator("POSTGRES_URI", mode="before") + @classmethod + def _normalize_postgres_uri(cls, v): + return normalize_postgres_uri(v) diff --git a/docsgpt/core/settings/embeddings.py b/docsgpt/core/settings/embeddings.py new file mode 100644 index 00000000..85de4d15 --- /dev/null +++ b/docsgpt/core/settings/embeddings.py @@ -0,0 +1,87 @@ +"""Embedding model selection and where it runs.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field, field_validator + +from docsgpt.core.paths import home_dir +from docsgpt.core.settings._shared import SettingsGroup, normalize_secret + + +class EmbeddingsSettings(SettingsGroup): + """The embedding model, remote or local, and the batching around it.""" + + EMBEDDINGS_NAME: str = Field( + default="huggingface_sentence-transformers/all-mpnet-base-v2", + description=( + "Embedding model. The legacy model is the default on purpose: an install that never pinned this " + "has vectors from it, and granite is the same width so a swap would fail silently. New installs " + "get granite from .env-template; existing ones switch by setting this and running " + "docsgpt.scripts.reembed." + ), + ) + EMBEDDINGS_BASE_URL: Optional[str] = Field( + default=None, description="Remote embeddings API URL (OpenAI-compatible)." + ) + EMBEDDINGS_KEY: Optional[str] = Field( + default=None, description="API key for embeddings (with OpenAI, the same value as API_KEY)." + ) + EMBEDDINGS_MAX_INPUT_TOKENS: Optional[int] = Field( + default=None, description="Truncate each remote embed input to N tokens (overflow is lost)." + ) + EMBEDDINGS_BATCH_SIZE: int = Field( + default=32, description="Chunks per store transaction and per remote embed request." + ) + EMBEDDINGS_MODEL_BATCH_SIZE: int = Field( + default=1, + description=( + "Documents per local ONNX forward pass. Each pass pads to its longest input, and that waste grows " + "with the square of chunk length: at 1250 tokens, 32 peaked at 6.6 GB, 1 at 2.9 GB." + ), + ) + EMBEDDINGS_THREADS: Optional[int] = Field( + default=None, + description=( + "Intra-op threads for the local ONNX runner; unset uses every core. It scales sub-linearly, so " + "several single-threaded workers beat one many-threaded process on the same cores." + ), + ) + EMBEDDINGS_CACHE_DIR: Optional[str] = Field( + default_factory=lambda: str(home_dir() / "models"), + description=( + "Where embedding models and their tokenizers are cached. Persistent by default: FastEmbed's own " + "default is the temp dir." + ), + ) + EMBEDDINGS_POOLING: Optional[str] = Field( + default=None, + description=( + 'Pooling strategy ("cls" or "mean"). Read from the model\'s own repository; set only for a ' + "repository that declares none, or to override what it declares." + ), + ) + EMBEDDINGS_NORMALIZE: Optional[bool] = Field( + default=None, + description=( + "L2-normalise embeddings. Read from the model's own repository; set only for a repository that " + "declares nothing, or to override what it declares." + ), + ) + EMBEDDINGS_DELEGATE_TO_WORKER: bool = Field( + default=True, + description=( + "Embed on the worker so the API holds no model (~890 MB), at one broker round trip per query. " + "Ignored when EMBEDDINGS_BASE_URL is set, which is the better answer for production." + ), + ) + EMBEDDINGS_QUEUE: str = Field(default="embeddings", description="Celery queue the embed task is routed to.") + EMBEDDINGS_DELEGATE_TIMEOUT: int = Field( + default=60, description="Seconds the API waits for the worker to return an embedding." + ) + + @field_validator("EMBEDDINGS_KEY", mode="before") + @classmethod + def _normalize_embeddings_secrets(cls, v): + return normalize_secret(v) diff --git a/docsgpt/core/settings/events.py b/docsgpt/core/settings/events.py new file mode 100644 index 00000000..0ab01d4c --- /dev/null +++ b/docsgpt/core/settings/events.py @@ -0,0 +1,86 @@ +"""Server-sent events, replay journal and remote-device sessions.""" + +from __future__ import annotations + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class EventsSettings(SettingsGroup): + """The internal push channel (notifications and durable replay) and the Redis pool behind it.""" + + ENABLE_SSE_PUSH: bool = Field( + default=True, + description=( + "Internal SSE push channel (notifications and durable replay journal). False makes /api/events emit " + '"push_disabled" and return; clients fall back to polling.' + ), + ) + EVENTS_STREAM_MAXLEN: int = Field( + default=1000, description="Per-user durable backlog cap in entries; ~24h of replay at typical rates." + ) + SSE_KEEPALIVE_SECONDS: int = Field(default=15, ge=1, description="Interval between SSE keepalive comments.") + SSE_MAX_CONCURRENT_PER_USER: int = Field( + default=8, + description=( + "Simultaneous SSE connections per user; each holds a pooled async Redis connection for its lifetime. " + "8 covers multi-tab use without one user starving the pool. 0 disables." + ), + ) + ASYNC_REDIS_MAX_CONNECTIONS: int = Field( + default=2000, + ge=1, + description=( + "Pool size of the async Redis client behind the event-loop routes, per process. Every open " + "notification tab, chat reconnect and device session holds one connection, so this caps concurrent " + "streams per worker (redis-py's own default is 100). Keep the total across workers below the Redis " + "server's maxclients (10000 by default)." + ), + ) + EVENTS_REPLAY_MAX_PER_REQUEST: int = Field( + default=200, + description=( + "Backlog entries XRANGE returns per /api/events snapshot. Bounds what one replay moves from Redis to " + "the wire: a client looping Last-Event-ID reconnects enumerates at most this many per round-trip." + ), + ) + EVENTS_REPLAY_MAX_AGE_HOURS: int = Field(default=48, description="Oldest backlog entry a replay will return.") + EVENTS_REPLAY_BUDGET_REQUESTS_PER_WINDOW: int = Field( + default=30, + description=( + "Sliding-window cap on snapshot replays per user; exhausting it returns 429 with the cursor pinned " + "so the client backs off until the window rolls over." + ), + ) + EVENTS_REPLAY_BUDGET_WINDOW_SECONDS: int = Field(default=60, description="Length of the replay budget window.") + MESSAGE_EVENTS_RETENTION_DAYS: int = Field( + default=14, + description=( + "Retention for the message_events journal, enforced by the cleanup_message_events beat task. Replay " + "only needs streams a client could still be tailing." + ), + ) + + # Remote Device feature. + REMOTE_DEVICE_SESSION_IDLE_SECONDS: int = Field( + default=60, description="Seconds without a heartbeat before a remote-device session is considered idle." + ) + REMOTE_DEVICE_REQUIRE_SIGNATURE: bool = Field( + default=False, description="Require signed commands from remote devices." + ) + REMOTE_DEVICE_PAIRING_TTL_SECONDS: int = Field(default=600, description="Lifetime of a pairing code.") + REMOTE_DEVICE_CMD_QUEUE_TTL_SECONDS: int = Field( + default=900, + description=( + "Redis TTL of the per-device command queue, routing invocations cross-process so a scheduled run " + "reaches the web-held device session. Must exceed the max drain deadline (605s) so a command for a " + "briefly-offline device isn't evicted before its own drain gives up." + ), + ) + REMOTE_DEVICE_INVOCATION_TTL_SECONDS: int = Field( + default=900, description="Redis TTL of a pending remote-device invocation." + ) + REMOTE_DEVICE_OUTPUT_STREAM_MAXLEN: int = Field( + default=10_000, description="Cap on buffered output entries per remote-device invocation stream." + ) diff --git a/docsgpt/core/settings/guardrails.py b/docsgpt/core/settings/guardrails.py new file mode 100644 index 00000000..81ac60f2 --- /dev/null +++ b/docsgpt/core/settings/guardrails.py @@ -0,0 +1,40 @@ +"""Agent guardrails.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class GuardrailSettings(SettingsGroup): + """Input/output checks every agent runs, and the floor no agent may weaken.""" + + GUARDRAILS_ENABLED: bool = Field(default=True, description="Master switch; False disables every stage.") + GUARDRAILS_CHECKS_ENABLED: list = Field( + default=[], description="Allowlist of GuardrailCreator.checks keys; empty means every registered check." + ) + GUARDRAILS_FLOOR: dict = Field( + default={}, + description=( + "A GuardrailsConfig fragment every agent inherits and cannot weaken; agents may add controls or " + 'make an action stricter, never looser. "enabled" is required; without it the floor parses but ' + 'applies to nothing. Example: {"enabled": true, "mode": "scan_all", "controls": [{"check": ' + '"secrets", "stage": "output", "action": "redact"}]}' + ), + ) + GUARDRAILS_JUDGE_MODEL: Optional[str] = Field( + default=None, description="Judge model for the topic/policy checks; unset reuses the request's model." + ) + GUARDRAILS_STORE_SCANNED_TEXT: bool = Field( + default=False, + description=( + "Persist scanned text alongside guardrail_events. Off by default: pre-redaction text is exactly the " + "material a PII control exists to keep out of storage." + ), + ) + GUARDRAILS_EVENTS_RETENTION_DAYS: int = Field( + default=30, ge=1, description="Days guardrail events are kept before the cleanup task removes them." + ) diff --git a/docsgpt/core/settings/ingestion.py b/docsgpt/core/settings/ingestion.py new file mode 100644 index 00000000..38da376d --- /dev/null +++ b/docsgpt/core/settings/ingestion.py @@ -0,0 +1,144 @@ +"""Uploads, document parsing and the size caps that keep one file from taking a worker down.""" + +from __future__ import annotations + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class IngestionSettings(SettingsGroup): + """Upload limits, the parser engine, and per-format byte caps for ingestion and attachments.""" + + UPLOAD_FOLDER: str = Field(default="inputs", description="Directory under the data home for uploaded sources.") + UPLOAD_MAX_REQUEST_BYTES: int = Field( + default=256 * 1024 * 1024, + gt=0, + description="Cap on an upload request body; applied by Flask before multipart parsing.", + ) + UPLOAD_MAX_FILE_BYTES: int = Field( + default=100 * 1024 * 1024, gt=0, description="Cap on a single uploaded file; also enforced while copying." + ) + PARSE_SPEC_MAX_BYTES: int = Field( + default=10 * 1024 * 1024, gt=0, description="Cap on an OpenAPI/tool spec file accepted for parsing." + ) + # ZIP limits apply cumulatively across nested archives in one extraction. + UPLOAD_MAX_ARCHIVE_BYTES: int = Field( + default=250 * 1024 * 1024, gt=0, description="Cap on total bytes extracted from one uploaded archive." + ) + UPLOAD_MAX_ARCHIVE_FILES: int = Field( + default=10_000, gt=0, description="Cap on files extracted from one uploaded archive." + ) + UPLOAD_MAX_ARCHIVE_RATIO: int = Field( + default=1000, gt=0, description="Maximum decompressed-to-compressed ratio before an archive is rejected." + ) + UPLOAD_MAX_ARCHIVE_DEPTH: int = Field( + default=3, ge=0, description="Maximum nesting depth of archives inside archives." + ) + PARSE_PDF_AS_IMAGE: bool = Field(default=False, description="Render PDF pages to images before parsing.") + PARSE_IMAGE_REMOTE: bool = Field(default=False, description="Send images to a remote parser.") + DOC_PARSER_ENGINE: str = Field( + default="anydoc", + description=( + 'Document parser for source ingestion, chat attachments and the read_document tool. "anydoc" ' + "(default): firecrawl-anydoc, a Rust converter with no ML models; milliseconds per file, ~100 MB " + 'peak RSS. "docling": the layout/table-model pipeline (optional install; needed for ' + "read_document's structured output and the docling OCR backend). Files anydoc cannot convert " + "(scanned PDFs, malformed input) fall back to docling when it is installed, otherwise to the native " + "OCR parsers (OCR on) or the legacy parsers. Rollback to the previous behaviour is this one variable." + ), + ) + DOCLING_PIPELINE_QUEUE_MAX_SIZE: int = Field( + default=2, + description=( + "Pages docling's threaded pipeline buffers in flight; the library default (100) drives worker RSS " + "to ~3 GB on a mid-size PDF." + ), + ) + DOCLING_COMPILE_TORCH_MODELS: bool = Field( + default=False, description="Let docling torch.compile its models (slower start, faster pages)." + ) + DOCLING_TABULAR_MAX_BYTES: int = Field( + default=2_000_000, description="Largest CSV/XLSX docling will parse, in bytes." + ) + DOCLING_MARKUP_MAX_BYTES: int = Field( + default=8_000_000, description="Largest HTML/XML docling will parse, in bytes." + ) + MARKUP_MAX_BYTES: int = Field( + default=8_000_000, + description=( + "HTML/XHTML larger than this (bytes) are head-truncated before the markdownify parser runs (the " + "anydoc engine's HTML path). The tree that path builds costs ~50x the input (30 MB of HTML measured " + "at 1.6 GB RSS) and the upload cap is 100 MB, so the gate is what keeps one upload from taking the " + "ingest worker down. 0 disables it." + ), + ) + PDF_TRUST_CHECK: bool = Field( + default=True, + description=( + "Trust-check anydoc's PDF output (docsgpt/parser/file/pdf_trust.py): flag composite (Type0) fonts " + "without a ToUnicode map, and CJK-declaring PDFs whose extracted text has almost no CJK, the two " + "classes where anydoc drops text silently. A flagged file re-parses on the docling fallback when " + "docling is installed; otherwise the anydoc output is kept and the document gets " + 'extra_info["parse_warnings"]. ~30 ms per scanned MB.' + ), + ) + ANYDOC_TABLEIZE: bool = Field( + default=False, + description=( + "Rewrite dot-leader / whitespace-aligned table runs in anydoc's PDF markdown into GFM tables " + "(docsgpt/parser/file/tableize.py). Off by default: it rewrites content on a heuristic (>=3 uniform " + "label+numbers lines) validated only on a small corpus so far." + ), + ) + ATTACHMENT_PDF_TEXT_FAST_PATH: bool = Field( + default=True, + description=( + "Read PDF attachments via their embedded text layer (pypdfium2) instead of docling, falling back to " + "docling when there is no text layer. Attachments go into a prompt, so docling's structural " + "markdown earns far less than the tens of seconds per file it costs; source ingestion is " + "unaffected because chunking and retrieval do depend on that structure." + ), + ) + ATTACHMENT_PDF_TEXT_MIN_MEDIAN_CHARS: int = Field( + default=32, + description=( + "Median chars per sampled page below which a PDF attachment is treated as a scan and handed to " + "docling. Measured on real uploads: scans at 0-17 chars/page, text-layer documents at 433-6834." + ), + ) + ATTACHMENT_TEXT_MAX_BYTES: int = Field(default=5_000_000, description="Cap on extracted attachment text.") + AGENT_IMAGE_MAX_BYTES: int = Field(default=5_000_000, description="Cap on an image passed to an agent.") + AGENT_IMAGE_MAX_PIXELS: int = Field( + default=16_777_216, description="Cap on the pixel count of an image passed to an agent." + ) + GITHUB_INGEST_MAX_FILE_BYTES: int = Field( + default=1048576, description="Skip GitHub repo blobs larger than this (0 = no cap)." + ) + GITHUB_INGEST_MAX_WORKERS: int = Field(default=8, description="Parallel file fetches per GitHub repo ingest.") + + # read_document parsing on a dedicated Celery queue (backend parser). + DOCUMENT_PARSE_QUEUE: str = Field(default="parsing", description="Celery queue the parse_document task is routed to.") + DOCUMENT_PARSE_TIMEOUT: int = Field( + default=120, description="Seconds the read_document tool awaits the enqueued parse before degrading." + ) + DOCUMENT_PARSE_TIMEOUT_PER_MB: int = Field( + default=60, + description=( + "Extra seconds of parse window per MiB of input. The base timeout is a FLOOR: the window grows with " + "document size because OCR cost scales with pages. Without this a large scan is silently dropped at " + "the base window." + ), + ) + DOCUMENT_PARSE_TIMEOUT_MAX: int = Field( + default=900, description="Absolute ceiling on the size-scaled parse window, in seconds." + ) + DOCUMENT_PARSE_MAX_BYTES: int = Field( + default=0, description="Cap on a parsed document's bytes (0 = reuse SANDBOX_MAX_INPUT_BYTES)." + ) + DOCUMENT_MAX_DECOMPRESSED_BYTES: int = Field( + default=300 * 1024 * 1024, description="Cap on bytes decompressed from an archive handed to read_document." + ) + DOCUMENT_MAX_ARCHIVE_ENTRIES: int = Field( + default=10000, description="Cap on entries in an archive handed to read_document." + ) diff --git a/docsgpt/core/settings/llm.py b/docsgpt/core/settings/llm.py new file mode 100644 index 00000000..232ad3be --- /dev/null +++ b/docsgpt/core/settings/llm.py @@ -0,0 +1,121 @@ +"""LLM providers, API keys and per-provider tunables.""" + +from __future__ import annotations + +import os +from typing import Optional + +from pydantic import Field, field_validator + +from docsgpt.core.paths import home_dir +from docsgpt.core.settings._shared import SettingsGroup, normalize_secret + + +class LLMSettings(SettingsGroup): + """Which model answers, how it is reached, and provider-specific behaviour.""" + + LLM_PROVIDER: str = Field(default="docsgpt", description="LLM provider key, e.g. openai, anthropic, docsgpt.") + LLM_NAME: Optional[str] = Field( + default=None, description="Model name for the provider; with openai, e.g. gpt-4 or gpt-3.5-turbo." + ) + API_KEY: Optional[str] = Field(default=None, description="LLM API key used by LLM_PROVIDER.") + + # Provider-specific API keys (for multi-model support). + OPENAI_API_KEY: Optional[str] = Field(default=None, description="OpenAI API key.") + ANTHROPIC_API_KEY: Optional[str] = Field(default=None, description="Anthropic API key.") + GOOGLE_API_KEY: Optional[str] = Field(default=None, description="Google AI API key.") + GROQ_API_KEY: Optional[str] = Field(default=None, description="Groq API key.") + HUGGINGFACE_API_KEY: Optional[str] = Field(default=None, description="Hugging Face API key.") + OPEN_ROUTER_API_KEY: Optional[str] = Field(default=None, description="OpenRouter API key.") + NOVITA_API_KEY: Optional[str] = Field(default=None, description="Novita API key.") + + OPENAI_API_BASE: Optional[str] = Field(default=None, description="Azure OpenAI API base URL.") + OPENAI_API_VERSION: Optional[str] = Field(default=None, description="Azure OpenAI API version.") + AZURE_DEPLOYMENT_NAME: Optional[str] = Field(default=None, description="Azure deployment name for answering.") + AZURE_EMBEDDINGS_DEPLOYMENT_NAME: Optional[str] = Field( + default=None, description="Azure deployment name for embeddings." + ) + OPENAI_BASE_URL: Optional[str] = Field( + default=None, description="Base URL for OpenAI-compatible model servers." + ) + LLM_PATH: str = Field( + default=os.path.join(str(home_dir()), "models/docsgpt-7b-f16.gguf"), + description="Path to the local GGUF model used by the llama.cpp provider.", + ) + + FALLBACK_LLM_PROVIDER: Optional[str] = Field(default=None, description="Provider for the fallback LLM.") + FALLBACK_LLM_NAME: Optional[str] = Field(default=None, description="Model name for the fallback LLM.") + FALLBACK_LLM_API_KEY: Optional[str] = Field(default=None, description="API key for the fallback LLM.") + TITLE_MODEL_ID: Optional[str] = Field( + default=None, description="Optional cheaper model for conversation titles; unset reuses the answer model." + ) + MODELS_CONFIG_DIR: Optional[str] = Field( + default=None, + description=( + "Directory of operator-supplied model YAMLs, loaded after the built-in catalog; later wins on " + "duplicate model id. See docsgpt/core/models/README.md." + ), + ) + DEFAULT_LLM_TOKEN_LIMIT: int = Field( + default=128000, description="Context window assumed when the model is not found in the registry." + ) + RESERVED_TOKENS: dict = Field( + default={"system_prompt": 500, "current_query": 500, "safety_buffer": 1000}, + description="Tokens held back from the context window for the system prompt, the query and a safety buffer.", + ) + CACHE_REDIS_URL: str = Field(default="redis://localhost:6379/2", description="Redis URL for the LLM cache.") + + # OpenAI Responses API. + OPENAI_RESPONSES_STORE: bool = Field( + default=False, + description=( + "True persists Responses API calls server-side so previous_response_id can chain turns. False keeps " + "them stateless, carrying reasoning across the tool loop as encrypted items." + ), + ) + OPENAI_RESPONSES_CHAIN_ACROSS_TURNS: bool = Field( + default=True, + description=( + "Cross-turn previous_response_id chaining (store mode only). The chained transcript lives on the " + "provider and is invisible to every local guard, so it is bounded: a turn starts from the local " + "history when the previous turn's reported prompt already reached the budget (default: the model's " + "context window) or when the conversation was compressed after that turn was produced." + ), + ) + OPENAI_RESPONSES_CHAIN_BUDGET_TOKENS: Optional[int] = Field( + default=None, description="Prompt-token budget for cross-turn chaining; unset uses the model's context window." + ) + OPENAI_RESPONSES_TRUNCATION_AUTO: bool = Field( + default=False, + description=( + 'Send truncation: "auto" so the provider drops the oldest input items instead of failing every ' + "request once a chain exceeds the model's window." + ), + ) + OPENAI_PROMPT_CACHE_KEY: bool = Field( + default=True, + description=( + "Route a user's Responses API calls to the same prompt-cache shard with an opaque per-user key." + ), + ) + OPENAI_PROMPT_CACHE_RETENTION: Optional[str] = Field( + default=None, description="Request extended prompt-cache retention where the provider offers it." + ) + OPENAI_REASONING_SUMMARY: str = Field( + default="auto", description="Reasoning summary mode requested from the Responses API." + ) + + @field_validator( + "API_KEY", + "OPENAI_API_KEY", + "ANTHROPIC_API_KEY", + "GOOGLE_API_KEY", + "GROQ_API_KEY", + "HUGGINGFACE_API_KEY", + "NOVITA_API_KEY", + "FALLBACK_LLM_API_KEY", + mode="before", + ) + @classmethod + def _normalize_llm_secrets(cls, v): + return normalize_secret(v) diff --git a/docsgpt/core/settings/ocr.py b/docsgpt/core/settings/ocr.py new file mode 100644 index 00000000..264b21c5 --- /dev/null +++ b/docsgpt/core/settings/ocr.py @@ -0,0 +1,91 @@ +"""OCR for scanned PDFs and images.""" + +from __future__ import annotations + +from pydantic import AliasChoices, Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class OCRSettings(SettingsGroup): + """Whether OCR runs, which stack performs it, and which engine it uses. + + OCR_ENABLED covers source ingestion, OCR_ATTACHMENTS_ENABLED chat attachments. Which stack performs + it is OCR_BACKEND; which engine, OCR_ENGINE. The DOCLING_OCR_* names are the pre-2026-09 spellings + and stay accepted as aliases. + """ + + OCR_ENABLED: bool = Field( + default=False, + validation_alias=AliasChoices("OCR_ENABLED", "DOCLING_OCR_ENABLED"), + description="OCR scanned PDFs and images during source ingestion.", + ) + OCR_ATTACHMENTS_ENABLED: bool = Field( + default=False, + validation_alias=AliasChoices("OCR_ATTACHMENTS_ENABLED", "DOCLING_OCR_ATTACHMENTS_ENABLED"), + description="OCR scanned PDFs and images attached to a chat.", + ) + OCR_BACKEND: str = Field( + default="auto", + description=( + "Which stack runs OCR when it is on. auto: docling when installed, otherwise native. docling: the " + "layout-model pipeline (hybrid region OCR, reading order, table structure); needs the optional " + "docling extra. native: pypdfium2/Pillow page rendering straight into tesseract or a DeepSeek-OCR " + "endpoint (docsgpt/parser/file/ocr_parser.py); no ML models in the worker, tables come out as text " + "lines under tesseract." + ), + ) + OCR_ENGINE: str = Field( + default="tesseract", + description=( + "OCR engine used when OCR is on. Benched 2026-08 on EN/ZH/table/degraded scans (docs/Guides/ocr has " + "the menu). tesseract (recommended): best classic-engine accuracy (perfect EN word recall, 0.000 " + "bilingual CER, 100% table cells), ~35 MB, CPU-only; needs the system binary and language packs, an " + "optional install like every OCR dependency (build with INSTALL_TESSERACT=true, or apt/brew install " + "tesseract-ocr for a local run); both backends. deepseek: DeepSeek-OCR against an Ollama/vLLM " + "endpoint (OCR_DEEPSEEK_*); best table/CJK quality, the worker stays light (no layout models) but " + "each page costs seconds on the model server; both backends. auto: docling's pick, ocrmac on macOS " + "(excellent), rapidocr on Linux (silently shreds some long text lines; avoid as a server default). " + "ocrmac | rapidocr: force one of those. auto/ocrmac/rapidocr exist only inside docling; the native " + "backend runs tesseract for them. An engine that is not installed degrades (docling: to auto) with " + "a warning instead of failing the parse." + ), + ) + OCR_LANGS: str = Field( + default="eng", + description=( + 'Tesseract language packs, "+"-separated (e.g. "eng+chi_sim+deu"). Other engines keep their own ' + "defaults; their language codes differ." + ), + ) + OCR_DEEPSEEK_URL: str = Field( + default="http://localhost:11434/v1/chat/completions", + description="Chat-completions URL of the DeepSeek-OCR endpoint (Ollama or vLLM).", + ) + OCR_DEEPSEEK_MODEL: str = Field(default="deepseek-ocr:3b", description="Model name at the DeepSeek-OCR endpoint.") + OCR_DEEPSEEK_TIMEOUT: float = Field( + default=300.0, + description=( + "Seconds allowed per page request to the DeepSeek endpoint, on both backends (native sends pages one " + "at a time; docling's VLM pipeline keeps its own concurrency). A 3B model on a laptop needs minutes; " + "a vLLM GPU deployment, seconds." + ), + ) + OCR_RENDER_DPI: int = Field( + default=200, + description=( + "Native backend only: resolution at which pages without a text layer are rendered before OCR. 200 " + "suits tesseract; clamped to 72-600." + ), + ) + OCR_MIN_CHARS_PER_PAGE: int = Field( + default=20, + validation_alias=AliasChoices("OCR_MIN_CHARS_PER_PAGE", "DOCLING_OCR_MIN_CHARS_PER_PAGE"), + description=( + "Chars-per-page floor below which an OCR'd PDF/image parse is treated as an OCR dropout rather than " + "as content (long-running docling workers were observed returning zero characters for every " + "scanned page after a long scanned PDF, with no error). docling retries once on a fresh full-page-OCR " + "converter; both backends then fail loudly instead of indexing an empty document. 0 disables the " + "guard." + ), + ) diff --git a/docsgpt/core/settings/retrieval.py b/docsgpt/core/settings/retrieval.py new file mode 100644 index 00000000..39672d29 --- /dev/null +++ b/docsgpt/core/settings/retrieval.py @@ -0,0 +1,40 @@ +"""Retrieval strategy and GraphRAG.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class RetrievalSettings(SettingsGroup): + """Which vector store answers searches and how retrieval fans out across sources.""" + + VECTOR_STORE: str = Field( + default="faiss", + description="Vector store backend: faiss, elasticsearch, mongodb, qdrant, milvus or pgvector.", + ) + RETRIEVERS_ENABLED: list = Field( + default=["classic", "default"], + description=( + "Retriever keys an agent may use; must match RetrieverCreator.retrievers registry keys, NOT the " + "legacy classic_rag label which never matched the registry." + ), + ) + RETRIEVAL_MAX_PARALLEL_SOURCES: int = Field( + default=4, + description="Concurrent per-source searches in one retrieval; the query is embedded once and shared.", + ) + PER_SOURCE_RETRIEVAL_ENABLED: bool = Field( + default=True, + description="Kill-switch for per-source retrieval dispatch; False collapses to a single retriever.", + ) + GRAPHRAG_ENABLED: bool = Field(default=False, description="Gates graph-aware ingestion and retrieval.") + GRAPHRAG_EXTRACTION_MODEL: Optional[str] = Field( + default=None, description="Model for ingest-time graph extraction; unset reuses LLM_PROVIDER/LLM_NAME." + ) + GRAPHRAG_MAX_CHUNKS_FOR_EXTRACTION: int = Field( + default=2000, description="Hard cap on chunks extracted per source (cost control)." + ) diff --git a/docsgpt/core/settings/sandbox.py b/docsgpt/core/settings/sandbox.py new file mode 100644 index 00000000..958734bc --- /dev/null +++ b/docsgpt/core/settings/sandbox.py @@ -0,0 +1,88 @@ +"""Code-execution sandbox: the Jupyter gateway runner or Daytona Cloud.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class SandboxSettings(SettingsGroup): + """The app is a CLIENT of an always-on runner; defaults are safe so app import never fails unconfigured.""" + + SANDBOX_BACKEND: str = Field( + default="jupyter", description="Sandbox backend: jupyter (self-host) or daytona (Daytona Cloud)." + ) + SANDBOX_GATEWAY_URL: str = Field( + default="http://localhost:8888", + description="URL of the Jupyter Kernel Gateway runner (the docsgpt-sandbox service).", + ) + SANDBOX_GATEWAY_AUTH_TOKEN: Optional[str] = Field(default=None, description="Gateway auth token, if set.") + SANDBOX_KERNEL_NAME: str = Field( + default="docsgpt-python", + description=( + "Kernelspec per session. The env-scrubbing docsgpt-python spec keeps kernel code from reading the " + "gateway token or operator secrets from os.environ; the stock python3 spec inherits the gateway env " + "verbatim and must not be used with untrusted code." + ), + ) + SANDBOX_MAX_TTL: int = Field(default=1200, description="Hard cap (s) on agent-selectable keep-alive TTL.") + SANDBOX_MAX_SESSIONS: int = Field( + default=32, + description=( + "Concurrent live sessions per process, backend-agnostic; at the cap an LRU-idle session is evicted. " + "0 or negative disables the cap." + ), + ) + SANDBOX_EXEC_TIMEOUT: int = Field(default=60, description="Default wall-clock cap (s) per exec call.") + SANDBOX_HTTP_TIMEOUT: int = Field( + default=10, description="Fixed cap (s) for REST control calls (create/delete/alive/interrupt)." + ) + SANDBOX_MAX_OUTPUT_BYTES: int = Field( + default=8 * 1024 * 1024, description="Cap on buffered stdout+stderr per exec." + ) + SANDBOX_MAX_FILE_BYTES: int = Field( + default=10 * 1024 * 1024, description="Cap on get_file size routed through stdout." + ) + SANDBOX_MAX_INPUT_BYTES: int = Field( + default=25 * 1024 * 1024, description="Cap on an input document staged into a sandbox session." + ) + # Runner container caps, consumed by the docsgpt-sandbox compose service, not the app. + SANDBOX_MEMORY: str = Field( + default="1g", + description=( + "Docker mem_limit for the runner container. Consumed by the docsgpt-sandbox compose service, not " + "the app; part of the untrusted-code security boundary." + ), + ) + SANDBOX_CPUS: str = Field( + default="1.0", + description=( + "Docker CPU quota for the runner container. Consumed by the docsgpt-sandbox compose service, not " + "the app; part of the untrusted-code security boundary." + ), + ) + + # Daytona Cloud backend (SANDBOX_BACKEND=daytona). All knobs are optional so app import never fails + # when the backend is unused. + DAYTONA_API_KEY: Optional[str] = Field(default=None, description="Daytona Cloud API key (secret).") + DAYTONA_API_URL: Optional[str] = Field( + default=None, description="Override for the Daytona API base URL, if self-targeting." + ) + DAYTONA_TARGET: Optional[str] = Field(default=None, description='Daytona region/target, e.g. "us".') + DAYTONA_SNAPSHOT: Optional[str] = Field( + default=None, + description="Image for new sandboxes; render libs via scripts/build_daytona_snapshot.py.", + ) + DAYTONA_LANGUAGE: str = Field(default="python", description="Default runtime language for created sandboxes.") + DAYTONA_AUTO_STOP_INTERVAL: int = Field( + default=15, description="Minutes idle before Daytona auto-stops a sandbox (0 disables)." + ) + DAYTONA_AUTO_DELETE_INTERVAL: int = Field( + default=60, description="Minutes after stop before Daytona auto-deletes a sandbox (-1 disables)." + ) + DAYTONA_MAX_SANDBOXES: int = Field( + default=50, description="Cap on concurrent live Daytona sandboxes (cost-DoS guard)." + ) diff --git a/docsgpt/core/settings/scheduler.py b/docsgpt/core/settings/scheduler.py new file mode 100644 index 00000000..00a9f455 --- /dev/null +++ b/docsgpt/core/settings/scheduler.py @@ -0,0 +1,28 @@ +"""Scheduled agent runs (see scheduler.md).""" + +from __future__ import annotations + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class SchedulerSettings(SettingsGroup): + """Cadence, quotas and timeouts of scheduled runs.""" + + SCHEDULE_DISPATCHER_INTERVAL: int = Field( + default=30, description="Seconds between dispatcher passes that enqueue due schedules." + ) + SCHEDULE_MIN_INTERVAL: int = Field(default=900, description="Smallest allowed recurrence interval in seconds.") + SCHEDULE_MAX_PER_USER: int = Field(default=50, description="Cap on schedules a user may own.") + SCHEDULE_RUN_TIMEOUT: int = Field(default=600, description="Wall-clock cap on one scheduled run, in seconds.") + SCHEDULE_MISFIRE_GRACE: int = Field( + default=60, description="Seconds past the due time within which a missed run still fires." + ) + SCHEDULE_AUTOPAUSE_FAILURES: int = Field( + default=3, description="Consecutive failures after which a schedule is paused automatically." + ) + SCHEDULE_ONCE_MAX_HORIZON: int = Field( + default=31_536_000, description="How far ahead a one-off run may be scheduled, in seconds (one year)." + ) + SCHEDULE_RUN_OUTPUT_RETENTION_DAYS: int = Field(default=90, description="Days scheduled-run output is kept.") diff --git a/docsgpt/core/settings/server.py b/docsgpt/core/settings/server.py new file mode 100644 index 00000000..84bb2510 --- /dev/null +++ b/docsgpt/core/settings/server.py @@ -0,0 +1,39 @@ +"""The API process itself.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class ServerSettings(SettingsGroup): + """Serving the UI, public URLs, and process-level knobs of the API server.""" + + SERVE_UI: bool = Field( + default=True, description="Serve the web UI shipped in the package (docsgpt/static) from the API process." + ) + FLASK_DEBUG_MODE: bool = Field(default=False, description="Run Flask in debug mode.") + VERSION_CHECK: bool = Field(default=True, description="Anonymous startup version check for security issues.") + PUBLIC_API_BASE_URL: Optional[str] = Field( + default=None, description="Public base URL for user-facing endpoint references in prompts." + ) + GRACEFUL_SHUTDOWN_TIMEOUT_SECONDS: int = Field( + default=30, + description=( + "Bounds uvicorn's shutdown drain (uvicorn_worker doesn't forward --graceful-timeout). Keep below the " + "gunicorn --timeout (180) watchdog. Used by BoundedDrainUvicornWorker." + ), + ) + WSGI_THREADPOOL_WORKERS: int = Field( + default=96, description="Threads serving the WSGI (Flask) part of the app under the ASGI server." + ) + V1_SESSION_TTL_SECONDS: int = Field( + default=24 * 60 * 60, + description=( + "Lets OpenAI-compatible clients identify a logical chat by session header, which chat-completions " + "itself has no field for; TTL of that session mapping." + ), + ) diff --git a/docsgpt/core/settings/speech.py b/docsgpt/core/settings/speech.py new file mode 100644 index 00000000..465c8fc5 --- /dev/null +++ b/docsgpt/core/settings/speech.py @@ -0,0 +1,31 @@ +"""Text-to-speech and speech-to-text.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field, field_validator + +from docsgpt.core.settings._shared import SettingsGroup, normalize_secret + + +class SpeechSettings(SettingsGroup): + """Voice providers and transcription options.""" + + TTS_PROVIDER: str = Field( + default="google_tts", description="Text-to-speech provider: google_tts, elevenlabs, or none to switch it off." + ) + ELEVENLABS_API_KEY: Optional[str] = Field(default=None, description="ElevenLabs API key.") + STT_PROVIDER: str = Field( + default="openai", description="Speech-to-text provider: openai, faster_whisper, or none to switch it off." + ) + OPENAI_STT_MODEL: str = Field(default="gpt-4o-mini-transcribe", description="OpenAI transcription model.") + STT_LANGUAGE: Optional[str] = Field(default=None, description="Language hint for transcription; unset auto-detects.") + STT_MAX_FILE_SIZE_MB: int = Field(default=50, description="Cap on an audio file accepted for transcription.") + STT_ENABLE_TIMESTAMPS: bool = Field(default=False, description="Return word/segment timestamps.") + STT_ENABLE_DIARIZATION: bool = Field(default=False, description="Label speakers in the transcript.") + + @field_validator("ELEVENLABS_API_KEY", mode="before") + @classmethod + def _normalize_speech_secrets(cls, v): + return normalize_secret(v) diff --git a/docsgpt/core/settings/storage.py b/docsgpt/core/settings/storage.py new file mode 100644 index 00000000..ac970a54 --- /dev/null +++ b/docsgpt/core/settings/storage.py @@ -0,0 +1,49 @@ +"""Where uploaded files and generated artifacts are stored.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class StorageSettings(SettingsGroup): + """Local disk or an S3-compatible bucket, and how download URLs are produced.""" + + STORAGE_TYPE: str = Field(default="local", description="File storage backend: local or s3.") + URL_STRATEGY: str = Field( + default="backend", + description="How download links are produced: backend (streamed through the API) or s3 (presigned URLs).", + ) + + # S3-compatible object storage (STORAGE_TYPE=s3): AWS S3, MinIO, R2, B2, Spaces, ... + # For non-AWS, set S3_ENDPOINT_URL and usually S3_PATH_STYLE=true. + S3_BUCKET_NAME: str = Field(default="docsgpt-test-bucket", description="Bucket name.") + S3_ENDPOINT_URL: Optional[str] = Field( + default=None, description="Custom endpoint for S3-compatible services (MinIO, R2, B2, Spaces); omit for AWS." + ) + S3_ACCESS_KEY_ID: Optional[str] = Field(default=None, description="Access key id.") + S3_SECRET_ACCESS_KEY: Optional[str] = Field(default=None, description="Secret access key.") + S3_REGION: Optional[str] = Field(default=None, description='AWS region; use "auto" for Cloudflare R2.') + S3_PATH_STYLE: bool = Field( + default=False, description="Path-style addressing (required by most non-AWS services)." + ) + + # Legacy AWS credentials from the retired SageMaker provider. + SAGEMAKER_REGION: Optional[str] = Field( + default=None, + description="Legacy AWS region from the retired SageMaker provider; deprecated fallback for S3_REGION.", + ) + SAGEMAKER_ACCESS_KEY: Optional[str] = Field( + default=None, + description="Legacy AWS access key from the retired SageMaker provider; deprecated fallback for S3_ACCESS_KEY_ID.", + ) + SAGEMAKER_SECRET_KEY: Optional[str] = Field( + default=None, + description=( + "Legacy AWS secret key from the retired SageMaker provider; deprecated fallback for " + "S3_SECRET_ACCESS_KEY." + ), + ) diff --git a/docsgpt/core/settings/vectorstores.py b/docsgpt/core/settings/vectorstores.py new file mode 100644 index 00000000..13c504df --- /dev/null +++ b/docsgpt/core/settings/vectorstores.py @@ -0,0 +1,89 @@ +"""Connection settings for each vector store backend.""" + +from __future__ import annotations + +from typing import Optional + +from pydantic import Field, field_validator + +from docsgpt.core.db_uri import normalize_pgvector_connection_string +from docsgpt.core.paths import home_dir +from docsgpt.core.settings._shared import SettingsGroup, normalize_secret + + +class VectorStoreSettings(SettingsGroup): + """Per-backend connection details; only the backend named by VECTOR_STORE is read.""" + + MONGO_URI: Optional[str] = Field( + default=None, + description=( + "Only consulted when VECTOR_STORE=mongodb or when running scripts/db/backfill.py; user data lives " + "in Postgres." + ), + ) + + # Elasticsearch. + ELASTIC_CLOUD_ID: Optional[str] = Field(default=None, description="Elastic Cloud id.") + ELASTIC_USERNAME: Optional[str] = Field(default=None, description="Elasticsearch username.") + ELASTIC_PASSWORD: Optional[str] = Field(default=None, description="Elasticsearch password.") + ELASTIC_URL: Optional[str] = Field(default=None, description="Elasticsearch URL.") + ELASTIC_INDEX: Optional[str] = Field(default="docsgpt", description="Elasticsearch index name.") + + # Qdrant. + QDRANT_COLLECTION_NAME: Optional[str] = Field(default="docsgpt", description="Qdrant collection name.") + QDRANT_LOCATION: Optional[str] = Field(default=None, description="Qdrant location (':memory:' or a URL).") + QDRANT_URL: Optional[str] = Field(default=None, description="Qdrant server URL.") + QDRANT_PORT: Optional[int] = Field(default=6333, description="Qdrant REST port.") + QDRANT_GRPC_PORT: int = Field(default=6334, description="Qdrant gRPC port.") + QDRANT_PREFER_GRPC: bool = Field(default=False, description="Use gRPC instead of REST where possible.") + QDRANT_HTTPS: Optional[bool] = Field(default=None, description="Use HTTPS for the Qdrant connection.") + QDRANT_API_KEY: Optional[str] = Field(default=None, description="Qdrant API key.") + QDRANT_PREFIX: Optional[str] = Field(default=None, description="URL prefix for a Qdrant behind a proxy.") + QDRANT_TIMEOUT: Optional[float] = Field(default=None, description="Qdrant request timeout in seconds.") + QDRANT_HOST: Optional[str] = Field(default=None, description="Qdrant host (alternative to QDRANT_URL).") + QDRANT_PATH: Optional[str] = Field(default=None, description="Path for an embedded on-disk Qdrant.") + QDRANT_DISTANCE_FUNC: str = Field(default="Cosine", description="Qdrant distance function.") + + # PGVector. + PGVECTOR_CONNECTION_STRING: Optional[str] = Field( + default=None, + description=( + "pgvector connection string. postgres://, postgresql:// and postgresql+psycopg:// are all accepted " + "and normalized internally for psycopg.connect(). Unset falls back to POSTGRES_URI." + ), + ) + PGVECTOR_POOL_MAX_SIZE: int = Field( + default=8, description="Per-process connection pool size; 0 uses one direct connection per store." + ) + PGVECTOR_IVFFLAT_PROBES: Optional[int] = Field( + default=None, + description="IVFFlat probes; unset derives sqrt(lists) from the index. Higher means better recall, more scan.", + ) + + # Milvus. + MILVUS_COLLECTION_NAME: Optional[str] = Field(default="docsgpt", description="Milvus collection name.") + MILVUS_URI: Optional[str] = Field( + default_factory=lambda: str(home_dir() / "milvus_local.db"), + description=( + "Milvus server URI. The default is a milvus-lite (embedded) database file under the data home, " + "like the other local stores." + ), + ) + MILVUS_TOKEN: Optional[str] = Field(default="", description="Milvus auth token.") + + # LanceDB. + LANCEDB_PATH: str = Field( + default_factory=lambda: str(home_dir() / "data" / "lancedb"), + description="LanceDB local data directory.", + ) + LANCEDB_TABLE_NAME: Optional[str] = Field(default="docsgpts", description="LanceDB table for stored vectors.") + + @field_validator("PGVECTOR_CONNECTION_STRING", mode="before") + @classmethod + def _normalize_pgvector_connection_string(cls, v): + return normalize_pgvector_connection_string(v) + + @field_validator("QDRANT_API_KEY", mode="before") + @classmethod + def _normalize_vectorstore_secrets(cls, v): + return normalize_secret(v) diff --git a/docsgpt/core/settings/workers.py b/docsgpt/core/settings/workers.py new file mode 100644 index 00000000..fbc7daae --- /dev/null +++ b/docsgpt/core/settings/workers.py @@ -0,0 +1,37 @@ +"""Celery broker, result backend and worker process limits.""" + +from __future__ import annotations + +from pydantic import Field + +from docsgpt.core.settings._shared import SettingsGroup + + +class WorkerSettings(SettingsGroup): + """How background tasks are queued and how worker processes are recycled.""" + + CELERY_BROKER_URL: str = Field(default="redis://localhost:6379/0", description="Celery broker URL.") + CELERY_RESULT_BACKEND: str = Field(default="redis://localhost:6379/1", description="Celery result backend URL.") + CELERY_WORKER_PREFETCH_MULTIPLIER: int = Field( + default=1, description="Tasks prefetched per worker process; 1 caps SIGKILL loss to one task." + ) + CELERY_VISIBILITY_TIMEOUT: int = Field( + default=3600, + description=( + "Broker visibility timeout in seconds. Must exceed the longest legitimate task runtime but stay " + "short enough that SIGKILLed tasks redeliver promptly." + ), + ) + CELERY_WORKER_MAX_MEMORY_PER_CHILD: int = Field( + default=4194304, + description=( + "Recycle a prefork child past this resident size in KB; backstops docling/torch heap growth. " + "Checked between tasks, so it does not bound the peak within one. 0 disables." + ), + ) + CELERY_WORKER_MAX_TASKS_PER_CHILD: int = Field( + default=0, description="Recycle a worker child after N tasks; 0 disables." + ) + API_URL: str = Field( + default="http://localhost:7091", description="Backend URL the Celery worker calls back into." + ) diff --git a/tests/test_remaining_coverage.py b/tests/test_remaining_coverage.py index b251a6b1..89fbded2 100644 --- a/tests/test_remaining_coverage.py +++ b/tests/test_remaining_coverage.py @@ -421,7 +421,7 @@ class TestBaseLLMAbstractRawGen: # --------------------------------------------------------------------------- -# docsgpt/core/settings.py (line 184 - clean_none_string) +# docsgpt/core/settings (normalize_api_key) # --------------------------------------------------------------------------- @pytest.mark.unit class TestSettingsNormalizeApiKey: