mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-03 07:11:56 +00:00
refactor: rename the application package to docsgpt
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
This commit is contained in:
1 parent
3e7f1f91b4
commit
574f96341e
984 files changed
+9847
-9682
No files matched your search
@@ -19,17 +19,17 @@ Run the full app under uvicorn (serves `/mcp` and the async SSE reconnect
|
||||
routes, and matches production):
|
||||
|
||||
```bash
|
||||
uvicorn application.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
|
||||
uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
|
||||
```
|
||||
|
||||
`flask --app application/app.py run --host=0.0.0.0 --port=7091` is faster but
|
||||
`flask --app docsgpt/app.py run --host=0.0.0.0 --port=7091` is faster but
|
||||
serves only the WSGI Flask app — it omits `/mcp` and the reconnect reader
|
||||
`GET /api/messages/<id>/events`, so a dropped stream won't auto-resume.
|
||||
|
||||
### Celery (Task Queue)
|
||||
|
||||
```bash
|
||||
celery -A application.app.celery worker -l INFO -Q docsgpt,parsing,embeddings
|
||||
celery -A docsgpt.app.celery worker -l INFO -Q docsgpt,parsing,embeddings
|
||||
```
|
||||
|
||||
The `parsing` queue serves document parsing (the `read_document` tool / workflow
|
||||
|
||||
@@ -22,8 +22,8 @@ fi
|
||||
|
||||
|
||||
# The embedding model is fetched on first use and cached, so nothing to download
|
||||
# here. For an offline container, run `python -m application.scripts.prefetch_models`
|
||||
# here. For an offline container, run `python -m docsgpt.scripts.prefetch_models`
|
||||
# after the install below.
|
||||
pip install -r application/requirements.txt
|
||||
pip install -r docsgpt/requirements.txt
|
||||
cd frontend
|
||||
npm install --include=dev
|
||||
@@ -0,0 +1,23 @@
|
||||
# Build context for docsgpt/Dockerfile is the repository root, so the image can
|
||||
# carry the `application` import alias next to the `docsgpt` package. Allow only
|
||||
# what the image needs; everything else (frontend, docs, tests, venvs) stays out.
|
||||
*
|
||||
!docsgpt/
|
||||
!application/
|
||||
|
||||
# Inside the package: caches, local runtime data and secrets never ship.
|
||||
**/__pycache__/
|
||||
**/*.py[cod]
|
||||
docsgpt/.pytest_cache/
|
||||
docsgpt/.ruff_cache/
|
||||
docsgpt/.coverage
|
||||
docsgpt/htmlcov/
|
||||
docsgpt/*.log
|
||||
docsgpt/indexes/
|
||||
docsgpt/inputs/
|
||||
docsgpt/vectors/
|
||||
docsgpt/*.faiss
|
||||
docsgpt/*.pkl
|
||||
docsgpt/.env
|
||||
docsgpt/.env.*
|
||||
docsgpt/Dockerfile
|
||||
+1
-1
@@ -21,7 +21,7 @@ INTERNAL_KEY=<internal key for worker-to-backend authentication>
|
||||
# searching a different vector space than the stored vectors -- which fails
|
||||
# silently, because both models are 768-dimensional. To switch, set it and then
|
||||
# run:
|
||||
# python -m application.scripts.reembed
|
||||
# python -m docsgpt.scripts.reembed
|
||||
# EMBEDDINGS_NAME=ibm-granite/granite-embedding-311m-multilingual-r2
|
||||
|
||||
# Remote Embeddings (Optional - for using a remote embeddings API instead of
|
||||
|
||||
@@ -10,7 +10,7 @@ DocsGPT ingests content (files/URLs/connectors), indexes it, and answers queries
|
||||
|
||||
Core components:
|
||||
- Backend API (`application/`)
|
||||
- Workers/ingestion (`application/worker.py` and related modules)
|
||||
- Workers/ingestion (`docsgpt/worker.py` and related modules)
|
||||
- Datastores (MongoDB/Redis/vector stores)
|
||||
- Frontend (`frontend/`)
|
||||
- Optional extensions/integrations (`extensions/`)
|
||||
|
||||
+1
-1
@@ -8,7 +8,7 @@ github:
|
||||
|
||||
application:
|
||||
- changed-files:
|
||||
- any-glob-to-any-file: 'application/**/*'
|
||||
- any-glob-to-any-file: 'docsgpt/**/*'
|
||||
|
||||
docs:
|
||||
- changed-files:
|
||||
|
||||
@@ -4,7 +4,7 @@ on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths:
|
||||
- 'application/version.py'
|
||||
- 'docsgpt/version.py'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
@@ -23,12 +23,12 @@ jobs:
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Read version from application/version.py
|
||||
- name: Read version from docsgpt/version.py
|
||||
id: ver
|
||||
run: |
|
||||
VERSION=$(python3 -c "g={}; exec(open('application/version.py').read(), g); print(g['__version__'])")
|
||||
VERSION=$(python3 -c "g={}; exec(open('docsgpt/version.py').read(), g); print(g['__version__'])")
|
||||
if [ -z "$VERSION" ]; then
|
||||
echo "::error::Could not read __version__ from application/version.py"
|
||||
echo "::error::Could not read __version__ from docsgpt/version.py"
|
||||
exit 1
|
||||
fi
|
||||
echo "version=$VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
@@ -28,13 +28,13 @@ jobs:
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
pip install bandit # Bandit is needed for this action
|
||||
if [ -f application/requirements.txt ]; then pip install -r application/requirements.txt; fi
|
||||
if [ -f docsgpt/requirements.txt ]; then pip install -r docsgpt/requirements.txt; fi
|
||||
|
||||
- name: Run Bandit scan
|
||||
uses: PyCQA/bandit-action@v1
|
||||
with:
|
||||
severity: medium
|
||||
confidence: medium
|
||||
targets: application/
|
||||
targets: docsgpt/
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -63,9 +63,9 @@ jobs:
|
||||
- name: Build and push platform-specific images
|
||||
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
|
||||
with:
|
||||
file: './application/Dockerfile'
|
||||
file: './docsgpt/Dockerfile'
|
||||
platforms: ${{ matrix.platform }}
|
||||
context: ./application
|
||||
context: .
|
||||
push: true
|
||||
build-args: |
|
||||
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
|
||||
|
||||
@@ -65,9 +65,9 @@ jobs:
|
||||
- name: Build and push platform-specific images
|
||||
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
|
||||
with:
|
||||
file: './application/Dockerfile'
|
||||
file: './docsgpt/Dockerfile'
|
||||
platforms: ${{ matrix.platform }}
|
||||
context: ./application
|
||||
context: .
|
||||
push: true
|
||||
build-args: |
|
||||
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
|
||||
|
||||
@@ -8,14 +8,15 @@ on:
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'application/Dockerfile'
|
||||
- 'application/.dockerignore'
|
||||
- 'docsgpt/Dockerfile'
|
||||
- '.dockerignore'
|
||||
- 'application/**'
|
||||
- 'application/requirements*.txt'
|
||||
- 'application/scripts/prefetch_models.py'
|
||||
- 'application/scripts/verify_offline.py'
|
||||
- 'application/vectorstore/model_registry.py'
|
||||
- 'application/parser/tokenization.py'
|
||||
- 'application/vectorstore/embeddings_local.py'
|
||||
- 'docsgpt/scripts/prefetch_models.py'
|
||||
- 'docsgpt/scripts/verify_offline.py'
|
||||
- 'docsgpt/vectorstore/model_registry.py'
|
||||
- 'docsgpt/parser/tokenization.py'
|
||||
- 'docsgpt/vectorstore/embeddings_local.py'
|
||||
- '.github/workflows/docker-image-verify.yml'
|
||||
|
||||
permissions:
|
||||
@@ -40,8 +41,8 @@ jobs:
|
||||
- name: Build the image
|
||||
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
|
||||
with:
|
||||
file: ./application/Dockerfile
|
||||
context: ./application
|
||||
file: ./docsgpt/Dockerfile
|
||||
context: .
|
||||
platforms: linux/amd64
|
||||
load: true
|
||||
tags: docsgpt:verify${{ matrix.variant }}
|
||||
@@ -63,4 +64,4 @@ jobs:
|
||||
IMAGE: docsgpt:verify${{ matrix.variant }}
|
||||
run: |
|
||||
docker run --rm --network none "$IMAGE" \
|
||||
python -m application.scripts.verify_offline
|
||||
python -m docsgpt.scripts.verify_offline
|
||||
@@ -36,4 +36,4 @@ jobs:
|
||||
run: |
|
||||
uv lock --check
|
||||
bash scripts/export_requirements.sh
|
||||
git diff --exit-code -- application/requirements.txt application/requirements-docling.txt application/requirements-milvus.txt
|
||||
git diff --exit-code -- docsgpt/requirements.txt docsgpt/requirements-docling.txt docsgpt/requirements-milvus.txt
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
|
||||
- name: Test with pytest and generate coverage report
|
||||
run: |
|
||||
python -m pytest -n auto --cov=application --cov-report=xml --cov-report=term-missing
|
||||
python -m pytest -n auto --cov=docsgpt --cov-report=xml --cov-report=term-missing
|
||||
- name: Upload coverage reports to Codecov
|
||||
if: github.event_name == 'pull_request' && matrix.python-version == '3.12'
|
||||
uses: codecov/codecov-action@v5
|
||||
|
||||
+2
-2
@@ -179,7 +179,7 @@ frontend/*.njsproj
|
||||
frontend/*.sln
|
||||
frontend/*.sw?
|
||||
|
||||
application/vectors/
|
||||
docsgpt/vectors/
|
||||
|
||||
**/inputs
|
||||
|
||||
@@ -202,6 +202,6 @@ tests/e2e/node_modules/
|
||||
tests/e2e/playwright-report/
|
||||
tests/e2e/test-results/
|
||||
tests/e2e/.e2e-last-run.json
|
||||
application/core/models/foundry_test_internal.yaml
|
||||
docsgpt/core/models/foundry_test_internal.yaml
|
||||
bench/
|
||||
/benchmark/
|
||||
Vendored
+2
-2
@@ -14,7 +14,7 @@
|
||||
"request": "launch",
|
||||
"module": "flask",
|
||||
"env": {
|
||||
"FLASK_APP": "application/app.py",
|
||||
"FLASK_APP": "docsgpt/app.py",
|
||||
"PYTHONPATH": "${workspaceFolder}",
|
||||
"FLASK_ENV": "development",
|
||||
"FLASK_DEBUG": "1",
|
||||
@@ -38,7 +38,7 @@
|
||||
},
|
||||
"args": [
|
||||
"-A",
|
||||
"application.app.celery",
|
||||
"docsgpt.app.celery",
|
||||
"worker",
|
||||
"-l",
|
||||
"INFO",
|
||||
|
||||
@@ -28,17 +28,17 @@ Use these commands once the dev prerequisites above are satisfied.
|
||||
|
||||
```bash
|
||||
source .venv/bin/activate # macOS/Linux
|
||||
uv pip install -r application/requirements.txt # or: pip install -r application/requirements.txt
|
||||
uv pip install -r docsgpt/requirements.txt # or: pip install -r docsgpt/requirements.txt
|
||||
# Optional extras (not installed by default; each file = core + the extra):
|
||||
# uv pip install -r application/requirements-docling.txt # docling parser engine (OCR backend, structured output)
|
||||
# uv pip install -r application/requirements-milvus.txt # VECTOR_STORE=milvus
|
||||
# uv pip install -r docsgpt/requirements-docling.txt # docling parser engine (OCR backend, structured output)
|
||||
# uv pip install -r docsgpt/requirements-milvus.txt # VECTOR_STORE=milvus
|
||||
# With uv alone: `uv sync --extra docling` (pyproject.toml + uv.lock are the source of truth).
|
||||
# `uv pip install -r application/requirements-docling.txt` needs UV_INDEX_STRATEGY=unsafe-best-match
|
||||
# `uv pip install -r docsgpt/requirements-docling.txt` needs UV_INDEX_STRATEGY=unsafe-best-match
|
||||
# (the file adds the PyTorch CPU index; prefer `uv sync --extra docling`).
|
||||
```
|
||||
|
||||
Dependencies are declared in `pyproject.toml` and locked in `uv.lock`; the
|
||||
`application/requirements*.txt` files are exported from the lock. To add or
|
||||
`docsgpt/requirements*.txt` files are exported from the lock. To add or
|
||||
bump a package: edit `pyproject.toml`, run `uv lock`, then
|
||||
`bash scripts/export_requirements.sh` (CI fails if the exports are stale).
|
||||
Never edit the requirements files by hand.
|
||||
@@ -47,13 +47,13 @@ Run the API. For local dev, prefer the ASGI entrypoint under uvicorn — it
|
||||
serves the **whole** app, matches production, and hot-reloads:
|
||||
|
||||
```bash
|
||||
uvicorn application.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
|
||||
uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
|
||||
```
|
||||
|
||||
`flask --app application/app.py run --host=0.0.0.0 --port=7091` is a faster
|
||||
`flask --app docsgpt/app.py run --host=0.0.0.0 --port=7091` is a faster
|
||||
inner loop (quick startup, the Werkzeug interactive debugger), but it serves
|
||||
**only** the WSGI Flask app and omits the routes mounted on the ASGI shell
|
||||
in `application/asgi.py`:
|
||||
in `docsgpt/asgi.py`:
|
||||
|
||||
- the `/mcp` FastMCP endpoint, and
|
||||
- the native-async SSE reconnect reader `GET /api/messages/<id>/events`.
|
||||
@@ -63,13 +63,13 @@ Flask route), but a stream interrupted by a disconnect won't auto-resume on
|
||||
reconnect. Use `flask run` only when you don't need those routes.
|
||||
|
||||
Production uses `gunicorn -k uvicorn_worker.UvicornWorker` against the same
|
||||
`application.asgi:asgi_app` target; see `application/Dockerfile` for the
|
||||
`docsgpt.asgi:asgi_app` target; see `docsgpt/Dockerfile` for the
|
||||
full flag set.
|
||||
|
||||
Run the Celery worker in a separate terminal:
|
||||
|
||||
```bash
|
||||
celery -A application.app.celery worker -l INFO
|
||||
celery -A docsgpt.app.celery worker -l INFO
|
||||
```
|
||||
|
||||
**The worker is required for retrieval, not optional.** `EMBEDDINGS_DELEGATE_TO_WORKER`
|
||||
@@ -83,7 +83,7 @@ loading a model of its own — which keeps the API process around 285 MB instead
|
||||
On macOS, prefer the solo pool for Celery:
|
||||
|
||||
```bash
|
||||
python -m celery -A application.app.celery worker -l INFO --pool=solo
|
||||
python -m celery -A docsgpt.app.celery worker -l INFO --pool=solo
|
||||
```
|
||||
|
||||
Note that `--pool=solo` costs roughly 350 ms per query embed against ~55 ms on the
|
||||
@@ -157,7 +157,7 @@ vale .
|
||||
|
||||
## Repository map
|
||||
|
||||
- `application/`: Flask backend, API routes, agent logic, retrieval, parsing, security, storage, Celery worker, and WSGI entrypoints.
|
||||
- `docsgpt/`: Flask backend, API routes, agent logic, retrieval, parsing, security, storage, Celery worker, and WSGI entrypoints.
|
||||
- `tests/`: backend unit/integration tests and test-only Python dependencies.
|
||||
- `frontend/`: Vite + React + TypeScript application.
|
||||
- `frontend/src/`: main UI code, including `components`, `conversation`, `hooks`, `locale`, `settings`, `upload`, and Redux store wiring in `store.ts`.
|
||||
@@ -177,12 +177,12 @@ vale .
|
||||
|
||||
### Backend Abstractions
|
||||
|
||||
- LLM providers implement a common interface in `application/llm/` (add new providers by extending the base class).
|
||||
- Vector stores are abstracted in `application/vectorstore/`.
|
||||
- Parsers live in `application/parser/` and handle different document formats in the ingestion stage.
|
||||
- Agents and tools are in `application/agents/` and `application/agents/tools/`.
|
||||
- Celery setup/config lives in `application/celery_init.py` and `application/celeryconfig.py`.
|
||||
- Settings and env vars are managed via Pydantic in `application/core/settings.py`.
|
||||
- LLM providers implement a common interface in `docsgpt/llm/` (add new providers by extending the base class).
|
||||
- Vector stores are abstracted in `docsgpt/vectorstore/`.
|
||||
- Parsers live in `docsgpt/parser/` and handle different document formats in the ingestion stage.
|
||||
- Agents and tools are in `docsgpt/agents/` and `docsgpt/agents/tools/`.
|
||||
- Celery setup/config lives in `docsgpt/celery_init.py` and `docsgpt/celeryconfig.py`.
|
||||
- Settings and env vars are managed via Pydantic in `docsgpt/core/settings.py`.
|
||||
|
||||
### Frontend
|
||||
|
||||
|
||||
+1
-1
@@ -49,7 +49,7 @@ Tech Stack Overview:
|
||||
|
||||
### 🖥 Backend Contributions (🐍 Python)
|
||||
|
||||
- Review our issues and contribute to [`/application`](https://github.com/arc53/DocsGPT/tree/main/application)
|
||||
- Review our issues and contribute to [`/docsgpt`](https://github.com/arc53/DocsGPT/tree/main/docsgpt)
|
||||
- All new code should be covered with unit tests ([pytest](https://github.com/pytest-dev/pytest)). Please find tests under [`/tests`](https://github.com/arc53/DocsGPT/tree/main/tests) folder.
|
||||
- Before submitting your Pull Request, ensure it can be queried after ingesting some test data.
|
||||
- **Coding Style:** We adhere to the [PEP 8](https://www.python.org/dev/peps/pep-0008/) style guide for Python code. We use `ruff` as our linter and code formatter. Please ensure your code is formatted correctly and passes `ruff` checks before submitting.
|
||||
|
||||
@@ -132,7 +132,7 @@ Please refer to the [CONTRIBUTING.md](CONTRIBUTING.md) file for information abou
|
||||
|
||||
## Project Structure
|
||||
|
||||
- **Application** - Backend Flask application.
|
||||
- **docsgpt** - Backend Flask application (the `docsgpt` Python package).
|
||||
|
||||
- **Extensions** - Integrations and widgets (e.g., Chatwoot, React widget).
|
||||
|
||||
|
||||
@@ -1,23 +0,0 @@
|
||||
# Build context is application/. Keep local state and caches out of the image.
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
.pytest_cache/
|
||||
.ruff_cache/
|
||||
.coverage
|
||||
htmlcov/
|
||||
*.log
|
||||
|
||||
# Runtime data: bind-mounted or created at run time, never baked in.
|
||||
indexes/
|
||||
inputs/
|
||||
vectors/
|
||||
*.faiss
|
||||
*.pkl
|
||||
|
||||
# Secrets and local config.
|
||||
.env
|
||||
.env.*
|
||||
|
||||
# Not needed inside the image.
|
||||
Dockerfile
|
||||
.dockerignore
|
||||
@@ -0,0 +1,57 @@
|
||||
"""``application`` is now ``docsgpt``; this alias keeps the old name importable for one release.
|
||||
|
||||
Every ``import application.x.y`` resolves to the already-imported ``docsgpt.x.y``
|
||||
module object, so there is exactly one settings object, one Celery app and one
|
||||
Flask app however a process refers to them. Entry points such as
|
||||
``celery -A application.app.celery`` and ``uvicorn application.asgi:asgi_app``
|
||||
keep working; update them to ``docsgpt.…`` before the alias is removed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import importlib.abc
|
||||
import importlib.util
|
||||
import sys
|
||||
import warnings
|
||||
|
||||
_OLD = __name__
|
||||
_NEW = "docsgpt"
|
||||
|
||||
|
||||
class _AliasLoader(importlib.abc.Loader):
|
||||
"""Hand back the ``docsgpt`` module instead of executing anything."""
|
||||
|
||||
def __init__(self, target: str) -> None:
|
||||
self._target = target
|
||||
|
||||
def create_module(self, spec):
|
||||
return importlib.import_module(self._target)
|
||||
|
||||
def exec_module(self, module) -> None:
|
||||
return None
|
||||
|
||||
|
||||
class _AliasFinder(importlib.abc.MetaPathFinder):
|
||||
"""Resolve ``application.<path>`` to ``docsgpt.<path>``."""
|
||||
|
||||
def find_spec(self, name, path=None, target=None):
|
||||
if name != _OLD and not name.startswith(_OLD + "."):
|
||||
return None
|
||||
new_name = _NEW + name[len(_OLD):]
|
||||
spec = importlib.util.find_spec(new_name)
|
||||
if spec is None:
|
||||
return None
|
||||
return importlib.util.spec_from_loader(
|
||||
name, _AliasLoader(new_name), is_package=spec.submodule_search_locations is not None
|
||||
)
|
||||
|
||||
|
||||
warnings.warn(
|
||||
"The 'application' package was renamed to 'docsgpt'. Update imports and entry points "
|
||||
"(celery -A docsgpt.app.celery, uvicorn docsgpt.asgi:asgi_app); this alias will be removed.",
|
||||
FutureWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
sys.meta_path.insert(0, _AliasFinder())
|
||||
sys.modules[_OLD] = importlib.import_module(_NEW)
|
||||
@@ -1,3 +0,0 @@
|
||||
from application.api.v1.routes import v1_bp
|
||||
|
||||
__all__ = ["v1_bp"]
|
||||
@@ -26,7 +26,8 @@ services:
|
||||
|
||||
backend:
|
||||
build:
|
||||
context: ../application
|
||||
context: ..
|
||||
dockerfile: docsgpt/Dockerfile
|
||||
args:
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
@@ -43,9 +44,9 @@ services:
|
||||
ports:
|
||||
- "7091:7091"
|
||||
volumes:
|
||||
- ../application/indexes:/app/application/indexes
|
||||
- ../application/inputs:/app/application/inputs
|
||||
- ../application/vectors:/app/application/vectors
|
||||
- ../docsgpt/indexes:/app/docsgpt/indexes
|
||||
- ../docsgpt/inputs:/app/docsgpt/inputs
|
||||
- ../docsgpt/vectors:/app/docsgpt/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
@@ -54,7 +55,8 @@ services:
|
||||
|
||||
worker:
|
||||
build:
|
||||
context: ../application
|
||||
context: ..
|
||||
dockerfile: docsgpt/Dockerfile
|
||||
args:
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
@@ -62,7 +64,7 @@ services:
|
||||
# must set INSTALL_TESSERACT=true before rebuilding (see docker-compose.yaml).
|
||||
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
|
||||
# `parsing` queue carries read_document/parse_document; required for its await to resolve.
|
||||
command: celery -A application.app.celery worker -l INFO -Q docsgpt,parsing,embeddings
|
||||
command: celery -A docsgpt.app.celery worker -l INFO -Q docsgpt,parsing,embeddings
|
||||
env_file:
|
||||
- ../.env
|
||||
environment:
|
||||
|
||||
@@ -44,9 +44,9 @@ services:
|
||||
ports:
|
||||
- "7091:7091"
|
||||
volumes:
|
||||
- ../application/indexes:/app/indexes
|
||||
- ../application/inputs:/app/inputs
|
||||
- ../application/vectors:/app/vectors
|
||||
- ../docsgpt/indexes:/app/indexes
|
||||
- ../docsgpt/inputs:/app/inputs
|
||||
- ../docsgpt/vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
@@ -58,7 +58,7 @@ services:
|
||||
user: root
|
||||
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-develop}${DOCSGPT_IMAGE_VARIANT:-}
|
||||
# `parsing` queue carries read_document/parse_document; required for its await to resolve.
|
||||
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
command: celery -A docsgpt.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
env_file:
|
||||
- ../.env
|
||||
environment:
|
||||
@@ -68,9 +68,9 @@ services:
|
||||
- CACHE_REDIS_URL=redis://redis:6379/2
|
||||
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
||||
volumes:
|
||||
- ../application/indexes:/app/indexes
|
||||
- ../application/inputs:/app/inputs
|
||||
- ../application/vectors:/app/vectors
|
||||
- ../docsgpt/indexes:/app/indexes
|
||||
- ../docsgpt/inputs:/app/inputs
|
||||
- ../docsgpt/vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
|
||||
@@ -82,7 +82,7 @@ services:
|
||||
user: root
|
||||
# Consumes the default queue plus `parsing` (read_document) and `embeddings`
|
||||
# (query embedding); without the latter every search times out.
|
||||
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
command: celery -A docsgpt.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
env_file:
|
||||
- path: .env
|
||||
required: false
|
||||
|
||||
@@ -32,7 +32,8 @@ services:
|
||||
backend:
|
||||
user: root
|
||||
build:
|
||||
context: ../application
|
||||
context: ..
|
||||
dockerfile: docsgpt/Dockerfile
|
||||
args:
|
||||
# Optional extras to bake in (comma-separated): docling, milvus. The
|
||||
# docling extra brings the layout-model parser/OCR backend and its
|
||||
@@ -57,9 +58,9 @@ services:
|
||||
ports:
|
||||
- "7091:7091"
|
||||
volumes:
|
||||
- ../application/indexes:/app/indexes
|
||||
- ../application/inputs:/app/inputs
|
||||
- ../application/vectors:/app/vectors
|
||||
- ../docsgpt/indexes:/app/indexes
|
||||
- ../docsgpt/inputs:/app/inputs
|
||||
- ../docsgpt/vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
@@ -69,7 +70,8 @@ services:
|
||||
worker:
|
||||
user: root
|
||||
build:
|
||||
context: ../application
|
||||
context: ..
|
||||
dockerfile: docsgpt/Dockerfile
|
||||
args:
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
@@ -80,7 +82,7 @@ services:
|
||||
# fails after EMBEDDINGS_DELEGATE_TIMEOUT, because EMBEDDINGS_DELEGATE_TO_WORKER
|
||||
# is on by default. For heavy/OCR parsing run a separate worker with `-Q parsing`;
|
||||
# to keep query latency off the ingest pool, another with `-Q embeddings`.
|
||||
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
command: celery -A docsgpt.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
env_file:
|
||||
- ../.env
|
||||
environment:
|
||||
@@ -91,9 +93,9 @@ services:
|
||||
- CACHE_REDIS_URL=redis://redis:6379/2
|
||||
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
||||
volumes:
|
||||
- ../application/indexes:/app/indexes
|
||||
- ../application/inputs:/app/inputs
|
||||
- ../application/vectors:/app/vectors
|
||||
- ../docsgpt/indexes:/app/indexes
|
||||
- ../docsgpt/inputs:/app/inputs
|
||||
- ../docsgpt/vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
|
||||
@@ -42,7 +42,7 @@ spec:
|
||||
name: docsgpt-secrets
|
||||
env:
|
||||
- name: FLASK_APP
|
||||
value: "application/app.py"
|
||||
value: "docsgpt/app.py"
|
||||
- name: DEPLOYMENT_TYPE
|
||||
value: "cloud"
|
||||
- name: POSTGRES_URI
|
||||
@@ -87,7 +87,7 @@ spec:
|
||||
image: arc53/docsgpt
|
||||
# `parsing` queue carries read_document/parse_document; required for its await to resolve.
|
||||
# For heavy/OCR parsing, run a separate deployment with `-Q parsing` (and GPU env).
|
||||
command: ["celery", "-A", "application.app.celery", "worker", "-l", "INFO", "-n", "worker.%h", "-Q", "docsgpt,parsing,embeddings"]
|
||||
command: ["celery", "-A", "docsgpt.app.celery", "worker", "-l", "INFO", "-n", "worker.%h", "-Q", "docsgpt,parsing,embeddings"]
|
||||
resources:
|
||||
limits:
|
||||
memory: "4Gi"
|
||||
|
||||
@@ -35,7 +35,7 @@ spec:
|
||||
name: docsgpt-secrets
|
||||
env:
|
||||
- name: FLASK_APP
|
||||
value: "application/app.py"
|
||||
value: "docsgpt/app.py"
|
||||
resources:
|
||||
limits:
|
||||
memory: "1Gi"
|
||||
|
||||
@@ -190,7 +190,7 @@ Document reading no longer runs in this sandbox. The `read_document` tool and th
|
||||
workflow native-file extract branch enqueue a `parse_document` Celery task that
|
||||
parses the document **in the backend** (the `DOC_PARSER_ENGINE` parser — anydoc
|
||||
by default; Docling only when the optional
|
||||
`application/requirements-docling.txt` extra is installed) and awaits the
|
||||
`docsgpt/requirements-docling.txt` extra is installed) and awaits the
|
||||
result. The task is routed to a
|
||||
dedicated **`parsing` queue** (`settings.DOCUMENT_PARSE_QUEUE`, default
|
||||
`"parsing"`) so a parse enqueued from inside a Celery worker (headless/scheduled
|
||||
@@ -199,7 +199,7 @@ agent) is served by a separate worker and never self-deadlocks the awaiting one.
|
||||
Run a dedicated parsing worker that consumes the `parsing` queue:
|
||||
|
||||
```bash
|
||||
celery -A application.app.celery worker -Q parsing -l INFO
|
||||
celery -A docsgpt.app.celery worker -Q parsing -l INFO
|
||||
```
|
||||
|
||||
It takes its own env, so parse-heavy work runs on a separate, optionally larger
|
||||
@@ -214,7 +214,7 @@ and leaves this worker light.
|
||||
worker must also consume `parsing`, or the tool's await never resolves:
|
||||
|
||||
```bash
|
||||
celery -A application.app.celery worker -Q docsgpt,parsing,embeddings -l INFO
|
||||
celery -A docsgpt.app.celery worker -Q docsgpt,parsing,embeddings -l INFO
|
||||
```
|
||||
|
||||
Tuning settings: `DOCUMENT_PARSE_TIMEOUT` (seconds the tool awaits before
|
||||
|
||||
@@ -44,7 +44,7 @@ The main set of instructions or system [prompt](/Guides/Customising-prompts) tha
|
||||
|
||||
## Understanding Agent Types
|
||||
|
||||
DocsGPT supports several agent types, each with a distinct way of processing information. The code for these can be found in the `application/agents/` directory.
|
||||
DocsGPT supports several agent types, each with a distinct way of processing information. The code for these can be found in the `docsgpt/agents/` directory.
|
||||
|
||||
### 1. Classic Agent
|
||||
|
||||
@@ -117,8 +117,8 @@ Once an agent is created, you can:
|
||||
|
||||
You can bootstrap a fresh DocsGPT deployment with a curated set of agents by seeding them directly into the user-data store (Postgres).
|
||||
|
||||
1. **Customize the configuration** – edit `application/seed/config/premade_agents.yaml` (or copy from `application/seed/config/agents_template.yaml`) to describe the agents you want to provision. Each entry lets you define prompts, tools, and optional data sources.
|
||||
1. **Customize the configuration** – edit `docsgpt/seed/config/premade_agents.yaml` (or copy from `docsgpt/seed/config/agents_template.yaml`) to describe the agents you want to provision. Each entry lets you define prompts, tools, and optional data sources.
|
||||
2. **Ensure dependencies are running** – Postgres must be reachable using `POSTGRES_URI` from `.env` (schema applied via `python scripts/db/init_postgres.py`), and a Celery worker should be available if any agent sources need to be ingested via `ingest_remote`.
|
||||
3. **Execute the seeder** – run `python -m application.seed.commands init`. Add `--force` when you need to reseed an existing environment.
|
||||
3. **Execute the seeder** – run `python -m docsgpt.seed.commands init`. Add `--force` when you need to reseed an existing environment.
|
||||
|
||||
The seeder keeps templates under the `system` user so they appear in the UI for anyone to clone or customize. Environment variable placeholders such as `${MY_TOKEN}` inside tool configs are resolved during the seeding process.
|
||||
@@ -70,7 +70,7 @@ GET /api/messages/<message_id>/events
|
||||
This replays the message's events past your last-seen sequence number and tails the rest live. It is backed by the Postgres `message_events` journal (retained for `MESSAGE_EVENTS_RETENTION_DAYS`, default 14).
|
||||
|
||||
<Callout type="warning" emoji="⚠️">
|
||||
The chat reconnect endpoint is a native-async route served by the ASGI entrypoint. Under a plain `flask run` dev server it returns `404`; run the backend via the ASGI app (`uvicorn application.asgi:asgi_app`) or the production gunicorn uvicorn worker to use it. See the [Development Environment](/Deploying/Development-Environment) guide.
|
||||
The chat reconnect endpoint is a native-async route served by the ASGI entrypoint. Under a plain `flask run` dev server it returns `404`; run the backend via the ASGI app (`uvicorn docsgpt.asgi:asgi_app`) or the production gunicorn uvicorn worker to use it. See the [Development Environment](/Deploying/Development-Environment) guide.
|
||||
</Callout>
|
||||
|
||||
## Settings
|
||||
|
||||
@@ -53,7 +53,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
|
||||
* **Option 1: Using a `.env` file (Recommended):**
|
||||
* If you haven't already, create a file named `.env` in the **root directory** of your DocsGPT project.
|
||||
* Modify the `.env` file to adjust settings as needed. You can find a comprehensive list of configurable options in [`application/core/settings.py`](https://github.com/arc53/DocsGPT/blob/main/application/core/settings.py).
|
||||
* Modify the `.env` file to adjust settings as needed. You can find a comprehensive list of configurable options in [`docsgpt/core/settings.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/core/settings.py).
|
||||
|
||||
* **Option 2: Exporting Environment Variables:**
|
||||
* Alternatively, you can export environment variables directly in your terminal. However, using a `.env` file is generally more organized for development.
|
||||
@@ -83,7 +83,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
For an offline or air-gapped machine, fetch it ahead of time instead:
|
||||
|
||||
```bash
|
||||
python -m application.scripts.prefetch_models
|
||||
python -m docsgpt.scripts.prefetch_models
|
||||
```
|
||||
|
||||
4. **Install Backend Dependencies:**
|
||||
@@ -91,7 +91,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
Navigate to the root of your DocsGPT repository and install the required Python packages:
|
||||
|
||||
```bash
|
||||
pip install -r application/requirements.txt
|
||||
pip install -r docsgpt/requirements.txt
|
||||
```
|
||||
|
||||
Dependencies are declared in `pyproject.toml` and locked in `uv.lock`; the
|
||||
@@ -103,8 +103,8 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
feature (each file is the core set plus the extra):
|
||||
|
||||
```bash
|
||||
pip install -r application/requirements-docling.txt # docling parser engine: OCR backend, read_document structured output
|
||||
pip install -r application/requirements-milvus.txt # VECTOR_STORE=milvus
|
||||
pip install -r docsgpt/requirements-docling.txt # docling parser engine: OCR backend, read_document structured output
|
||||
pip install -r docsgpt/requirements-milvus.txt # VECTOR_STORE=milvus
|
||||
# or with uv: uv sync --extra docling --extra milvus
|
||||
```
|
||||
|
||||
@@ -122,25 +122,25 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
For local development, run the ASGI composition under uvicorn. It serves the **whole** application, hot-reloads on source changes, and matches the production runtime:
|
||||
|
||||
```bash
|
||||
uvicorn application.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
|
||||
uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
|
||||
```
|
||||
|
||||
This makes the backend accessible on `http://localhost:7091`. Production uses `gunicorn -k uvicorn_worker.UvicornWorker` against the same `application.asgi:asgi_app` target.
|
||||
This makes the backend accessible on `http://localhost:7091`. Production uses `gunicorn -k uvicorn_worker.UvicornWorker` against the same `docsgpt.asgi:asgi_app` target.
|
||||
|
||||
A plain Flask run is a faster inner loop (quick startup, the Werkzeug interactive debugger):
|
||||
|
||||
```bash
|
||||
flask --app application/app.py run --host=0.0.0.0 --port=7091
|
||||
flask --app docsgpt/app.py run --host=0.0.0.0 --port=7091
|
||||
```
|
||||
|
||||
But it serves **only** the WSGI Flask app and omits the routes mounted on the ASGI shell in `application/asgi.py`: the `/mcp` FastMCP endpoint and the native-async SSE reconnect reader `GET /api/messages/<id>/events`. Under `flask run` those paths return 404 — chat still works (`POST /stream` is a Flask route), but a stream interrupted by a disconnect won't auto-resume on reconnect. Use `flask run` only when you don't need those routes.
|
||||
But it serves **only** the WSGI Flask app and omits the routes mounted on the ASGI shell in `docsgpt/asgi.py`: the `/mcp` FastMCP endpoint and the native-async SSE reconnect reader `GET /api/messages/<id>/events`. Under `flask run` those paths return 404 — chat still works (`POST /stream` is a Flask route), but a stream interrupted by a disconnect won't auto-resume on reconnect. Use `flask run` only when you don't need those routes.
|
||||
|
||||
6. **Start the Celery Worker:**
|
||||
|
||||
Open a new terminal window (and activate your virtual environment if you used one). Start the Celery worker to handle background tasks:
|
||||
|
||||
```bash
|
||||
celery -A application.app.celery worker -l INFO
|
||||
celery -A docsgpt.app.celery worker -l INFO
|
||||
```
|
||||
|
||||
This command will start the Celery worker, which processes tasks such as document parsing and vector embedding.
|
||||
@@ -148,7 +148,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
**macOS note:** Due to a threading issue, start Celery with the solo pool:
|
||||
|
||||
```bash
|
||||
python -m celery -A application.app.celery worker -l INFO --pool=solo
|
||||
python -m celery -A docsgpt.app.celery worker -l INFO --pool=solo
|
||||
```
|
||||
|
||||
**Running in Debugger (VSCode):**
|
||||
|
||||
@@ -69,8 +69,8 @@ checkout.
|
||||
## Using the Source Checkout
|
||||
|
||||
With a clone of the repository, `deployment/docker-compose-hub.yaml` runs the
|
||||
same pre-built images while keeping your data in `application/indexes`,
|
||||
`application/inputs` and `application/vectors`, and `deployment/docker-compose.yaml`
|
||||
same pre-built images while keeping your data in `docsgpt/indexes`,
|
||||
`docsgpt/inputs` and `docsgpt/vectors`, and `deployment/docker-compose.yaml`
|
||||
builds the images from your working tree (for local changes, or a build with
|
||||
extra packages: `EXTRAS=docling` in `.env`).
|
||||
|
||||
|
||||
@@ -29,11 +29,11 @@ LLM_NAME=gpt-4o
|
||||
|
||||
### 2. Configuration via `settings.py` file (Advanced)
|
||||
|
||||
For more advanced configurations or if you prefer to manage settings directly in code, you can modify the `settings.py` file. This file is located in the `application/core` directory of your DocsGPT project.
|
||||
For more advanced configurations or if you prefer to manage settings directly in code, you can modify the `settings.py` file. This file is located in the `docsgpt/core` directory of your DocsGPT project.
|
||||
|
||||
While modifying `settings.py` offers more flexibility, it's generally recommended to use the `.env` file for basic settings and reserve `settings.py` for more complex adjustments or when you need to configure settings programmatically.
|
||||
|
||||
**Location of `settings.py`:** `application/core/settings.py`
|
||||
**Location of `settings.py`:** `docsgpt/core/settings.py`
|
||||
|
||||
## Basic Settings Explained
|
||||
|
||||
@@ -61,7 +61,7 @@ Here are some of the most fundamental settings you'll likely want to configure:
|
||||
|
||||
- **Default value:** leave it unset and DocsGPT picks for you at first boot, recording the choice so it never changes underneath you: a fresh install is pinned to `ibm-granite/granite-embedding-311m-multilingual-r2` (multilingual, 32k context, same 768 dimensions), and an install that already has sources is pinned to `huggingface_sentence-transformers/all-mpnet-base-v2` so its index stays readable. Setting it here overrides that pin.
|
||||
- **Other options:** Any FastEmbed built-in model, or any Hugging Face repository shipping an ONNX export. See [Embeddings](/Models/embeddings).
|
||||
- **Changing it on an existing index requires re-embedding** — same-width models swap without any error and silently degrade retrieval. Run `python -m application.scripts.reembed`.
|
||||
- **Changing it on an existing index requires re-embedding** — same-width models swap without any error and silently degrade retrieval. Run `python -m docsgpt.scripts.reembed`.
|
||||
|
||||
- **`API_KEY`**: Required for most cloud-based LLM providers. This is your authentication key to access the LLM provider's API. You'll need to obtain this key from your chosen provider's platform.
|
||||
|
||||
@@ -133,7 +133,7 @@ models:
|
||||
|
||||
After restart, those models appear in `/api/models` and are selectable
|
||||
in the UI. A working template lives at
|
||||
`application/core/models/examples/mistral.yaml.example`.
|
||||
`docsgpt/core/models/examples/mistral.yaml.example`.
|
||||
|
||||
**What you can do:**
|
||||
|
||||
@@ -148,8 +148,8 @@ in the UI. A working template lives at
|
||||
|
||||
**What you cannot do via `MODELS_CONFIG_DIR`:** add a brand-new
|
||||
non-OpenAI provider. That requires a Python plugin under
|
||||
`application/llm/providers/`. See
|
||||
`application/core/models/README.md` for the full schema reference.
|
||||
`docsgpt/llm/providers/`. See
|
||||
`docsgpt/core/models/README.md` for the full schema reference.
|
||||
|
||||
### Docker
|
||||
|
||||
@@ -207,7 +207,7 @@ for the engines and flows.
|
||||
|
||||
| Setting | Default | Description |
|
||||
| --- | --- | --- |
|
||||
| `DOC_PARSER_ENGINE` | `anydoc` | Parser engine: `anydoc` (fast Rust converter, no ML models) or `docling` (layout/table models, structured output). Files anydoc cannot read fall back to Docling when installed. Docling is an optional extra: `pip install -r application/requirements-docling.txt`, or Docker builds with `--build-arg INSTALL_DOCLING=true`. |
|
||||
| `DOC_PARSER_ENGINE` | `anydoc` | Parser engine: `anydoc` (fast Rust converter, no ML models) or `docling` (layout/table models, structured output). Files anydoc cannot read fall back to Docling when installed. Docling is an optional extra: `pip install -r docsgpt/requirements-docling.txt`, or Docker builds with `--build-arg INSTALL_DOCLING=true`. |
|
||||
| `OCR_ENABLED` | `false` | OCR for source ingestion. Alias: `DOCLING_OCR_ENABLED`. |
|
||||
| `OCR_ATTACHMENTS_ENABLED` | `false` | OCR for chat attachments. Alias: `DOCLING_OCR_ATTACHMENTS_ENABLED`. |
|
||||
| `OCR_BACKEND` | `auto` | Who performs OCR: `auto` (Docling when installed, else native), `native` (pypdfium2 + Pillow rendering into tesseract or DeepSeek-OCR; no docling needed), or `docling` (layout-model hybrid OCR). See the [OCR guide](/Guides/ocr#ocr-backends). |
|
||||
@@ -469,7 +469,7 @@ See [Embeddings](/Models/embeddings) for full guidance.
|
||||
|
||||
| Setting | Default | Description |
|
||||
| --- | --- | --- |
|
||||
| `EMBEDDINGS_NAME` | `huggingface_sentence-transformers/all-mpnet-base-v2` | The embedding model. New installs use `ibm-granite/granite-embedding-311m-multilingual-r2`. Changing it on a populated index requires `application.scripts.reembed`. |
|
||||
| `EMBEDDINGS_NAME` | `huggingface_sentence-transformers/all-mpnet-base-v2` | The embedding model. New installs use `ibm-granite/granite-embedding-311m-multilingual-r2`. Changing it on a populated index requires `docsgpt.scripts.reembed`. |
|
||||
| `EMBEDDINGS_BASE_URL` | unset | Base URL of a remote OpenAI-compatible embeddings server. Setting it routes all embedding calls there. |
|
||||
| `EMBEDDINGS_THREADS` | unset (Docker image: `4`) | Threads one local FastEmbed/onnxruntime session may use. onnxruntime otherwise sizes its pool to the host's core count, which a CPU-limited container still reports, so the image pins it like `OMP_NUM_THREADS`. Raise it on a large dedicated worker. |
|
||||
| `EMBEDDINGS_KEY` | unset | Optional bearer token for the remote embeddings server. |
|
||||
@@ -550,4 +550,4 @@ These are just the basic settings to get you started. The `settings.py` file con
|
||||
- Cache settings (`CACHE_REDIS_URL`)
|
||||
- And many more!
|
||||
|
||||
For a complete list of available settings and their descriptions, refer to the `settings.py` file in `application/core`. Remember to restart your Docker containers after making changes to your `.env` file or `settings.py` for the changes to take effect.
|
||||
For a complete list of available settings and their descriptions, refer to the `settings.py` file in `docsgpt/core`. Remember to restart your Docker containers after making changes to your `.env` file or `settings.py` for the changes to take effect.
|
||||
@@ -8,7 +8,7 @@ import { Callout } from 'nextra/components'
|
||||
# Observability
|
||||
|
||||
DocsGPT bundles the OpenTelemetry SDK and auto-instrumentation packages
|
||||
in `application/requirements.txt` — they install with the rest of the
|
||||
in `docsgpt/requirements.txt` — they install with the rest of the
|
||||
backend deps. Telemetry is **off by default**; opt in by prefixing the
|
||||
launch command with `opentelemetry-instrument` and setting OTLP env
|
||||
vars.
|
||||
@@ -42,12 +42,12 @@ services:
|
||||
backend:
|
||||
command: >
|
||||
opentelemetry-instrument gunicorn -w 1 -k uvicorn_worker.UvicornWorker
|
||||
--bind 0.0.0.0:7091 --config application/gunicorn_conf.py
|
||||
application.asgi:asgi_app
|
||||
--bind 0.0.0.0:7091 --config docsgpt/gunicorn_conf.py
|
||||
docsgpt.asgi:asgi_app
|
||||
environment:
|
||||
- OTEL_SERVICE_NAME=docsgpt-backend
|
||||
worker:
|
||||
command: opentelemetry-instrument celery -A application.app.celery worker -l INFO -B
|
||||
command: opentelemetry-instrument celery -A docsgpt.app.celery worker -l INFO -B
|
||||
environment:
|
||||
- OTEL_SERVICE_NAME=docsgpt-celery-worker
|
||||
```
|
||||
@@ -56,14 +56,14 @@ For local dev, prepend `dotenv run --` so the `OTEL_*` vars from `.env`
|
||||
reach `opentelemetry-instrument` before it boots the SDK:
|
||||
|
||||
```bash
|
||||
dotenv run -- opentelemetry-instrument flask --app application/app.py run --port=7091
|
||||
dotenv run -- opentelemetry-instrument celery -A application.app.celery worker -l INFO --pool=solo
|
||||
dotenv run -- opentelemetry-instrument flask --app docsgpt/app.py run --port=7091
|
||||
dotenv run -- opentelemetry-instrument celery -A docsgpt.app.celery worker -l INFO --pool=solo
|
||||
```
|
||||
|
||||
|
||||
<Callout type="info" emoji="ℹ️">
|
||||
Logs are exported in-process when `OTEL_LOGS_EXPORTER=otlp` is set —
|
||||
`application/core/logging_config.py` detects the flag and preserves
|
||||
`docsgpt/core/logging_config.py` detects the flag and preserves
|
||||
the OTEL log handler. Without it, `logging` writes only to stdout.
|
||||
</Callout>
|
||||
|
||||
|
||||
@@ -37,7 +37,7 @@ schema on first boot.
|
||||
|
||||
```bash
|
||||
export POSTGRES_URI="postgresql://user:pass@host/docsgpt?sslmode=require"
|
||||
uvicorn application.asgi:asgi_app --host 0.0.0.0 --port 7091
|
||||
uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091
|
||||
```
|
||||
|
||||
### Bare-metal Postgres
|
||||
@@ -47,7 +47,7 @@ First boot creates both the database and the schema.
|
||||
|
||||
```bash
|
||||
export POSTGRES_URI="postgresql://postgres@localhost/docsgpt"
|
||||
uvicorn application.asgi:asgi_app --host 0.0.0.0 --port 7091
|
||||
uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091
|
||||
```
|
||||
|
||||
Prefer a dedicated non-superuser role? Create it once as superuser — the
|
||||
@@ -93,7 +93,7 @@ init-container ahead of the app rollout:
|
||||
```bash
|
||||
python scripts/db/init_postgres.py
|
||||
# equivalently:
|
||||
alembic -c application/alembic.ini upgrade head
|
||||
alembic -c docsgpt/alembic.ini upgrade head
|
||||
```
|
||||
|
||||
The reasoning: the app's runtime role shouldn't carry DDL privileges,
|
||||
@@ -112,7 +112,7 @@ One-shot, offline, app stopped. The app itself will create the
|
||||
Postgres schema when it boots — you only need to run the data copy.
|
||||
|
||||
```bash
|
||||
pip install -r application/requirements.txt
|
||||
pip install -r docsgpt/requirements.txt
|
||||
pip install 'pymongo>=4.6'
|
||||
|
||||
export POSTGRES_URI="postgresql://docsgpt:docsgpt@localhost:5432/docsgpt"
|
||||
|
||||
@@ -359,7 +359,7 @@ Technical documentation about...
|
||||
### Template Validation
|
||||
Test your template syntax before saving:
|
||||
```python
|
||||
from application.api.answer.services.prompt_renderer import PromptRenderer
|
||||
from docsgpt.api.answer.services.prompt_renderer import PromptRenderer
|
||||
|
||||
renderer = PromptRenderer()
|
||||
is_valid = renderer.validate_template("Your prompt with {{ variables }}")
|
||||
@@ -487,7 +487,7 @@ Provide detailed answers appropriate for {{ passthrough.access_level }} access l
|
||||
|
||||
### Render Prompt via API
|
||||
```python
|
||||
from application.api.answer.services.prompt_renderer import PromptRenderer
|
||||
from docsgpt.api.answer.services.prompt_renderer import PromptRenderer
|
||||
|
||||
renderer = PromptRenderer()
|
||||
rendered = renderer.render_prompt(
|
||||
|
||||
@@ -19,7 +19,7 @@ The compression system operates on a "summarize and truncate" principle:
|
||||
|
||||
## Configuration
|
||||
|
||||
You can configure the compression behavior in your `.env` file or `application/core/settings.py`:
|
||||
You can configure the compression behavior in your `.env` file or `docsgpt/core/settings.py`:
|
||||
|
||||
| Setting | Default | Description |
|
||||
| :--- | :--- | :--- |
|
||||
|
||||
@@ -57,7 +57,7 @@ apt-get install tesseract-ocr tesseract-ocr-eng brew install tesseract
|
||||
Docker images build without it by default; opt in with the build argument:
|
||||
|
||||
```bash
|
||||
docker build --build-arg INSTALL_TESSERACT=true ./application
|
||||
docker build -f docsgpt/Dockerfile --build-arg INSTALL_TESSERACT=true .
|
||||
```
|
||||
|
||||
`deployment/docker-compose.yaml` forwards the same switch, so setting
|
||||
@@ -94,7 +94,7 @@ docling is not part of the base install, and OCR does not need it (see
|
||||
output:
|
||||
|
||||
```bash
|
||||
pip install -r application/requirements-docling.txt # or: uv sync --extra docling
|
||||
pip install -r docsgpt/requirements-docling.txt # or: uv sync --extra docling
|
||||
```
|
||||
|
||||
That file is the core set plus the `docling` extra, exported from the same
|
||||
@@ -108,7 +108,7 @@ layout, table-structure and RapidOCR models in so the first parse does not
|
||||
download them. Local builds opt in with the build argument:
|
||||
|
||||
```bash
|
||||
docker build --build-arg EXTRAS=docling ./application
|
||||
docker build -f docsgpt/Dockerfile --build-arg EXTRAS=docling .
|
||||
```
|
||||
|
||||
`deployment/docker-compose.yaml` forwards the same switch, so setting
|
||||
|
||||
@@ -33,7 +33,7 @@ DocsGPT offers direct, streamlined support for the following cloud LLM providers
|
||||
| Novita AI | `novita` | (See Novita docs) |
|
||||
| HuggingFace Inference API | `huggingface` | `meta-llama/Llama-3.1-8B-Instruct` |
|
||||
|
||||
DocsGPT also ships a **model catalog** (`application/core/models/*.yaml`) that the in-app model picker reads, so common models from these providers — including DeepSeek — appear ready to select once the matching API key is set.
|
||||
DocsGPT also ships a **model catalog** (`docsgpt/core/models/*.yaml`) that the in-app model picker reads, so common models from these providers — including DeepSeek — appear ready to select once the matching API key is set.
|
||||
|
||||
## Connecting to OpenAI-Compatible Cloud APIs
|
||||
|
||||
@@ -75,4 +75,4 @@ See [App Configuration](/Deploying/DocsGPT-Settings) for the full settings refer
|
||||
|
||||
## Adding Support for Other Cloud Providers
|
||||
|
||||
If you wish to connect to a cloud provider that is not explicitly listed above or doesn't offer OpenAI API compatibility, you can extend DocsGPT to support it. Within the DocsGPT repository, navigate to the `application/llm` directory. Here, you will find Python files defining the existing LLM integrations. You can use these files as examples to create a new module for your desired cloud provider. After creating your new LLM module, you will need to register it within the `llm_creator.py` file. This process involves some coding, but it allows for virtually unlimited extensibility to connect to any cloud-based LLM service with an accessible API.
|
||||
If you wish to connect to a cloud provider that is not explicitly listed above or doesn't offer OpenAI API compatibility, you can extend DocsGPT to support it. Within the DocsGPT repository, navigate to the `docsgpt/llm` directory. Here, you will find Python files defining the existing LLM integrations. You can use these files as examples to create a new module for your desired cloud provider. After creating your new LLM module, you will need to register it within the `llm_creator.py` file. This process involves some coding, but it allows for virtually unlimited extensibility to connect to any cloud-based LLM service with an accessible API.
|
||||
@@ -46,7 +46,7 @@ Models with a Dense projection layer (for example `sentence-transformers/LaBSE`)
|
||||
For an offline or air-gapped install, pre-fetch the model at build or setup time:
|
||||
|
||||
```bash
|
||||
python -m application.scripts.prefetch_models
|
||||
python -m docsgpt.scripts.prefetch_models
|
||||
```
|
||||
|
||||
## Using OpenAI Embeddings
|
||||
@@ -100,7 +100,7 @@ Retrieval then depends on a worker consuming `EMBEDDINGS_QUEUE` (`embeddings` by
|
||||
Sharing one worker also shares its concurrency with ingest, so a query can queue behind a long parse. Run a dedicated worker to isolate query latency:
|
||||
|
||||
```bash
|
||||
celery -A application.app.celery worker -Q embeddings
|
||||
celery -A docsgpt.app.celery worker -Q embeddings
|
||||
```
|
||||
|
||||
Set `EMBEDDINGS_DELEGATE_TO_WORKER=false` if you run the API without a worker; it will load the model in-process instead.
|
||||
@@ -139,7 +139,7 @@ The dimension check is a guard against a corrupt index, not a guarantee that a s
|
||||
Switching between same-width models therefore still requires re-embedding:
|
||||
|
||||
```bash
|
||||
python -m application.scripts.reembed
|
||||
python -m docsgpt.scripts.reembed
|
||||
```
|
||||
|
||||
Run it after changing `EMBEDDINGS_NAME` and before serving queries. See [Upgrading](/upgrading) for the granite migration specifically.
|
||||
@@ -150,6 +150,6 @@ With `GRAPHRAG_ENABLED`, the script also rewrites `graph_nodes.name_embedding` o
|
||||
|
||||
## Adding Support for Other Embedding Models
|
||||
|
||||
To teach DocsGPT about a new model — so it carries a known pooling, width and context window rather than being inferred — add an `EmbeddingModel` entry to `MODELS` in `application/vectorstore/model_registry.py`. That registry is the single source of truth the local runner, the remote client, the schema bootstrap and the chunker all read.
|
||||
To teach DocsGPT about a new model — so it carries a known pooling, width and context window rather than being inferred — add an `EmbeddingModel` entry to `MODELS` in `docsgpt/vectorstore/model_registry.py`. That registry is the single source of truth the local runner, the remote client, the schema bootstrap and the chunker all read.
|
||||
|
||||
Specifically, pay attention to the `EmbeddingsWrapper` and `EmbeddingsSingleton` classes. `EmbeddingsWrapper` provides a way to wrap different embedding model libraries into a consistent interface for DocsGPT. `EmbeddingsSingleton` manages the instantiation and retrieval of embedding model instances. By understanding these classes and the existing embedding model implementations, you can create your own custom integration for virtually any embedding model library you desire.
|
||||
@@ -86,4 +86,4 @@ You can configure multiple API keys simultaneously (e.g., both `OPENAI_API_KEY`
|
||||
|
||||
## Adding Support for Other Local Engines
|
||||
|
||||
While DocsGPT currently focuses on OpenAI API compatible local engines, you can extend its capabilities to support other local inference solutions. To do this, navigate to the `application/llm` directory in the DocsGPT repository. Examine the existing Python files for examples of LLM integrations. You can create a new module for your desired local engine, and then register it in the `llm_creator.py` file within the same directory. This allows for custom integration with a wide range of local LLM servers beyond those listed above.
|
||||
While DocsGPT currently focuses on OpenAI API compatible local engines, you can extend its capabilities to support other local inference solutions. To do this, navigate to the `docsgpt/llm` directory in the DocsGPT repository. Examine the existing Python files for examples of LLM integrations. You can create a new module for your desired local engine, and then register it in the `llm_creator.py` file within the same directory. This allows for custom integration with a wide range of local LLM servers beyond those listed above.
|
||||
@@ -38,37 +38,37 @@ DocsGPT includes a suite of pre-built tools designed to expand its capabilities
|
||||
},
|
||||
{
|
||||
title: 'Brave Search',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/brave.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/brave.py',
|
||||
description: 'Enables DocsGPT to perform real-time web and image searches using the Brave Search API. Requires an API key.'
|
||||
},
|
||||
{
|
||||
title: 'DuckDuckGo Search',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/duckduckgo.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/duckduckgo.py',
|
||||
description: 'Performs web and image searches using DuckDuckGo. No API key required.'
|
||||
},
|
||||
{
|
||||
title: 'CryptoPrice',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/cryptoprice.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/cryptoprice.py',
|
||||
description: 'Fetches the current price of specified cryptocurrencies using the CryptoCompare public API.'
|
||||
},
|
||||
{
|
||||
title: 'Ntfy',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/ntfy.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/ntfy.py',
|
||||
description: 'Allows DocsGPT to send push notifications to ntfy topics on a specified server, ideal for alerts and updates.'
|
||||
},
|
||||
{
|
||||
title: 'Telegram Bot',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/telegram.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/telegram.py',
|
||||
description: 'Allows DocsGPT to send messages or images to Telegram chats via a Telegram Bot. Requires a bot token and chat ID.'
|
||||
},
|
||||
{
|
||||
title: 'PostgreSQL Database',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/postgres.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/postgres.py',
|
||||
description: 'Connects to a PostgreSQL database to execute SQL queries and retrieve schema information.'
|
||||
},
|
||||
{
|
||||
title: 'Read Webpage (browser)',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/read_webpage.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/read_webpage.py',
|
||||
description: 'Fetches the HTML content of a URL and converts it to Markdown for the agent to read.'
|
||||
},
|
||||
{
|
||||
@@ -83,17 +83,17 @@ DocsGPT includes a suite of pre-built tools designed to expand its capabilities
|
||||
},
|
||||
{
|
||||
title: 'Memory',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/memory.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/memory.py',
|
||||
description: 'Stores and retrieves information across conversations through a per-user memory file directory.'
|
||||
},
|
||||
{
|
||||
title: 'Notepad',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/notes.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/notes.py',
|
||||
description: 'A single editable note. Supports viewing, overwriting, and string replacement.'
|
||||
},
|
||||
{
|
||||
title: 'Todo List',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/application/agents/tools/todo_list.py',
|
||||
link: 'https://github.com/arc53/DocsGPT/blob/main/docsgpt/agents/tools/todo_list.py',
|
||||
description: 'Manages todo items — creating, viewing, updating, and deleting todos.'
|
||||
}
|
||||
]}
|
||||
|
||||
@@ -27,7 +27,7 @@ While DocsGPT offers a range of built-in tools and a versatile API Tool, there a
|
||||
Before you begin, ensure you have:
|
||||
|
||||
* A solid understanding of Python programming.
|
||||
* Familiarity with the DocsGPT project structure, particularly the `application/agents/tools/` directory where custom tools reside.
|
||||
* Familiarity with the DocsGPT project structure, particularly the `docsgpt/agents/tools/` directory where custom tools reside.
|
||||
* Basic knowledge of how APIs work, as many tools involve interacting with external or internal APIs.
|
||||
* Your DocsGPT development environment set up. If not, please refer to the [Setting Up a Development Environment](/Deploying/Development-Environment) guide.
|
||||
|
||||
@@ -35,7 +35,7 @@ Before you begin, ensure you have:
|
||||
|
||||
Custom tools in DocsGPT are Python classes that inherit from a base `Tool` class and implement specific methods to define their behavior, capabilities, and configuration needs.
|
||||
|
||||
The **foundation** for all custom tools is the abstract base class, located in `application/agents/tools/base.py`. Your custom tool class **must** inherit from this class.
|
||||
The **foundation** for all custom tools is the abstract base class, located in `docsgpt/agents/tools/base.py`. Your custom tool class **must** inherit from this class.
|
||||
|
||||
### Essential Methods to Implement
|
||||
|
||||
@@ -150,11 +150,11 @@ Your custom tool class needs to implement the following methods:
|
||||
|
||||
## Tool Registration and Discovery
|
||||
|
||||
DocsGPT's ToolManager (located in application/agents/tools/tool_manager.py) automatically discovers and loads tools.
|
||||
DocsGPT's ToolManager (located in docsgpt/agents/tools/tool_manager.py) automatically discovers and loads tools.
|
||||
|
||||
As long as your custom tool:
|
||||
|
||||
1. Is placed in a Python file within the `application/agents/tools/` directory (and the filename is not `base.py` or starts with `__`).
|
||||
1. Is placed in a Python file within the `docsgpt/agents/tools/` directory (and the filename is not `base.py` or starts with `__`).
|
||||
2. Correctly inherits from the `Tool` base class.
|
||||
3. Implements all the abstract methods (`execute_action`, `get_actions_metadata`, `get_config_requirements`).
|
||||
|
||||
|
||||
@@ -19,8 +19,8 @@ DocsGPT now runs embeddings through [FastEmbed](https://github.com/qdrant/fastem
|
||||
**Your worker command does need one change.** Query embedding now runs on the Celery worker (`EMBEDDINGS_DELEGATE_TO_WORKER`, on by default), which keeps the API from loading a model of its own. If you start your worker with an explicit `-Q`, add the `embeddings` queue:
|
||||
|
||||
```diff
|
||||
- celery -A application.app.celery worker -l INFO -Q docsgpt,parsing
|
||||
+ celery -A application.app.celery worker -l INFO -Q docsgpt,parsing,embeddings
|
||||
- celery -A docsgpt.app.celery worker -l INFO -Q docsgpt,parsing
|
||||
+ celery -A docsgpt.app.celery worker -l INFO -Q docsgpt,parsing,embeddings
|
||||
```
|
||||
|
||||
The bundled Compose and Kubernetes manifests already do this — pull them along with the code. Without it, every search blocks for `EMBEDDINGS_DELEGATE_TIMEOUT` (60s) and then answers with no retrieved context rather than raising, so the symptom is bad answers, not an error. To keep the model out of the worker too, set `EMBEDDINGS_BASE_URL`; to run the API on its own, set `EMBEDDINGS_DELEGATE_TO_WORKER=false`.
|
||||
@@ -41,8 +41,8 @@ Set the model, then rebuild the vectors:
|
||||
EMBEDDINGS_NAME=ibm-granite/granite-embedding-311m-multilingual-r2
|
||||
|
||||
# 2. Rebuild the vectors from the chunk text already in your index
|
||||
docker compose exec backend python -m application.scripts.reembed --dry-run
|
||||
docker compose exec backend python -m application.scripts.reembed
|
||||
docker compose exec backend python -m docsgpt.scripts.reembed --dry-run
|
||||
docker compose exec backend python -m docsgpt.scripts.reembed
|
||||
```
|
||||
|
||||
Re-embedding reads the chunk text already stored in your index. It does not re-download, re-parse or re-chunk your documents, so no source files are needed and the run is proportional to index size, not corpus size. Both `pgvector` and `faiss` are supported.
|
||||
@@ -93,7 +93,7 @@ need attention:
|
||||
## Check your version
|
||||
|
||||
```bash
|
||||
docker compose exec backend python -c "from application.version import get_version; print(get_version())"
|
||||
docker compose exec backend python -c "from docsgpt.version import get_version; print(get_version())"
|
||||
```
|
||||
|
||||
Release notes: [changelog](/changelog). Tags: [GitHub releases](https://github.com/arc53/DocsGPT/releases).
|
||||
@@ -135,7 +135,7 @@ Full manifests: [Kubernetes deployment guide](/Deploying/Kubernetes-Deploying).
|
||||
Alembic migrations run on worker startup. To apply manually:
|
||||
|
||||
```bash
|
||||
docker compose exec backend alembic -c application/alembic.ini upgrade head
|
||||
docker compose exec backend alembic -c docsgpt/alembic.ini upgrade head
|
||||
```
|
||||
|
||||
`upgrade head` is idempotent.
|
||||
|
||||
@@ -126,7 +126,7 @@ reconnect HTTP errored. Common cases:
|
||||
- The user's JWT rotated mid-stream → 401 on the GET. Frontend
|
||||
doesn't auto-refresh; the user reloads.
|
||||
- The user is on a different host than the API and CORS is rejecting
|
||||
the GET → check `application/asgi.py` allow-headers.
|
||||
the GET → check `docsgpt/asgi.py` allow-headers.
|
||||
|
||||
### D. "The dev install never delivers any notifications at all"
|
||||
|
||||
@@ -316,7 +316,7 @@ redis-cli -n 2 DEL user:<id>:stream
|
||||
|
||||
## Settings reference
|
||||
|
||||
Everything in `application/core/settings.py`:
|
||||
Everything in `docsgpt/core/settings.py`:
|
||||
|
||||
| Setting | Default | Purpose |
|
||||
| --------------------------------------------- | ------- | --------------------------------------------- |
|
||||
@@ -364,16 +364,16 @@ re-surface UX is too aggressive for v2.
|
||||
|
||||
The chat-stream reconnect reader `GET /api/messages/<id>/events`
|
||||
is a native-async Starlette route mounted in
|
||||
`application/asgi.py`, not a Flask route. Plain `flask run`
|
||||
`docsgpt/asgi.py`, not a Flask route. Plain `flask run`
|
||||
serves only the WSGI Flask app, so under it that endpoint 404s
|
||||
and reconnect-after-disconnect can't resume. Run the backend via
|
||||
`uvicorn application.asgi:asgi_app --reload` (or the production
|
||||
`uvicorn docsgpt.asgi:asgi_app --reload` (or the production
|
||||
gunicorn uvicorn-worker) to exercise it.
|
||||
|
||||
### Werkzeug doesn't auto-reload route files
|
||||
|
||||
The dev server (`flask run`) doesn't watch
|
||||
`application/api/events/routes.py` for changes by default.
|
||||
`docsgpt/api/events/routes.py` for changes by default.
|
||||
After editing the route, restart Flask manually — `--reload`
|
||||
isn't on. (Production gunicorn reloads via deploy.)
|
||||
|
||||
|
||||
@@ -12,7 +12,7 @@
|
||||
#
|
||||
# Everything the default configuration needs is inside the image: embedding
|
||||
# models, their tokenizers, tiktoken's encoding and, with the docling extra,
|
||||
# docling's layout/table/OCR models. `python -m application.scripts.verify_offline`
|
||||
# docling's layout/table/OCR models. `python -m docsgpt.scripts.verify_offline`
|
||||
# under `docker run --network none` proves it.
|
||||
|
||||
FROM ubuntu:24.04 AS builder
|
||||
@@ -25,7 +25,9 @@ RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends python3.12 python3.12-venv ca-certificates && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY requirements.txt requirements-docling.txt requirements-milvus.txt ./
|
||||
# Build context is the repository root (see .dockerignore there):
|
||||
# docker build -f docsgpt/Dockerfile .
|
||||
COPY docsgpt/requirements.txt docsgpt/requirements-docling.txt docsgpt/requirements-milvus.txt ./
|
||||
|
||||
RUN python3.12 -m venv /venv
|
||||
ENV PATH="/venv/bin:$PATH"
|
||||
@@ -97,7 +99,7 @@ WORKDIR /app
|
||||
RUN groupadd -r appuser && \
|
||||
useradd -r -g appuser -d /app -s /sbin/nologin -c "Docker image user" appuser && \
|
||||
chown appuser:appuser /app && \
|
||||
install -d -o appuser -g appuser /app/models /app/application
|
||||
install -d -o appuser -g appuser /app/models /app/docsgpt
|
||||
|
||||
COPY --from=builder /venv /venv
|
||||
|
||||
@@ -117,14 +119,14 @@ ENV EMBEDDINGS_CACHE_DIR=/app/models \
|
||||
|
||||
# Only the modules the prefetch imports are copied first, so an unrelated
|
||||
# source edit does not invalidate the model layer.
|
||||
COPY --chown=appuser:appuser __init__.py /app/application/__init__.py
|
||||
COPY --chown=appuser:appuser scripts/__init__.py scripts/prefetch_models.py /app/application/scripts/
|
||||
COPY --chown=appuser:appuser vectorstore/__init__.py vectorstore/model_registry.py /app/application/vectorstore/
|
||||
COPY --chown=appuser:appuser docsgpt/__init__.py /app/docsgpt/__init__.py
|
||||
COPY --chown=appuser:appuser docsgpt/scripts/__init__.py docsgpt/scripts/prefetch_models.py /app/docsgpt/scripts/
|
||||
COPY --chown=appuser:appuser docsgpt/vectorstore/__init__.py docsgpt/vectorstore/model_registry.py /app/docsgpt/vectorstore/
|
||||
|
||||
USER appuser
|
||||
|
||||
ARG EMBEDDINGS_PREFETCH=""
|
||||
RUN PYTHONPATH=/app python -m application.scripts.prefetch_models ${EMBEDDINGS_PREFETCH} && \
|
||||
RUN PYTHONPATH=/app python -m docsgpt.scripts.prefetch_models ${EMBEDDINGS_PREFETCH} && \
|
||||
rm -rf /app/models/.locks /app/.cache
|
||||
|
||||
# docling downloads its layout, table-structure and OCR models on first parse;
|
||||
@@ -134,12 +136,14 @@ RUN if python -c "import docling" 2>/dev/null; then \
|
||||
rm -rf /app/.cache; \
|
||||
fi
|
||||
|
||||
COPY --chown=appuser:appuser . /app/application
|
||||
COPY --chown=appuser:appuser docsgpt /app/docsgpt
|
||||
# One-release alias so `-A application.app.celery` style entry points keep working.
|
||||
COPY --chown=appuser:appuser application /app/application
|
||||
|
||||
# Runtime data directories, owned by the process user so a named volume
|
||||
# mounted on them (docker-compose-standalone.yaml) inherits that ownership
|
||||
# and uploads work without running the container as root.
|
||||
RUN mkdir -p /app/application/inputs/local /app/inputs /app/indexes /app/vectors
|
||||
RUN mkdir -p /app/docsgpt/inputs/local /app/inputs /app/indexes /app/vectors
|
||||
|
||||
ENV FLASK_APP=app.py
|
||||
|
||||
@@ -155,11 +159,11 @@ ENV MALLOC_ARENA_MAX=2 \
|
||||
EXPOSE 7091
|
||||
|
||||
# BoundedDrainUvicornWorker makes max_requests recycles safe with held-open SSE
|
||||
# connections (see application/gunicorn_worker.py); with recycles now safe,
|
||||
# connections (see docsgpt/gunicorn_worker.py); with recycles now safe,
|
||||
# --max-requests is raised (kept for memory hygiene) to cut churn.
|
||||
CMD ["gunicorn", \
|
||||
"-w", "1", \
|
||||
"-k", "application.gunicorn_worker.BoundedDrainUvicornWorker", \
|
||||
"-k", "docsgpt.gunicorn_worker.BoundedDrainUvicornWorker", \
|
||||
"--bind", "0.0.0.0:7091", \
|
||||
"--timeout", "180", \
|
||||
"--graceful-timeout", "120", \
|
||||
@@ -167,5 +171,5 @@ CMD ["gunicorn", \
|
||||
"--worker-tmp-dir", "/dev/shm", \
|
||||
"--max-requests", "5000", \
|
||||
"--max-requests-jitter", "500", \
|
||||
"--config", "application/gunicorn_conf.py", \
|
||||
"application.asgi:asgi_app"]
|
||||
"--config", "docsgpt/gunicorn_conf.py", \
|
||||
"docsgpt.asgi:asgi_app"]
|
||||
File renamed without changes.
File renamed without changes.
@@ -1,9 +1,9 @@
|
||||
import logging
|
||||
|
||||
from application.agents.agentic_agent import AgenticAgent
|
||||
from application.agents.classic_agent import ClassicAgent
|
||||
from application.agents.research_agent import ResearchAgent
|
||||
from application.agents.workflow_agent import WorkflowAgent
|
||||
from docsgpt.agents.agentic_agent import AgenticAgent
|
||||
from docsgpt.agents.classic_agent import ClassicAgent
|
||||
from docsgpt.agents.research_agent import ResearchAgent
|
||||
from docsgpt.agents.workflow_agent import WorkflowAgent
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
import logging
|
||||
from typing import Dict, Generator, Optional
|
||||
|
||||
from application.agents.base import BaseAgent
|
||||
from application.agents.tools.internal_search import add_internal_search_tool
|
||||
from application.agents.tools.wiki import add_wiki_tool
|
||||
from application.logging import LogContext
|
||||
from docsgpt.agents.base import BaseAgent
|
||||
from docsgpt.agents.tools.internal_search import add_internal_search_tool
|
||||
from docsgpt.agents.tools.wiki import add_wiki_tool
|
||||
from docsgpt.logging import LogContext
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -6,30 +6,30 @@ from abc import ABC, abstractmethod
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, Generator, List, Optional
|
||||
|
||||
from application.agents.tool_executor import (
|
||||
from docsgpt.agents.tool_executor import (
|
||||
ToolExecutor,
|
||||
result_status,
|
||||
truncate_tool_result,
|
||||
)
|
||||
from application.core.json_schema_utils import (
|
||||
from docsgpt.core.json_schema_utils import (
|
||||
JsonSchemaValidationError,
|
||||
normalize_json_schema_payload,
|
||||
)
|
||||
from application.core.settings import settings
|
||||
from application.llm.handlers.base import (
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.llm.handlers.base import (
|
||||
ToolCall,
|
||||
_bound_tool_response_for_llm,
|
||||
)
|
||||
from application.guardrails.config import DEFAULT_BLOCK_MESSAGE as GUARDRAIL_DEFAULT_MESSAGE
|
||||
from application.guardrails.runtime import (
|
||||
from docsgpt.guardrails.config import DEFAULT_BLOCK_MESSAGE as GUARDRAIL_DEFAULT_MESSAGE
|
||||
from docsgpt.guardrails.runtime import (
|
||||
build_engine as build_guardrail_engine,
|
||||
resolve_config as resolve_guardrails_config,
|
||||
)
|
||||
from application.guardrails.stream import StreamingOutputGuard
|
||||
from application.guardrails.types import Action, Stage, resolve_tool_result
|
||||
from application.llm.handlers.handler_creator import LLMHandlerCreator
|
||||
from application.llm.llm_creator import LLMCreator
|
||||
from application.logging import build_stack_data, log_activity, LogContext
|
||||
from docsgpt.guardrails.stream import StreamingOutputGuard
|
||||
from docsgpt.guardrails.types import Action, Stage, resolve_tool_result
|
||||
from docsgpt.llm.handlers.handler_creator import LLMHandlerCreator
|
||||
from docsgpt.llm.llm_creator import LLMCreator
|
||||
from docsgpt.logging import build_stack_data, log_activity, LogContext
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -424,7 +424,7 @@ class BaseAgent(ABC):
|
||||
return meta["response_id"]
|
||||
budget = getattr(settings, "OPENAI_RESPONSES_CHAIN_BUDGET_TOKENS", None)
|
||||
if not budget:
|
||||
from application.core.model_utils import get_token_limit
|
||||
from docsgpt.core.model_utils import get_token_limit
|
||||
|
||||
budget = get_token_limit(
|
||||
getattr(self, "model_id", None),
|
||||
@@ -677,13 +677,13 @@ class BaseAgent(ABC):
|
||||
# ---- Context / token management ----
|
||||
|
||||
def _calculate_current_context_tokens(self, messages: List[Dict]) -> int:
|
||||
from application.api.answer.services.compression.token_counter import (
|
||||
from docsgpt.api.answer.services.compression.token_counter import (
|
||||
TokenCounter,
|
||||
)
|
||||
return TokenCounter.count_message_tokens(messages)
|
||||
|
||||
def _check_context_limit(self, messages: List[Dict]) -> bool:
|
||||
from application.core.model_utils import get_token_limit
|
||||
from docsgpt.core.model_utils import get_token_limit
|
||||
|
||||
try:
|
||||
current_tokens = self._calculate_current_context_tokens(messages)
|
||||
@@ -705,7 +705,7 @@ class BaseAgent(ABC):
|
||||
return False
|
||||
|
||||
def _validate_context_size(self, messages: List[Dict]) -> None:
|
||||
from application.core.model_utils import get_token_limit
|
||||
from docsgpt.core.model_utils import get_token_limit
|
||||
|
||||
current_tokens = self._calculate_current_context_tokens(messages)
|
||||
self.current_token_count = current_tokens
|
||||
@@ -728,7 +728,7 @@ class BaseAgent(ABC):
|
||||
)
|
||||
|
||||
def _truncate_text_middle(self, text: str, max_tokens: int) -> str:
|
||||
from application.utils import num_tokens_from_string
|
||||
from docsgpt.utils import num_tokens_from_string
|
||||
|
||||
current_tokens = num_tokens_from_string(text)
|
||||
if current_tokens <= max_tokens:
|
||||
@@ -767,8 +767,8 @@ class BaseAgent(ABC):
|
||||
(the usual culprit) and raises when even that cannot fit — BEFORE
|
||||
the usage decorators run, so a hopeless payload costs nothing.
|
||||
"""
|
||||
from application.core.model_utils import get_token_limit
|
||||
from application.utils import num_tokens_from_string
|
||||
from docsgpt.core.model_utils import get_token_limit
|
||||
from docsgpt.utils import num_tokens_from_string
|
||||
|
||||
context_limit = get_token_limit(
|
||||
self.model_id, user_id=self.model_user_id or self.user
|
||||
@@ -847,7 +847,7 @@ class BaseAgent(ABC):
|
||||
"""
|
||||
if getattr(self, "prompt_embeds_documents", False):
|
||||
return ""
|
||||
from application.api.answer.services.prompt_renderer import (
|
||||
from docsgpt.api.answer.services.prompt_renderer import (
|
||||
format_docs_for_prompt,
|
||||
)
|
||||
|
||||
@@ -887,7 +887,7 @@ class BaseAgent(ABC):
|
||||
attacker-influenceable material. The prompt was rendered before the
|
||||
agent ran, so the verdict is applied by patching it here.
|
||||
"""
|
||||
from application.api.answer.services.prompt_renderer import (
|
||||
from docsgpt.api.answer.services.prompt_renderer import (
|
||||
format_docs_for_prompt,
|
||||
)
|
||||
|
||||
@@ -959,7 +959,7 @@ class BaseAgent(ABC):
|
||||
"""Merge the cached InternalSearchTool's docs into ``retrieved_docs``,
|
||||
deduped, preserving any pre-fetched docs so a mixed-exposure agent cites
|
||||
both pre-fetched and tool-retrieved sources (not just the tool's)."""
|
||||
from application.agents.tools.internal_search import INTERNAL_TOOL_ID
|
||||
from docsgpt.agents.tools.internal_search import INTERNAL_TOOL_ID
|
||||
|
||||
executor = getattr(self, "tool_executor", None)
|
||||
loaded = getattr(executor, "_loaded_tools", None) or {}
|
||||
@@ -1004,8 +1004,8 @@ class BaseAgent(ABC):
|
||||
query: str,
|
||||
) -> List[Dict]:
|
||||
"""Build messages using pre-rendered system prompt"""
|
||||
from application.core.model_utils import get_token_limit
|
||||
from application.utils import num_tokens_from_string
|
||||
from docsgpt.core.model_utils import get_token_limit
|
||||
from docsgpt.utils import num_tokens_from_string
|
||||
|
||||
# Retrieval controls run inside _build_document_block for the usual
|
||||
# path; a prompt that embeds the documents skips that block entirely,
|
||||
@@ -1197,7 +1197,7 @@ class BaseAgent(ABC):
|
||||
history: List[Dict],
|
||||
max_tokens: int,
|
||||
) -> List[Dict]:
|
||||
from application.utils import num_tokens_from_string
|
||||
from docsgpt.utils import num_tokens_from_string
|
||||
|
||||
if not history or max_tokens <= 0:
|
||||
return []
|
||||
@@ -1,10 +1,10 @@
|
||||
import logging
|
||||
from typing import Dict, Generator, Optional
|
||||
|
||||
from application.agents.base import BaseAgent
|
||||
from application.agents.tools.internal_search import add_internal_search_tool
|
||||
from application.agents.tools.wiki import add_wiki_tool
|
||||
from application.logging import LogContext
|
||||
from docsgpt.agents.base import BaseAgent
|
||||
from docsgpt.agents.tools.internal_search import add_internal_search_tool
|
||||
from docsgpt.agents.tools.wiki import add_wiki_tool
|
||||
from docsgpt.logging import LogContext
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -8,7 +8,7 @@ import logging
|
||||
import uuid
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from application.core.settings import settings
|
||||
from docsgpt.core.settings import settings
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -61,15 +61,15 @@ _builtin_loaded_cache: Dict[tuple, List[str]] = {}
|
||||
def _load_tool(tool_name: str) -> Optional[Any]:
|
||||
"""Return a metadata-only instance of a tool, or None if it has no class."""
|
||||
# Imports just the named module (not the whole package) — avoids the
|
||||
# circular import via ``mcp_tool`` → ``application.api.user``.
|
||||
# circular import via ``mcp_tool`` → ``docsgpt.api.user``.
|
||||
if tool_name in _tool_cache:
|
||||
return _tool_cache[tool_name]
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
instance: Optional[Any] = None
|
||||
try:
|
||||
module = importlib.import_module(f"application.agents.tools.{tool_name}")
|
||||
module = importlib.import_module(f"docsgpt.agents.tools.{tool_name}")
|
||||
except ModuleNotFoundError:
|
||||
_tool_cache[tool_name] = None
|
||||
return None
|
||||
@@ -5,19 +5,19 @@ from __future__ import annotations
|
||||
import logging
|
||||
from typing import Any, Dict, Iterable, List, Optional
|
||||
|
||||
from application.agents.agent_creator import AgentCreator
|
||||
from application.agents.tool_executor import ToolExecutor
|
||||
from application.api.answer.services.prompt_renderer import (
|
||||
from docsgpt.agents.agent_creator import AgentCreator
|
||||
from docsgpt.agents.tool_executor import ToolExecutor
|
||||
from docsgpt.api.answer.services.prompt_renderer import (
|
||||
PromptRenderer,
|
||||
format_docs_for_prompt,
|
||||
prompt_embeds_documents,
|
||||
resolve_prompt_skeleton,
|
||||
)
|
||||
from application.api.answer.services.stream_processor import get_prompt
|
||||
from application.core.settings import settings
|
||||
from application.retriever.retriever_creator import RetrieverCreator
|
||||
from application.storage.db.repositories.sources import SourcesRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.api.answer.services.stream_processor import get_prompt
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.retriever.retriever_creator import RetrieverCreator
|
||||
from docsgpt.storage.db.repositories.sources import SourcesRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -70,13 +70,13 @@ def run_agent_headless(
|
||||
conversation_id: Optional[str] = None,
|
||||
) -> Dict[str, Any]:
|
||||
"""Run an agent with no live client; returns a structured outcome dict."""
|
||||
from application.core.model_utils import (
|
||||
from docsgpt.core.model_utils import (
|
||||
get_api_key_for_provider,
|
||||
get_default_model_id,
|
||||
get_provider_from_model_id,
|
||||
validate_model_id,
|
||||
)
|
||||
from application.utils import calculate_doc_token_budget
|
||||
from docsgpt.utils import calculate_doc_token_budget
|
||||
|
||||
owner = _resolve_owner(agent_config)
|
||||
if not owner:
|
||||
@@ -210,7 +210,7 @@ def run_agent_headless(
|
||||
# by raising. Dropping it here (as this loop used to) makes a broken run
|
||||
# indistinguishable from one that simply had nothing to say, and the
|
||||
# caller records it as a success. Mirrors the sentinel in
|
||||
# ``application/logging.py`` so an error carrying no message is still
|
||||
# ``docsgpt/logging.py`` so an error carrying no message is still
|
||||
# truthy instead of reading as "ok".
|
||||
if event.get("type") == "error":
|
||||
stream_error = str(event.get("error") or "")[:500] or "unspecified"
|
||||
@@ -4,15 +4,15 @@ import os
|
||||
import time
|
||||
from typing import Dict, Generator, List, Optional
|
||||
|
||||
from application.agents.base import BaseAgent
|
||||
from application.agents.tool_executor import ToolExecutor
|
||||
from application.agents.tools.internal_search import (
|
||||
from docsgpt.agents.base import BaseAgent
|
||||
from docsgpt.agents.tool_executor import ToolExecutor
|
||||
from docsgpt.agents.tools.internal_search import (
|
||||
INTERNAL_TOOL_ID,
|
||||
add_internal_search_tool,
|
||||
)
|
||||
from application.agents.tools.wiki import add_wiki_tool
|
||||
from application.agents.tools.think import THINK_TOOL_ENTRY, THINK_TOOL_ID
|
||||
from application.logging import LogContext
|
||||
from docsgpt.agents.tools.wiki import add_wiki_tool
|
||||
from docsgpt.agents.tools.think import THINK_TOOL_ENTRY, THINK_TOOL_ID
|
||||
from docsgpt.logging import LogContext
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -673,7 +673,7 @@ class ResearchAgent(BaseAgent):
|
||||
)
|
||||
|
||||
if log_context:
|
||||
from application.logging import build_stack_data
|
||||
from docsgpt.logging import build_stack_data
|
||||
|
||||
log_context.stacks.append(
|
||||
{"component": "synthesis_llm", "data": build_stack_data(self.llm)}
|
||||
File renamed without changes.
@@ -6,25 +6,25 @@ from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from sqlalchemy.exc import IntegrityError
|
||||
|
||||
from application.agents.default_tools import (
|
||||
from docsgpt.agents.default_tools import (
|
||||
BUILTIN_AGENT_TOOLS,
|
||||
is_headless_excluded_tool,
|
||||
is_synthesized_tool_id,
|
||||
resolve_tool_by_id,
|
||||
synthesized_default_tools,
|
||||
)
|
||||
from application.agents.tools.tool_action_parser import ToolActionParser
|
||||
from application.agents.tools.tool_manager import ToolManager
|
||||
from application.guardrails.types import Stage as GuardrailStage, resolve_tool_result
|
||||
from application.security.encryption import decrypt_credentials
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from application.storage.db.repositories.agents import AgentsRepository
|
||||
from application.storage.db.repositories.tool_call_attempts import (
|
||||
from docsgpt.agents.tools.tool_action_parser import ToolActionParser
|
||||
from docsgpt.agents.tools.tool_manager import ToolManager
|
||||
from docsgpt.guardrails.types import Stage as GuardrailStage, resolve_tool_result
|
||||
from docsgpt.security.encryption import decrypt_credentials
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.repositories.agents import AgentsRepository
|
||||
from docsgpt.storage.db.repositories.tool_call_attempts import (
|
||||
ToolCallAttemptsRepository,
|
||||
)
|
||||
from application.storage.db.repositories.user_tools import UserToolsRepository
|
||||
from application.storage.db.repositories.users import UsersRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.storage.db.repositories.user_tools import UserToolsRepository
|
||||
from docsgpt.storage.db.repositories.users import UsersRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -51,7 +51,7 @@ def _dedupable_tool_names() -> frozenset:
|
||||
Only these are safe to collapse — an MCP or user-added tool may
|
||||
legitimately appear more than once under a single name.
|
||||
"""
|
||||
from application.core.settings import settings
|
||||
from docsgpt.core.settings import settings
|
||||
|
||||
return frozenset(BUILTIN_AGENT_TOOLS) | frozenset(getattr(settings, "DEFAULT_CHAT_TOOLS", None) or [])
|
||||
|
||||
@@ -846,7 +846,7 @@ class ToolExecutor:
|
||||
silently bypasses the prompt — not even via the headless allowlist.
|
||||
"""
|
||||
try:
|
||||
from application.agents.tools.remote_device import RemoteDeviceTool
|
||||
from docsgpt.agents.tools.remote_device import RemoteDeviceTool
|
||||
|
||||
tool = RemoteDeviceTool(
|
||||
config=tool_data.get("config") or {},
|
||||
@@ -872,7 +872,7 @@ class ToolExecutor:
|
||||
error so a misconfigured tool never silently runs untrusted code.
|
||||
"""
|
||||
try:
|
||||
from application.agents.tools.code_executor import CodeExecutorTool
|
||||
from docsgpt.agents.tools.code_executor import CodeExecutorTool
|
||||
|
||||
tool = CodeExecutorTool(
|
||||
tool_config=tool_data.get("config") or {},
|
||||
File renamed without changes.
@@ -6,12 +6,12 @@ from urllib.parse import quote, urlencode
|
||||
|
||||
import requests
|
||||
|
||||
from application.agents.tools.api_body_serializer import (
|
||||
from docsgpt.agents.tools.api_body_serializer import (
|
||||
ContentType,
|
||||
RequestBodySerializer,
|
||||
)
|
||||
from application.agents.tools.base import Tool
|
||||
from application.security.safe_url import UnsafeUserUrlError, pinned_request
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.security.safe_url import UnsafeUserUrlError, pinned_request
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
+7
-7
@@ -17,17 +17,17 @@ import logging
|
||||
import uuid
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from application.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from application.agents.tools.base import Tool
|
||||
from application.core.settings import settings
|
||||
from application.sandbox.artifacts_capture import (
|
||||
from docsgpt.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.sandbox.artifacts_capture import (
|
||||
QuotaExceeded,
|
||||
append_artifact_version,
|
||||
persist_new_artifact,
|
||||
)
|
||||
from application.sandbox.sandbox_creator import SandboxCreator
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.sandbox.sandbox_creator import SandboxCreator
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -13,7 +13,7 @@ from __future__ import annotations
|
||||
import re
|
||||
from typing import Any, Optional
|
||||
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
|
||||
_REF_RE = re.compile(r"^[Aa](\d+)$")
|
||||
|
||||
+6
-6
@@ -12,12 +12,12 @@ from __future__ import annotations
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from application.core.settings import settings
|
||||
from application.sandbox.artifacts_capture import QuotaExceeded, persist_new_artifact
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.repositories.attachments import AttachmentsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from application.storage.storage_creator import StorageCreator
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.sandbox.artifacts_capture import QuotaExceeded, persist_new_artifact
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.repositories.attachments import AttachmentsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
from docsgpt.storage.storage_creator import StorageCreator
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
File renamed without changes.
@@ -2,7 +2,7 @@ import logging
|
||||
|
||||
import requests
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -6,32 +6,32 @@ import logging
|
||||
import re
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from application.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from application.agents.tools.attachment_bridge import (
|
||||
from docsgpt.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from docsgpt.agents.tools.attachment_bridge import (
|
||||
AttachmentBridgeError,
|
||||
bridge_attachment,
|
||||
match_attachment,
|
||||
)
|
||||
from application.agents.tools.base import Tool
|
||||
from application.core.settings import settings
|
||||
from application.sandbox.artifacts_capture import (
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.sandbox.artifacts_capture import (
|
||||
MAX_CAPTURED_FILES,
|
||||
capture_artifacts,
|
||||
snapshot_signatures,
|
||||
unique_input_path,
|
||||
)
|
||||
from application.sandbox.artifacts_capture import (
|
||||
from docsgpt.sandbox.artifacts_capture import (
|
||||
infer_mime as _infer_mime,
|
||||
)
|
||||
from application.sandbox.artifacts_capture import (
|
||||
from docsgpt.sandbox.artifacts_capture import (
|
||||
kind_for_mime as _kind_for_mime,
|
||||
)
|
||||
from application.sandbox.base import ExecResult
|
||||
from application.sandbox.sandbox_creator import SandboxCreator
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from application.storage.storage_creator import StorageCreator
|
||||
from application.utils import safe_filename
|
||||
from docsgpt.sandbox.base import ExecResult
|
||||
from docsgpt.sandbox.sandbox_creator import SandboxCreator
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
from docsgpt.storage.storage_creator import StorageCreator
|
||||
from docsgpt.utils import safe_filename
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import requests
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
|
||||
class CryptoPriceTool(Tool):
|
||||
@@ -2,7 +2,7 @@ import logging
|
||||
import time
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
+7
-7
@@ -2,10 +2,10 @@ import json
|
||||
import logging
|
||||
from typing import Dict, List, Optional
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from application.core.settings import settings
|
||||
from application.retriever.dispatcher import build_dispatcher
|
||||
from application.retriever.retriever_creator import RetrieverCreator
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.retriever.dispatcher import build_dispatcher
|
||||
from docsgpt.retriever.retriever_creator import RetrieverCreator
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -78,10 +78,10 @@ class InternalSearchTool(Tool):
|
||||
# Per-operation session: this tool runs inside the answer
|
||||
# generator hot path, so we open a short-lived read
|
||||
# connection for the batch lookup and release immediately.
|
||||
from application.storage.db.repositories.sources import (
|
||||
from docsgpt.storage.db.repositories.sources import (
|
||||
SourcesRepository,
|
||||
)
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
if isinstance(active_docs, str):
|
||||
active_docs = [active_docs]
|
||||
@@ -402,7 +402,7 @@ def sources_have_directory_structure(source: Dict) -> bool:
|
||||
# sites are updated to propagate user context.
|
||||
from sqlalchemy import text as _text
|
||||
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
if isinstance(active_docs, str):
|
||||
active_docs = [active_docs]
|
||||
@@ -19,13 +19,13 @@ from mcp.shared.auth import OAuthClientInformationFull, OAuthClientMetadata, OAu
|
||||
from pydantic import AnyHttpUrl, ValidationError
|
||||
from redis import Redis
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from application.api.user.tasks import mcp_oauth_task
|
||||
from application.cache import get_redis_instance
|
||||
from application.core.settings import settings
|
||||
from application.core.url_validation import SSRFError, validate_url
|
||||
from application.events.keys import stream_key
|
||||
from application.security.encryption import decrypt_credentials
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.api.user.tasks import mcp_oauth_task
|
||||
from docsgpt.cache import get_redis_instance
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.core.url_validation import SSRFError, validate_url
|
||||
from docsgpt.events.keys import stream_key
|
||||
from docsgpt.security.encryption import decrypt_credentials
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -856,10 +856,10 @@ class DBTokenStorage(TokenStorage):
|
||||
|
||||
def _fetch_session_data(self) -> dict:
|
||||
"""Read the JSONB ``session_data`` blob for this MCP server row."""
|
||||
from application.storage.db.repositories.connector_sessions import (
|
||||
from docsgpt.storage.db.repositories.connector_sessions import (
|
||||
ConnectorSessionsRepository,
|
||||
)
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
base_url = self.get_base_url(self.server_url)
|
||||
with db_readonly() as conn:
|
||||
@@ -893,10 +893,10 @@ class DBTokenStorage(TokenStorage):
|
||||
the scalar column — ``get_by_user_and_server_url`` needs that to
|
||||
resolve the row (``NULL = 'https://...'`` is UNKNOWN in SQL).
|
||||
"""
|
||||
from application.storage.db.repositories.connector_sessions import (
|
||||
from docsgpt.storage.db.repositories.connector_sessions import (
|
||||
ConnectorSessionsRepository,
|
||||
)
|
||||
from application.storage.db.session import db_session
|
||||
from docsgpt.storage.db.session import db_session
|
||||
|
||||
base_url = self.get_base_url(self.server_url)
|
||||
with db_session() as conn:
|
||||
@@ -905,10 +905,10 @@ class DBTokenStorage(TokenStorage):
|
||||
)
|
||||
|
||||
def _delete(self) -> None:
|
||||
from application.storage.db.repositories.connector_sessions import (
|
||||
from docsgpt.storage.db.repositories.connector_sessions import (
|
||||
ConnectorSessionsRepository,
|
||||
)
|
||||
from application.storage.db.session import db_session
|
||||
from docsgpt.storage.db.session import db_session
|
||||
|
||||
with db_session() as conn:
|
||||
ConnectorSessionsRepository(conn).delete(
|
||||
@@ -980,7 +980,7 @@ class DBTokenStorage(TokenStorage):
|
||||
"""
|
||||
from sqlalchemy import text
|
||||
|
||||
from application.storage.db.session import db_session
|
||||
from docsgpt.storage.db.session import db_session
|
||||
|
||||
def _delete_all() -> None:
|
||||
with db_session() as conn:
|
||||
@@ -4,8 +4,8 @@ import uuid
|
||||
|
||||
from .base import Tool
|
||||
from .path_utils import validate_tool_path
|
||||
from application.storage.db.repositories.memories import MemoriesRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.storage.db.repositories.memories import MemoriesRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -59,7 +59,7 @@ class MemoryTool(Tool):
|
||||
tool_id,
|
||||
)
|
||||
return False
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
|
||||
if not looks_like_uuid(tool_id):
|
||||
logger.debug(
|
||||
@@ -2,8 +2,8 @@ from typing import Any, Dict, List, Optional
|
||||
import uuid
|
||||
|
||||
from .base import Tool
|
||||
from application.storage.db.repositories.notes import NotesRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.storage.db.repositories.notes import NotesRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
|
||||
# Stable synthetic title used in the Postgres ``notes.title`` column.
|
||||
@@ -55,7 +55,7 @@ class NotesTool(Tool):
|
||||
return False
|
||||
if tool_id.startswith("default_"):
|
||||
return False
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
|
||||
return looks_like_uuid(tool_id)
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
from application.agents.tools.base import Tool
|
||||
from application.security.safe_url import UnsafeUserUrlError, pinned_request
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.security.safe_url import UnsafeUserUrlError, pinned_request
|
||||
|
||||
class NtfyTool(Tool):
|
||||
"""
|
||||
File renamed without changes.
@@ -2,7 +2,7 @@ import logging
|
||||
|
||||
import psycopg
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -19,20 +19,20 @@ from typing import Any, Callable, Dict, List, Optional
|
||||
|
||||
from celery import current_task
|
||||
|
||||
from application.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from application.agents.tools.attachment_bridge import (
|
||||
from docsgpt.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from docsgpt.agents.tools.attachment_bridge import (
|
||||
AttachmentBridgeError,
|
||||
bridge_attachment,
|
||||
match_attachment,
|
||||
)
|
||||
from application.agents.tools.base import Tool
|
||||
from application.core.json_schema_utils import (
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.core.json_schema_utils import (
|
||||
JsonSchemaValidationError,
|
||||
normalize_json_schema_payload,
|
||||
)
|
||||
from application.core.settings import settings
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -223,7 +223,7 @@ class ReadDocumentTool(Tool):
|
||||
until the OUTER task's limit (webhook runs have none) kills the whole agent run.
|
||||
"""
|
||||
parent = self._parent()
|
||||
from application.api.user.tasks import parse_timeout_for_size
|
||||
from docsgpt.api.user.tasks import parse_timeout_for_size
|
||||
|
||||
# OCR cost scales with pages, so the parse window grows with the document's size
|
||||
# (floored at DOCUMENT_PARSE_TIMEOUT).
|
||||
@@ -232,7 +232,7 @@ class ReadDocumentTool(Tool):
|
||||
# ``current_task`` is a Celery proxy: truthy only while this runs inside a worker task,
|
||||
# falsy in the web process (the bare proxy is NOT identity-None, so test truthiness).
|
||||
if current_task:
|
||||
from application.worker import run_parse_document
|
||||
from docsgpt.worker import run_parse_document
|
||||
|
||||
try:
|
||||
result = self._run_inline_bounded(
|
||||
@@ -250,7 +250,7 @@ class ReadDocumentTool(Tool):
|
||||
|
||||
from celery.exceptions import TimeoutError as CeleryTimeoutError
|
||||
|
||||
from application.api.user.tasks import parse_document, parse_task_time_limits
|
||||
from docsgpt.api.user.tasks import parse_document, parse_task_time_limits
|
||||
|
||||
# The task's per-call time limits are raised to match the awaited window: bound to
|
||||
# the base timeout at import, the worker would otherwise self-terminate a large
|
||||
@@ -350,7 +350,7 @@ class ReadDocumentTool(Tool):
|
||||
# are non-daemon and registered with ``concurrent.futures``' atexit hook,
|
||||
# which joins them -- so an abandoned parse would hold up worker
|
||||
# shutdown for the rest of its (size-scaled) window. Same reasoning as
|
||||
# application/guardrails/engine.py. ``shutdown(cancel_futures=True)``
|
||||
# docsgpt/guardrails/engine.py. ``shutdown(cancel_futures=True)``
|
||||
# is not an alternative: it only drops queued work items, never the one
|
||||
# already running.
|
||||
slot: Dict[str, Any] = {}
|
||||
@@ -2,8 +2,8 @@ import codecs
|
||||
|
||||
from markdownify import markdownify
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from application.security.safe_url import (
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.security.safe_url import (
|
||||
ResponseTooLargeError,
|
||||
UnsafeUserUrlError,
|
||||
pinned_fetch_bytes,
|
||||
@@ -11,18 +11,18 @@ import uuid
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, Optional
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from application.devices.broker import get_broker
|
||||
from application.devices.denylist import check_denylist
|
||||
from application.devices.normalizer import normalize_command
|
||||
from application.storage.db.repositories.device_audit_log import (
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.devices.broker import get_broker
|
||||
from docsgpt.devices.denylist import check_denylist
|
||||
from docsgpt.devices.normalizer import normalize_command
|
||||
from docsgpt.storage.db.repositories.device_audit_log import (
|
||||
DeviceAuditLogRepository,
|
||||
)
|
||||
from application.storage.db.repositories.device_auto_approve_patterns import (
|
||||
from docsgpt.storage.db.repositories.device_auto_approve_patterns import (
|
||||
DeviceAutoApprovePatternsRepository,
|
||||
)
|
||||
from application.storage.db.repositories.devices import DevicesRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.storage.db.repositories.devices import DevicesRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
@@ -7,16 +7,16 @@ import logging
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from application.agents.scheduler_utils import (
|
||||
from docsgpt.agents.scheduler_utils import (
|
||||
ScheduleValidationError,
|
||||
clamp_once_horizon,
|
||||
parse_delay,
|
||||
parse_run_at,
|
||||
)
|
||||
from application.core.settings import settings
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from application.storage.db.repositories.schedules import SchedulesRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.repositories.schedules import SchedulesRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
from .base import Tool
|
||||
|
||||
@@ -297,13 +297,13 @@ def _safe_default_allowlist(
|
||||
chat tools (resolved against ``settings.DEFAULT_CHAT_TOOLS`` and the
|
||||
user's ``tool_preferences.disabled_default_tools`` opt-outs).
|
||||
"""
|
||||
from application.agents.default_tools import (
|
||||
from docsgpt.agents.default_tools import (
|
||||
resolve_tool_by_id,
|
||||
synthesized_default_tools,
|
||||
)
|
||||
from application.storage.db.repositories.agents import AgentsRepository
|
||||
from application.storage.db.repositories.user_tools import UserToolsRepository
|
||||
from application.storage.db.repositories.users import UsersRepository
|
||||
from docsgpt.storage.db.repositories.agents import AgentsRepository
|
||||
from docsgpt.storage.db.repositories.user_tools import UserToolsRepository
|
||||
from docsgpt.storage.db.repositories.users import UsersRepository
|
||||
|
||||
def _is_safe(row: Dict[str, Any]) -> bool:
|
||||
actions = row.get("actions") or []
|
||||
File renamed without changes.
@@ -2,7 +2,7 @@ import logging
|
||||
|
||||
import requests
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
|
||||
THINK_TOOL_ID = "think"
|
||||
@@ -2,8 +2,8 @@ from typing import Any, Dict, List, Optional
|
||||
import uuid
|
||||
|
||||
from .base import Tool
|
||||
from application.storage.db.repositories.todos import TodosRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.storage.db.repositories.todos import TodosRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
|
||||
def _status_from_completed(completed: Any) -> str:
|
||||
@@ -60,7 +60,7 @@ class TodoListTool(Tool):
|
||||
return False
|
||||
if tool_id.startswith("default_"):
|
||||
return False
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
|
||||
return looks_like_uuid(tool_id)
|
||||
|
||||
File renamed without changes.
@@ -3,7 +3,7 @@ import inspect
|
||||
import os
|
||||
import pkgutil
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
|
||||
|
||||
class ToolManager:
|
||||
@@ -17,7 +17,7 @@ class ToolManager:
|
||||
for finder, name, ispkg in pkgutil.iter_modules([tools_dir]):
|
||||
if name == "base" or name.startswith("__"):
|
||||
continue
|
||||
module = importlib.import_module(f"application.agents.tools.{name}")
|
||||
module = importlib.import_module(f"docsgpt.agents.tools.{name}")
|
||||
for member_name, obj in inspect.getmembers(module, inspect.isclass):
|
||||
if issubclass(obj, Tool) and obj is not Tool and not obj.internal:
|
||||
tool_config = self.config.get(name, {})
|
||||
@@ -25,7 +25,7 @@ class ToolManager:
|
||||
|
||||
def load_tool(self, tool_name, tool_config, user_id=None):
|
||||
self.config[tool_name] = tool_config
|
||||
module = importlib.import_module(f"application.agents.tools.{tool_name}")
|
||||
module = importlib.import_module(f"docsgpt.agents.tools.{tool_name}")
|
||||
for member_name, obj in inspect.getmembers(module, inspect.isclass):
|
||||
if issubclass(obj, Tool) and obj is not Tool:
|
||||
if (
|
||||
@@ -1,15 +1,15 @@
|
||||
import logging
|
||||
from typing import Any, Dict, List, Optional
|
||||
|
||||
from application.agents.tools.base import Tool
|
||||
from application.agents.tools.path_utils import validate_tool_path
|
||||
from application.storage.db.repositories.wiki_pages import (
|
||||
from docsgpt.agents.tools.base import Tool
|
||||
from docsgpt.agents.tools.path_utils import validate_tool_path
|
||||
from docsgpt.storage.db.repositories.wiki_pages import (
|
||||
WikiPageConflict,
|
||||
WikiPagesRepository,
|
||||
_content_hash,
|
||||
rebuild_wiki_directory_structure,
|
||||
)
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -207,7 +207,7 @@ class WikiTool(Tool):
|
||||
idempotency key guards each edit independently and dedups broker
|
||||
redeliveries without colliding across pages of the same source.
|
||||
"""
|
||||
from application.api.user.tasks import reembed_wiki_page
|
||||
from docsgpt.api.user.tasks import reembed_wiki_page
|
||||
|
||||
reembed_wiki_page.delay(
|
||||
self.source_id,
|
||||
@@ -2,11 +2,11 @@ import logging
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, Generator, List, Optional, Tuple
|
||||
|
||||
from application.agents.base import BaseAgent
|
||||
from application.guardrails.config import (
|
||||
from docsgpt.agents.base import BaseAgent
|
||||
from docsgpt.guardrails.config import (
|
||||
DEFAULT_BLOCK_MESSAGE as GUARDRAIL_DEFAULT_MESSAGE,
|
||||
)
|
||||
from application.agents.workflows.schemas import (
|
||||
from docsgpt.agents.workflows.schemas import (
|
||||
ExecutionStatus,
|
||||
Workflow,
|
||||
WorkflowEdge,
|
||||
@@ -14,16 +14,16 @@ from application.agents.workflows.schemas import (
|
||||
WorkflowNode,
|
||||
WorkflowRun,
|
||||
)
|
||||
from application.agents.workflows.workflow_engine import WorkflowEngine
|
||||
from application.core.settings import settings
|
||||
from application.logging import LogContext, log_activity
|
||||
from application.sandbox.artifacts_capture import QuotaExceeded
|
||||
from application.storage.db.base_repository import looks_like_uuid
|
||||
from application.storage.db.repositories.workflow_edges import WorkflowEdgesRepository
|
||||
from application.storage.db.repositories.workflow_nodes import WorkflowNodesRepository
|
||||
from application.storage.db.repositories.workflow_runs import WorkflowRunsRepository
|
||||
from application.storage.db.repositories.workflows import WorkflowsRepository
|
||||
from application.storage.db.session import db_readonly, db_session
|
||||
from docsgpt.agents.workflows.workflow_engine import WorkflowEngine
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.logging import LogContext, log_activity
|
||||
from docsgpt.sandbox.artifacts_capture import QuotaExceeded
|
||||
from docsgpt.storage.db.base_repository import looks_like_uuid
|
||||
from docsgpt.storage.db.repositories.workflow_edges import WorkflowEdgesRepository
|
||||
from docsgpt.storage.db.repositories.workflow_nodes import WorkflowNodesRepository
|
||||
from docsgpt.storage.db.repositories.workflow_runs import WorkflowRunsRepository
|
||||
from docsgpt.storage.db.repositories.workflows import WorkflowsRepository
|
||||
from docsgpt.storage.db.session import db_readonly, db_session
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
@@ -327,8 +327,8 @@ class WorkflowAgent(BaseAgent):
|
||||
# parent), so skip the bridge for unowned/draft ids.
|
||||
if not persisted:
|
||||
return [], []
|
||||
from application.sandbox.artifacts_capture import persist_new_artifact
|
||||
from application.storage.storage_creator import StorageCreator
|
||||
from docsgpt.sandbox.artifacts_capture import persist_new_artifact
|
||||
from docsgpt.storage.storage_creator import StorageCreator
|
||||
|
||||
storage = StorageCreator.get_storage()
|
||||
max_bytes = int(getattr(settings, "ARTIFACT_MAX_BYTES", 0) or 0)
|
||||
File renamed without changes.
@@ -2,11 +2,11 @@
|
||||
|
||||
from typing import Dict, List, Optional, Type
|
||||
|
||||
from application.agents.agentic_agent import AgenticAgent
|
||||
from application.agents.base import BaseAgent
|
||||
from application.agents.classic_agent import ClassicAgent
|
||||
from application.agents.research_agent import ResearchAgent
|
||||
from application.agents.workflows.schemas import AgentType
|
||||
from docsgpt.agents.agentic_agent import AgenticAgent
|
||||
from docsgpt.agents.base import BaseAgent
|
||||
from docsgpt.agents.classic_agent import ClassicAgent
|
||||
from docsgpt.agents.research_agent import ResearchAgent
|
||||
from docsgpt.agents.workflows.schemas import AgentType
|
||||
|
||||
|
||||
class _WorkflowNodeMixin:
|
||||
File renamed without changes.
+36
-36
@@ -6,9 +6,9 @@ import uuid
|
||||
from datetime import datetime, timezone
|
||||
from typing import Any, Dict, Generator, List, Optional, TYPE_CHECKING
|
||||
|
||||
from application.agents.workflows.cel_evaluator import CelEvaluationError, evaluate_cel
|
||||
from application.agents.workflows.node_agent import WorkflowNodeAgentFactory
|
||||
from application.agents.workflows.schemas import (
|
||||
from docsgpt.agents.workflows.cel_evaluator import CelEvaluationError, evaluate_cel
|
||||
from docsgpt.agents.workflows.node_agent import WorkflowNodeAgentFactory
|
||||
from docsgpt.agents.workflows.schemas import (
|
||||
AgentNodeConfig,
|
||||
AgentType,
|
||||
CodeNodeConfig,
|
||||
@@ -19,13 +19,13 @@ from application.agents.workflows.schemas import (
|
||||
WorkflowGraph,
|
||||
WorkflowNode,
|
||||
)
|
||||
from application.core.json_schema_utils import (
|
||||
from docsgpt.core.json_schema_utils import (
|
||||
JsonSchemaValidationError,
|
||||
normalize_json_schema_payload,
|
||||
)
|
||||
from application.error import sanitize_api_error
|
||||
from application.templates.namespaces import NamespaceManager
|
||||
from application.templates.template_engine import TemplateEngine, TemplateRenderError
|
||||
from docsgpt.error import sanitize_api_error
|
||||
from docsgpt.templates.namespaces import NamespaceManager
|
||||
from docsgpt.templates.template_engine import TemplateEngine, TemplateRenderError
|
||||
|
||||
try:
|
||||
import jsonschema
|
||||
@@ -33,7 +33,7 @@ except ImportError: # pragma: no cover - optional dependency in some deployment
|
||||
jsonschema = None
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from application.agents.base import BaseAgent
|
||||
from docsgpt.agents.base import BaseAgent
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
StateValue = Any
|
||||
@@ -101,7 +101,7 @@ class WorkflowEngine:
|
||||
# node and agent-node tool in this run, so it is torn down exactly once
|
||||
# here rather than per node. peek_manager() never builds the manager, so
|
||||
# a run that never opened a session closes nothing.
|
||||
from application.sandbox.sandbox_creator import SandboxCreator
|
||||
from docsgpt.sandbox.sandbox_creator import SandboxCreator
|
||||
|
||||
mgr = SandboxCreator.peek_manager()
|
||||
if mgr is not None:
|
||||
@@ -312,13 +312,13 @@ class WorkflowEngine:
|
||||
def _execute_agent_node(
|
||||
self, node: WorkflowNode
|
||||
) -> Generator[Dict[str, str], None, None]:
|
||||
from application.core.model_utils import (
|
||||
from docsgpt.core.model_utils import (
|
||||
get_api_key_for_provider,
|
||||
get_model_capabilities,
|
||||
resolve_dispatch_provider,
|
||||
)
|
||||
|
||||
from application.api.answer.services.prompt_renderer import (
|
||||
from docsgpt.api.answer.services.prompt_renderer import (
|
||||
prompt_embeds_documents as _prompt_embeds_documents,
|
||||
)
|
||||
|
||||
@@ -512,8 +512,8 @@ class WorkflowEngine:
|
||||
self, node: WorkflowNode
|
||||
) -> Generator[Dict[str, str], None, None]:
|
||||
"""Run code in the run-scoped sandbox, persist produced files, and write an artifact reference."""
|
||||
from application.sandbox.artifacts_capture import capture_artifacts, snapshot_signatures
|
||||
from application.sandbox.sandbox_creator import SandboxCreator
|
||||
from docsgpt.sandbox.artifacts_capture import capture_artifacts, snapshot_signatures
|
||||
from docsgpt.sandbox.sandbox_creator import SandboxCreator
|
||||
|
||||
config = CodeNodeConfig(**node.config.get("config", node.config))
|
||||
code = config.code or ""
|
||||
@@ -627,13 +627,13 @@ class WorkflowEngine:
|
||||
self, manager: Any, session_id: str, inputs: List[str], user_id: str
|
||||
) -> List[str]:
|
||||
"""Stage referenced input artifacts (run-scoped, never cross-tenant) into the workspace."""
|
||||
from application.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from application.core.settings import settings
|
||||
from application.sandbox.artifacts_capture import unique_input_path
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from application.storage.storage_creator import StorageCreator
|
||||
from application.utils import safe_filename
|
||||
from docsgpt.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.sandbox.artifacts_capture import unique_input_path
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
from docsgpt.storage.storage_creator import StorageCreator
|
||||
from docsgpt.utils import safe_filename
|
||||
|
||||
loaded: List[str] = []
|
||||
raw_ids = self._resolve_input_artifact_ids(inputs)
|
||||
@@ -693,9 +693,9 @@ class WorkflowEngine:
|
||||
selection, so it never widens per-node document access. Best-effort:
|
||||
a resolution failure drops the manifest, never the node.
|
||||
"""
|
||||
from application.agents.tools.artifact_ref import make_ref, resolve_artifact_id
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.agents.tools.artifact_ref import make_ref, resolve_artifact_id
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
try:
|
||||
raw_ids = self._resolve_input_artifact_ids(node_config.input_documents)
|
||||
@@ -738,10 +738,10 @@ class WorkflowEngine:
|
||||
supported_types: List[str],
|
||||
) -> List[Dict[str, Any]]:
|
||||
"""Resolve a node's selected documents to native/extracted attachment dicts for its LLM."""
|
||||
from application.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from application.core.settings import settings
|
||||
from application.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.agents.tools.artifact_ref import resolve_artifact_id
|
||||
from docsgpt.core.settings import settings
|
||||
from docsgpt.storage.db.repositories.artifacts import ArtifactsRepository
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
raw_ids = self._resolve_input_artifact_ids(node_config.input_documents)
|
||||
if not raw_ids:
|
||||
@@ -909,8 +909,8 @@ class WorkflowEngine:
|
||||
deadline: ``time.monotonic()`` value past which no further blocking
|
||||
parse may run, shared across the node's documents.
|
||||
"""
|
||||
from application.parser.document_reader import truncate_text_head_tail
|
||||
from application.storage.storage_creator import StorageCreator
|
||||
from docsgpt.parser.document_reader import truncate_text_head_tail
|
||||
from docsgpt.storage.storage_creator import StorageCreator
|
||||
|
||||
# An uploaded chat attachment was already parsed (and possibly OCR'd) when it
|
||||
# was stored, so re-parsing it here would repeat the dominant cost of the run
|
||||
@@ -965,12 +965,12 @@ class WorkflowEngine:
|
||||
"""
|
||||
from celery.exceptions import TimeoutError as CeleryTimeoutError
|
||||
|
||||
from application.api.user.tasks import (
|
||||
from docsgpt.api.user.tasks import (
|
||||
parse_document,
|
||||
parse_task_time_limits,
|
||||
parse_timeout_for_size,
|
||||
)
|
||||
from application.core.settings import settings
|
||||
from docsgpt.core.settings import settings
|
||||
|
||||
user_id = self._resolve_user_id()
|
||||
if not user_id:
|
||||
@@ -999,7 +999,7 @@ class WorkflowEngine:
|
||||
# default ``disable_sync_subtasks=True`` makes ``get()`` raise RuntimeError
|
||||
# ("Never call result.get() within a task!"). The dedicated parsing queue +
|
||||
# separate workers already avoid the real self-deadlock, so opt out
|
||||
# explicitly (mirrors application/agents/tools/read_document.py).
|
||||
# explicitly (mirrors docsgpt/agents/tools/read_document.py).
|
||||
result = async_result.get(timeout=timeout, disable_sync_subtasks=False)
|
||||
except (CeleryTimeoutError, TimeoutError):
|
||||
logger.warning("Workflow node: document parse timed out for %s", artifact_id)
|
||||
@@ -1082,7 +1082,7 @@ class WorkflowEngine:
|
||||
|
||||
def _resolve_code_timeout(self, requested: Optional[int]) -> float:
|
||||
"""Return the stricter of the node's requested timeout and the sandbox cap."""
|
||||
from application.core.settings import settings
|
||||
from docsgpt.core.settings import settings
|
||||
|
||||
cap = float(getattr(settings, "SANDBOX_EXEC_TIMEOUT", 60))
|
||||
if requested is None:
|
||||
@@ -1331,8 +1331,8 @@ class WorkflowEngine:
|
||||
logger.warning("Workflow node sources dropped: no owner to authorize.")
|
||||
return []
|
||||
|
||||
from application.api.user.team_sharing import can_access
|
||||
from application.storage.db.session import db_readonly
|
||||
from docsgpt.api.user.team_sharing import can_access
|
||||
from docsgpt.storage.db.session import db_readonly
|
||||
|
||||
allowed = []
|
||||
try:
|
||||
@@ -1359,7 +1359,7 @@ class WorkflowEngine:
|
||||
Returns:
|
||||
list: Retrieved documents, empty when there was nothing to fetch.
|
||||
"""
|
||||
from application.retriever.retriever_creator import RetrieverCreator
|
||||
from docsgpt.retriever.retriever_creator import RetrieverCreator
|
||||
|
||||
query = self.state.get("query", "")
|
||||
if not query:
|
||||
@@ -1,11 +1,11 @@
|
||||
# Alembic configuration for the DocsGPT user-data Postgres database.
|
||||
#
|
||||
# The SQLAlchemy URL is deliberately NOT set here — env.py reads it from
|
||||
# ``application.core.settings.settings.POSTGRES_URI`` so the same config
|
||||
# ``docsgpt.core.settings.settings.POSTGRES_URI`` so the same config
|
||||
# source serves the running app and migrations. To run from the project
|
||||
# root::
|
||||
#
|
||||
# alembic -c application/alembic.ini upgrade head
|
||||
# alembic -c docsgpt/alembic.ini upgrade head
|
||||
|
||||
[alembic]
|
||||
script_location = %(here)s/alembic
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Alembic environment for the DocsGPT user-data Postgres database.
|
||||
|
||||
The URL is pulled from ``application.core.settings`` rather than
|
||||
The URL is pulled from ``docsgpt.core.settings`` rather than
|
||||
``alembic.ini`` so that a single ``POSTGRES_URI`` env var drives both the
|
||||
running app and ``alembic`` CLI invocations.
|
||||
"""
|
||||
@@ -10,7 +10,7 @@ from logging.config import fileConfig
|
||||
from pathlib import Path
|
||||
|
||||
# Make the project root importable regardless of cwd. env.py lives at
|
||||
# <repo>/application/alembic/env.py, so parents[2] is the repo root.
|
||||
# <repo>/docsgpt/alembic/env.py, so parents[2] is the repo root.
|
||||
_PROJECT_ROOT = Path(__file__).resolve().parents[2]
|
||||
if str(_PROJECT_ROOT) not in sys.path:
|
||||
sys.path.insert(0, str(_PROJECT_ROOT))
|
||||
@@ -18,8 +18,8 @@ if str(_PROJECT_ROOT) not in sys.path:
|
||||
from alembic import context # noqa: E402
|
||||
from sqlalchemy import engine_from_config, pool # noqa: E402
|
||||
|
||||
from application.core.settings import settings # noqa: E402
|
||||
from application.storage.db.models import metadata as target_metadata # noqa: E402
|
||||
from docsgpt.core.settings import settings # noqa: E402
|
||||
from docsgpt.storage.db.models import metadata as target_metadata # noqa: E402
|
||||
|
||||
config = context.config
|
||||
|
||||
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
File renamed without changes.
Loaded 100 of 984 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user