Files
DocsGPT/deployment/docker-compose.yaml
T
Alex 574f96341e refactor: rename the application package to docsgpt
The backend import package is now docsgpt, the name it will carry on PyPI;
application was far too generic to install into anyone's site-packages.
git mv plus a mechanical rewrite of every import, dotted string and path
reference: 734 Python files, the compose files, Dockerfile, workflows, docs,
setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage
config, .gitignore. Behaviour is unchanged.

Kept for one release:
- A top-level application package whose meta-path finder resolves
  application.x.y to the already-imported docsgpt.x.y object, so old imports
  and entry points (celery -A application.app.celery,
  uvicorn application.asgi:asgi_app) keep working with a FutureWarning.
- Celery registers every application.* task name as an alias of its
  docsgpt.* task on start-up, so messages queued by the previous release still
  run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries
  the previous release wrote are left unread instead of firing twice.

The backend image builds from the repository root (docker build -f
docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore
allow-lists docsgpt/ and application/ and keeps caches, local data, .env
files, the sample index files and the Dockerfile out. Compose and the image
workflows point at the new context.
2026-09-07 10:20:43 +01:00

128 lines
4.3 KiB
YAML

name: docsgpt-oss
services:
frontend:
build:
context: ../frontend
# Vite dev server with hot reload over the bind mount below. The default
# target (what Docker Hub publishes) is a static build behind nginx.
target: dev
volumes:
- ../frontend/src:/app/src
environment:
# Every VITE_* the app reads. A bare name is passed through only when it is
# set in the shell or the --env-file, so an unset one does not reach the
# container as an empty string and override the image's own default.
- VITE_API_HOST=http://localhost:7091
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
- VITE_BASE_URL
- VITE_GOOGLE_CLIENT_ID
- VITE_GOOGLE_PICKER_API_KEY
- VITE_SHARE_POINT_CLIENT_ID
- VITE_CONFLUENCE_CLIENT_ID
- VITE_NOTIFICATION_TEXT
- VITE_NOTIFICATION_LINK
- VITE_ENABLE_VOICE_INPUT
- VITE_DISABLE_SOURCE_FE
- VITE_USE_V
ports:
- "5173:5173"
depends_on:
- backend
backend:
user: root
build:
context: ..
dockerfile: docsgpt/Dockerfile
args:
# Optional extras to bake in (comma-separated): docling, milvus. The
# docling extra brings the layout-model parser/OCR backend and its
# models. INSTALL_DOCLING=true is the older spelling of EXTRAS=docling.
EXTRAS: ${EXTRAS:-}
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
# Bake the tesseract binary behind OCR_ENABLED=true (~35 MB): set
# INSTALL_TESSERACT=true in ../.env or the shell (setup.sh does this
# when OCR is enabled). Off by default, like docling. Upgrading a
# deployment that already runs OCR_ENABLED=true with tesseract: add
# INSTALL_TESSERACT=true to ../.env before rebuilding, or scanned
# pages fail with an install hint.
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
env_file:
- ../.env
environment:
# Override URLs to use docker service names
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
ports:
- "7091:7091"
volumes:
- ../docsgpt/indexes:/app/indexes
- ../docsgpt/inputs:/app/inputs
- ../docsgpt/vectors:/app/vectors
depends_on:
redis:
condition: service_started
postgres:
condition: service_healthy
worker:
user: root
build:
context: ..
dockerfile: docsgpt/Dockerfile
args:
EXTRAS: ${EXTRAS:-}
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
# Consumes the default queue AND the dedicated `parsing` (read_document /
# parse_document) and `embeddings` (query embedding) queues. Without `parsing`
# the read_document await never resolves; without `embeddings` every search
# fails after EMBEDDINGS_DELEGATE_TIMEOUT, because EMBEDDINGS_DELEGATE_TO_WORKER
# is on by default. For heavy/OCR parsing run a separate worker with `-Q parsing`;
# to keep query latency off the ingest pool, another with `-Q embeddings`.
command: celery -A docsgpt.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
env_file:
- ../.env
environment:
# Override URLs to use docker service names
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- API_URL=http://backend:7091
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
volumes:
- ../docsgpt/indexes:/app/indexes
- ../docsgpt/inputs:/app/inputs
- ../docsgpt/vectors:/app/vectors
depends_on:
redis:
condition: service_started
postgres:
condition: service_healthy
redis:
image: redis:6-alpine
ports:
- 6379:6379
postgres:
image: postgres:16-alpine
environment:
- POSTGRES_USER=docsgpt
- POSTGRES_PASSWORD=docsgpt
- POSTGRES_DB=docsgpt
ports:
- "5432:5432"
volumes:
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U docsgpt -d docsgpt"]
interval: 5s
timeout: 5s
retries: 10
volumes:
postgres_data: