Files
DocsGPT/deployment/docker-compose.yaml
T
Alex 37d93cbd86 Parse documents on a Celery parsing worker via a read_document tool
Replace the sandbox Docling extractor with read_document, backed by the in-process
backend parser (the same one ingestion uses) and offloaded to a dedicated
'parsing' Celery queue so it can run on GPU-capable workers with predictable RAM.
The tool resolves the input ref under the run-scoped gate, enqueues the parse,
and awaits it with a timeout (degrading to an error rather than hanging); the
worker independently re-resolves the artifact through the same gate and never
trusts a raw path. Untrusted files get the upload path's safeguards (extension
whitelist, size cap, sanitized temp file, cleanup). Options: output
(markdown/text/structured/chunks), ocr, pages, engine, max_chars, include_tables,
persist, json_schema. The workflow native-file 'extract' fallback now uses the
same worker path, so document parsing no longer needs the sandbox and works on
every backend.

Also fixes the branch's periodic-task test (the sandbox reaper made it 12) and
points the dev and e2e Celery workers at the parsing queue.
2026-06-25 13:24:12 +01:00

137 lines
4.6 KiB
YAML

name: docsgpt-oss
services:
frontend:
build: ../frontend
volumes:
- ../frontend/src:/app/src
environment:
- VITE_API_HOST=http://localhost:7091
- VITE_API_STREAMING=$VITE_API_STREAMING
- VITE_GOOGLE_CLIENT_ID=$VITE_GOOGLE_CLIENT_ID
ports:
- "5173:5173"
depends_on:
- backend
backend:
user: root
build: ../application
env_file:
- ../.env
environment:
# Override URLs to use docker service names
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
# Code-execution runner reached over HTTP + WebSocket (no docker socket).
- SANDBOX_GATEWAY_URL=http://docsgpt-sandbox:8888
# Select the runner's env-scrubbing kernelspec (distinct name; never
# shadowed by the stock "python3" spec). Must match the kernel the
# docsgpt-sandbox image installs.
- SANDBOX_KERNEL_NAME=docsgpt-python
ports:
- "7091:7091"
volumes:
- ../application/indexes:/app/indexes
- ../application/inputs:/app/inputs
- ../application/vectors:/app/vectors
depends_on:
redis:
condition: service_started
postgres:
condition: service_healthy
worker:
user: root
build: ../application
# Consumes the default queue AND the dedicated `parsing` queue (read_document /
# parse_document). Without `parsing` here the read_document await never resolves.
# For heavy/OCR parsing run a separate worker with `-Q parsing` (see
# deployment/sandbox/README.md).
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing
env_file:
- ../.env
environment:
# Override URLs to use docker service names
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- API_URL=http://backend:7091
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
- SANDBOX_GATEWAY_URL=http://docsgpt-sandbox:8888
# Env-scrubbing kernelspec selected by name (see backend service).
- SANDBOX_KERNEL_NAME=docsgpt-python
volumes:
- ../application/indexes:/app/indexes
- ../application/inputs:/app/inputs
- ../application/vectors:/app/vectors
depends_on:
redis:
condition: service_started
postgres:
condition: service_healthy
# Always-on code-execution runner (Jupyter Kernel Gateway). Sessions are
# in-process kernels, never child containers; the Docker socket is NOT
# mounted. On an internal-only network — no host port is published, so the
# runner is reachable only from backend/worker, not from the host/internet.
# Egress/SSRF blocks, the gVisor `runsc` runtime, and seccomp profile come in
# the hardening slice.
#
# SINGLE TRUST DOMAIN: all sessions share this one container/uid and are
# isolated by working directory only (per-session cwd) — not by a kernel/OS
# boundary. The custom kernelspec scrubs secrets from the kernel env, but
# sibling workspaces are readable under the shared uid and kernels share one
# address space. Do NOT add `env_file: ../.env` here (the runner needs no app
# secrets). For cross-tenant / untrusted multi-tenant workloads use a
# per-session VM via SANDBOX_BACKEND=daytona instead.
docsgpt-sandbox:
build: ./sandbox
mem_limit: ${SANDBOX_MEMORY:-1g}
cpus: ${SANDBOX_CPUS:-1.0}
pids_limit: 256
read_only: true
environment:
# Keep Jupyter's runtime/connection files on the writable tmpfs.
- JUPYTER_RUNTIME_DIR=/tmp/jupyter-runtime
- JUPYTER_DATA_DIR=/tmp/jupyter-data
tmpfs:
# Per-session workspaces (/tmp/docsgpt-sandbox/<session_id>) and Jupyter
# runtime files live on tmpfs; the root FS is read-only everywhere else.
- /tmp
networks:
- sandbox-net
- default
redis:
image: redis:6-alpine
ports:
- 6379:6379
postgres:
image: postgres:16-alpine
environment:
- POSTGRES_USER=docsgpt
- POSTGRES_PASSWORD=docsgpt
- POSTGRES_DB=docsgpt
ports:
- "5432:5432"
volumes:
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U docsgpt -d docsgpt"]
interval: 5s
timeout: 5s
retries: 10
networks:
# Internal-only network for the sandbox runner (no external egress route via
# this network; the runner still reaches the internet via the default bridge).
sandbox-net:
internal: true
volumes:
postgres_data: