mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 20:13:04 +00:00
Replace the sandbox Docling extractor with read_document, backed by the in-process backend parser (the same one ingestion uses) and offloaded to a dedicated 'parsing' Celery queue so it can run on GPU-capable workers with predictable RAM. The tool resolves the input ref under the run-scoped gate, enqueues the parse, and awaits it with a timeout (degrading to an error rather than hanging); the worker independently re-resolves the artifact through the same gate and never trusts a raw path. Untrusted files get the upload path's safeguards (extension whitelist, size cap, sanitized temp file, cleanup). Options: output (markdown/text/structured/chunks), ocr, pages, engine, max_chars, include_tables, persist, json_schema. The workflow native-file 'extract' fallback now uses the same worker path, so document parsing no longer needs the sandbox and works on every backend. Also fixes the branch's periodic-task test (the sandbox reaper made it 12) and points the dev and e2e Celery workers at the parsing queue.
137 lines
4.6 KiB
YAML
137 lines
4.6 KiB
YAML
name: docsgpt-oss
|
|
services:
|
|
frontend:
|
|
build: ../frontend
|
|
volumes:
|
|
- ../frontend/src:/app/src
|
|
environment:
|
|
- VITE_API_HOST=http://localhost:7091
|
|
- VITE_API_STREAMING=$VITE_API_STREAMING
|
|
- VITE_GOOGLE_CLIENT_ID=$VITE_GOOGLE_CLIENT_ID
|
|
ports:
|
|
- "5173:5173"
|
|
depends_on:
|
|
- backend
|
|
|
|
backend:
|
|
user: root
|
|
build: ../application
|
|
env_file:
|
|
- ../.env
|
|
environment:
|
|
# Override URLs to use docker service names
|
|
- CELERY_BROKER_URL=redis://redis:6379/0
|
|
- CELERY_RESULT_BACKEND=redis://redis:6379/1
|
|
- CACHE_REDIS_URL=redis://redis:6379/2
|
|
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
|
# Code-execution runner reached over HTTP + WebSocket (no docker socket).
|
|
- SANDBOX_GATEWAY_URL=http://docsgpt-sandbox:8888
|
|
# Select the runner's env-scrubbing kernelspec (distinct name; never
|
|
# shadowed by the stock "python3" spec). Must match the kernel the
|
|
# docsgpt-sandbox image installs.
|
|
- SANDBOX_KERNEL_NAME=docsgpt-python
|
|
ports:
|
|
- "7091:7091"
|
|
volumes:
|
|
- ../application/indexes:/app/indexes
|
|
- ../application/inputs:/app/inputs
|
|
- ../application/vectors:/app/vectors
|
|
depends_on:
|
|
redis:
|
|
condition: service_started
|
|
postgres:
|
|
condition: service_healthy
|
|
|
|
worker:
|
|
user: root
|
|
build: ../application
|
|
# Consumes the default queue AND the dedicated `parsing` queue (read_document /
|
|
# parse_document). Without `parsing` here the read_document await never resolves.
|
|
# For heavy/OCR parsing run a separate worker with `-Q parsing` (see
|
|
# deployment/sandbox/README.md).
|
|
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing
|
|
env_file:
|
|
- ../.env
|
|
environment:
|
|
# Override URLs to use docker service names
|
|
- CELERY_BROKER_URL=redis://redis:6379/0
|
|
- CELERY_RESULT_BACKEND=redis://redis:6379/1
|
|
- API_URL=http://backend:7091
|
|
- CACHE_REDIS_URL=redis://redis:6379/2
|
|
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
|
- SANDBOX_GATEWAY_URL=http://docsgpt-sandbox:8888
|
|
# Env-scrubbing kernelspec selected by name (see backend service).
|
|
- SANDBOX_KERNEL_NAME=docsgpt-python
|
|
volumes:
|
|
- ../application/indexes:/app/indexes
|
|
- ../application/inputs:/app/inputs
|
|
- ../application/vectors:/app/vectors
|
|
depends_on:
|
|
redis:
|
|
condition: service_started
|
|
postgres:
|
|
condition: service_healthy
|
|
|
|
# Always-on code-execution runner (Jupyter Kernel Gateway). Sessions are
|
|
# in-process kernels, never child containers; the Docker socket is NOT
|
|
# mounted. On an internal-only network — no host port is published, so the
|
|
# runner is reachable only from backend/worker, not from the host/internet.
|
|
# Egress/SSRF blocks, the gVisor `runsc` runtime, and seccomp profile come in
|
|
# the hardening slice.
|
|
#
|
|
# SINGLE TRUST DOMAIN: all sessions share this one container/uid and are
|
|
# isolated by working directory only (per-session cwd) — not by a kernel/OS
|
|
# boundary. The custom kernelspec scrubs secrets from the kernel env, but
|
|
# sibling workspaces are readable under the shared uid and kernels share one
|
|
# address space. Do NOT add `env_file: ../.env` here (the runner needs no app
|
|
# secrets). For cross-tenant / untrusted multi-tenant workloads use a
|
|
# per-session VM via SANDBOX_BACKEND=daytona instead.
|
|
docsgpt-sandbox:
|
|
build: ./sandbox
|
|
mem_limit: ${SANDBOX_MEMORY:-1g}
|
|
cpus: ${SANDBOX_CPUS:-1.0}
|
|
pids_limit: 256
|
|
read_only: true
|
|
environment:
|
|
# Keep Jupyter's runtime/connection files on the writable tmpfs.
|
|
- JUPYTER_RUNTIME_DIR=/tmp/jupyter-runtime
|
|
- JUPYTER_DATA_DIR=/tmp/jupyter-data
|
|
tmpfs:
|
|
# Per-session workspaces (/tmp/docsgpt-sandbox/<session_id>) and Jupyter
|
|
# runtime files live on tmpfs; the root FS is read-only everywhere else.
|
|
- /tmp
|
|
networks:
|
|
- sandbox-net
|
|
- default
|
|
|
|
redis:
|
|
image: redis:6-alpine
|
|
ports:
|
|
- 6379:6379
|
|
|
|
postgres:
|
|
image: postgres:16-alpine
|
|
environment:
|
|
- POSTGRES_USER=docsgpt
|
|
- POSTGRES_PASSWORD=docsgpt
|
|
- POSTGRES_DB=docsgpt
|
|
ports:
|
|
- "5432:5432"
|
|
volumes:
|
|
- postgres_data:/var/lib/postgresql/data
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U docsgpt -d docsgpt"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 10
|
|
|
|
networks:
|
|
# Internal-only network for the sandbox runner (no external egress route via
|
|
# this network; the runner still reaches the internet via the default bridge).
|
|
sandbox-net:
|
|
internal: true
|
|
|
|
volumes:
|
|
postgres_data:
|
|
|