mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 16:13:23 +00:00
Conflicts, and how each was taken: - application/core/settings.py — ours. The renamed OCR_ENABLED / OCR_ATTACHMENTS_ENABLED / OCR_MIN_CHARS_PER_PAGE accept main's DOCLING_OCR_* spellings as AliasChoices, so nothing is dropped. - application/Dockerfile — both. Main's install layers plus the INSTALL_DOCLING build arg. - application/parser/file/constants.py — both imports. - deployment/docker-compose.yaml — both. The INSTALL_DOCLING / INSTALL_TESSERACT build args on backend and worker, and main's -Q docsgpt,parsing,embeddings, which query embedding needs. - tests/conftest.py — theirs. Both sides fixed the same pytest-postgresql 9.0.0 autocommit= breakage; main's spelling is the one already on main. - application/requirements.txt — the comments claimed different reasons torch is in core. Main's is the true one now: it removed sentence-transformers, so docling is torch's only remaining consumer. Two things the merge broke without conflicting: - onnxruntime. This branch moved it out of core into the docling extra; main meanwhile made it the runtime local embeddings execute on (fastembed). Git took the deletion, leaving fastembed with no pinned runtime in a repo that pins everything. Restored to core, and no longer pinned twice from the extra. - The frontend copy of ATTACHMENT_PARSER_EXTENSIONS. The backend list is derived and picked up the anydoc suffixes; the hand-kept frontend mirror did not, so the composer would refuse files the API accepts. tests/parser/file/test_constants.py is what caught it. ruff, pytest (9897 passed), frontend build and docs build all pass. The image build is unverified: no Docker daemon on this machine.
107 lines
3.5 KiB
YAML
107 lines
3.5 KiB
YAML
name: docsgpt-oss
|
|
services:
|
|
frontend:
|
|
build: ../frontend
|
|
volumes:
|
|
- ../frontend/src:/app/src
|
|
environment:
|
|
- VITE_API_HOST=http://localhost:7091
|
|
- VITE_API_STREAMING=$VITE_API_STREAMING
|
|
- VITE_GOOGLE_CLIENT_ID=$VITE_GOOGLE_CLIENT_ID
|
|
ports:
|
|
- "5173:5173"
|
|
depends_on:
|
|
- backend
|
|
|
|
backend:
|
|
user: root
|
|
build:
|
|
context: ../application
|
|
args:
|
|
# Bake the optional docling engine (layout-model OCR backend, structured
|
|
# output) into the image: set INSTALL_DOCLING=true in ../.env or the shell.
|
|
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
|
# Bake the tesseract binary behind OCR_ENABLED=true (~35 MB): set
|
|
# INSTALL_TESSERACT=true in ../.env or the shell (setup.sh does this
|
|
# when OCR is enabled). Off by default, like docling. Upgrading a
|
|
# deployment that already runs OCR_ENABLED=true with tesseract: add
|
|
# INSTALL_TESSERACT=true to ../.env before rebuilding, or scanned
|
|
# pages fail with an install hint.
|
|
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
|
|
env_file:
|
|
- ../.env
|
|
environment:
|
|
# Override URLs to use docker service names
|
|
- CELERY_BROKER_URL=redis://redis:6379/0
|
|
- CELERY_RESULT_BACKEND=redis://redis:6379/1
|
|
- CACHE_REDIS_URL=redis://redis:6379/2
|
|
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
|
ports:
|
|
- "7091:7091"
|
|
volumes:
|
|
- ../application/indexes:/app/indexes
|
|
- ../application/inputs:/app/inputs
|
|
- ../application/vectors:/app/vectors
|
|
depends_on:
|
|
redis:
|
|
condition: service_started
|
|
postgres:
|
|
condition: service_healthy
|
|
|
|
worker:
|
|
user: root
|
|
build:
|
|
context: ../application
|
|
args:
|
|
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
|
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
|
|
# Consumes the default queue AND the dedicated `parsing` (read_document /
|
|
# parse_document) and `embeddings` (query embedding) queues. Without `parsing`
|
|
# the read_document await never resolves; without `embeddings` every search
|
|
# fails after EMBEDDINGS_DELEGATE_TIMEOUT, because EMBEDDINGS_DELEGATE_TO_WORKER
|
|
# is on by default. For heavy/OCR parsing run a separate worker with `-Q parsing`;
|
|
# to keep query latency off the ingest pool, another with `-Q embeddings`.
|
|
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
|
env_file:
|
|
- ../.env
|
|
environment:
|
|
# Override URLs to use docker service names
|
|
- CELERY_BROKER_URL=redis://redis:6379/0
|
|
- CELERY_RESULT_BACKEND=redis://redis:6379/1
|
|
- API_URL=http://backend:7091
|
|
- CACHE_REDIS_URL=redis://redis:6379/2
|
|
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
|
volumes:
|
|
- ../application/indexes:/app/indexes
|
|
- ../application/inputs:/app/inputs
|
|
- ../application/vectors:/app/vectors
|
|
depends_on:
|
|
redis:
|
|
condition: service_started
|
|
postgres:
|
|
condition: service_healthy
|
|
|
|
redis:
|
|
image: redis:6-alpine
|
|
ports:
|
|
- 6379:6379
|
|
|
|
postgres:
|
|
image: postgres:16-alpine
|
|
environment:
|
|
- POSTGRES_USER=docsgpt
|
|
- POSTGRES_PASSWORD=docsgpt
|
|
- POSTGRES_DB=docsgpt
|
|
ports:
|
|
- "5432:5432"
|
|
volumes:
|
|
- postgres_data:/var/lib/postgresql/data
|
|
healthcheck:
|
|
test: ["CMD-SHELL", "pg_isready -U docsgpt -d docsgpt"]
|
|
interval: 5s
|
|
timeout: 5s
|
|
retries: 10
|
|
|
|
volumes:
|
|
postgres_data:
|