mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 10:13:06 +00:00
Replace the sandbox Docling extractor with read_document, backed by the in-process backend parser (the same one ingestion uses) and offloaded to a dedicated 'parsing' Celery queue so it can run on GPU-capable workers with predictable RAM. The tool resolves the input ref under the run-scoped gate, enqueues the parse, and awaits it with a timeout (degrading to an error rather than hanging); the worker independently re-resolves the artifact through the same gate and never trusts a raw path. Untrusted files get the upload path's safeguards (extension whitelist, size cap, sanitized temp file, cleanup). Options: output (markdown/text/structured/chunks), ocr, pages, engine, max_chars, include_tables, persist, json_schema. The workflow native-file 'extract' fallback now uses the same worker path, so document parsing no longer needs the sandbox and works on every backend. Also fixes the branch's periodic-task test (the sandbox reaper made it 12) and points the dev and e2e Celery workers at the parsing queue.
49 lines
2.0 KiB
Python
49 lines
2.0 KiB
Python
from application.core.settings import settings
|
|
|
|
# Pydantic loads .env into ``settings`` but does not inject values into
|
|
# ``os.environ`` — read directly from settings so beat startup (which
|
|
# imports this module before any explicit env load) sees a real URL.
|
|
broker_url = settings.CELERY_BROKER_URL
|
|
result_backend = settings.CELERY_RESULT_BACKEND
|
|
|
|
task_serializer = 'json'
|
|
result_serializer = 'json'
|
|
accept_content = ['json']
|
|
|
|
# Autodiscover tasks
|
|
imports = ('application.api.user.tasks',)
|
|
|
|
# Project-scoped queue so a stray sibling worker on the same broker
|
|
# (other repo, same default ``celery`` queue) can't grab DocsGPT tasks.
|
|
task_default_queue = "docsgpt"
|
|
task_default_exchange = "docsgpt"
|
|
task_default_routing_key = "docsgpt"
|
|
|
|
# Route document parsing to a dedicated queue so a parse enqueued from inside a
|
|
# Celery worker (headless/scheduled agent) is served by a separate parsing worker
|
|
# and never self-deadlocks the awaiting worker. The tool also passes the queue at
|
|
# apply_async time, so this routing is the default for any other enqueuer.
|
|
task_routes = {
|
|
"application.api.user.tasks.parse_document": {"queue": settings.DOCUMENT_PARSE_QUEUE},
|
|
}
|
|
|
|
beat_scheduler = "redbeat.RedBeatScheduler"
|
|
redbeat_redis_url = broker_url
|
|
redbeat_key_prefix = "redbeat:docsgpt:"
|
|
redbeat_lock_timeout = 90
|
|
|
|
# Survive worker SIGKILL/OOM without silently dropping in-flight tasks.
|
|
task_acks_late = True
|
|
task_reject_on_worker_lost = True
|
|
worker_prefetch_multiplier = settings.CELERY_WORKER_PREFETCH_MULTIPLIER
|
|
broker_transport_options = {"visibility_timeout": settings.CELERY_VISIBILITY_TIMEOUT}
|
|
result_expires = 86400 * 7
|
|
task_track_started = True
|
|
|
|
# Recycle the prefork worker child to bound native-heap growth from
|
|
# docling/torch parsing. Left unset (Celery's unlimited default) when 0.
|
|
if settings.CELERY_WORKER_MAX_MEMORY_PER_CHILD > 0:
|
|
worker_max_memory_per_child = settings.CELERY_WORKER_MAX_MEMORY_PER_CHILD
|
|
if settings.CELERY_WORKER_MAX_TASKS_PER_CHILD > 0:
|
|
worker_max_tasks_per_child = settings.CELERY_WORKER_MAX_TASKS_PER_CHILD
|