Files
DocsGPT/pyproject.toml
T
Alex ae9348bb2a fix(package): review pass on the PyPI package
- docsgpt api binds 127.0.0.1 by default, like gunicorn and uvicorn do;
  --host 0.0.0.0 exposes it. The docs say so.
- The embedded Milvus and LanceDB defaults derive from the data home, so
  they follow DOCSGPT_HOME like the faiss indexes and uploads do. A checkout
  run from its root and the Docker image resolve to the same paths as before.
- The docs and the pyproject comment describe the CPU torch install as two
  steps (torch and torchvision from the PyTorch CPU index first, then the
  docling extra): pip picks the highest version across indexes, so
  --extra-index-url only yields the CPU build while that index keeps pace
  with PyPI.
- AGENTS.md separates DOCSGPT_HOME (moves the data home) from
  DOCSGPT_ENV_FILE (selects the .env file); the docs example uses a password
  placeholder.
2026-09-07 15:25:30 +01:00

211 lines
6.8 KiB
TOML

[build-system]
requires = ["hatchling>=1.27,<2"]
build-backend = "hatchling.build"
[project]
name = "docsgpt"
# The version is read from docsgpt/version.py (the release workflow reads the
# same file); bump it there.
dynamic = ["version"]
description = "DocsGPT backend: chat with your documents, agents, and tools."
readme = "README.md"
requires-python = ">=3.12"
license = "MIT"
license-files = ["LICENSE"]
authors = [{ name = "Arc53" }]
keywords = ["rag", "llm", "agents", "documents", "chat", "search"]
classifiers = [
"Development Status :: 5 - Production/Stable",
"Environment :: Web Environment",
"Framework :: Flask",
"Intended Audience :: Developers",
"Operating System :: OS Independent",
"Programming Language :: Python :: 3",
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.13",
"Topic :: Scientific/Engineering :: Artificial Intelligence",
]
# Direct dependencies only, as compatible ranges: the floor is the version
# uv.lock pins, the ceiling the next major (next minor below 1.0), so the
# PyPI package installs next to other packages. The pip-facing files under
# docsgpt/ (requirements*.txt) are exported from the lock by
# scripts/export_requirements.sh and must not be edited by hand; Docker, CI
# and the checkout install those exact versions.
dependencies = [
"a2wsgi>=1.10.10,<2",
"alembic>=1.13,<2",
"anthropic>=0.121.0,<0.122",
"beautifulsoup4>=4.15.0,<5",
"boto3>=1.43.67,<2",
"cel-python>=0.5.0,<0.6",
"celery>=5.6.3,<6",
"celery-redbeat>=2.4.2,<3",
"croniter>=6.2.4,<7",
"cryptography>=50.0.0,<51",
"dataclasses-json>=0.6.7,<0.7",
"daytona>=0.205.1,<0.206",
"ddgs>=8.0.0,<10",
"defusedxml>=0.7.1,<0.8",
"docx2txt>=0.9,<0.10",
"elevenlabs>=2.62.0,<3",
"faiss-cpu>=1.15.0,<2",
"fast-ebook>=0.2.0,<0.3",
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
# with no model downloads. The docling engine is the `docling` extra.
"firecrawl-anydoc>=0.2.3,<0.3",
"fastembed>=0.8.0,<0.9",
"fastmcp>=3.4.6,<4",
"Flask>=3.1.3,<4",
"flask-restx>=1.3.2,<2",
"google-api-python-client>=2.198.0,<3",
"google-auth-oauthlib>=1.4.0,<2",
"google-genai>=2.17.0,<3",
"gTTS>=2.5.4,<3",
"gunicorn>=26.0.0,<27",
"jinja2>=3.1.6,<4",
"kombu>=5.6.2,<6",
"markdownify>=1.2.3,<2",
"msal>=1.37.0,<2",
"networkx>=3.6.1,<4",
"numpy>=2.5.1,<3",
# fastembed's runtime: local embeddings execute on it.
"onnxruntime>=1.28.0,<2",
"openai>=2.53.0,<3",
"openapi3-parser>=1.1.22,<2",
# pandas reads .xlsx through openpyxl but does not depend on it.
"openpyxl>=3.1.5,<4",
"opentelemetry-distro>=0.50b0,<1",
"opentelemetry-exporter-otlp>=1.29.0,<2",
"opentelemetry-instrumentation-celery>=0.50b0,<1",
"opentelemetry-instrumentation-flask>=0.50b0,<1",
"opentelemetry-instrumentation-logging>=0.50b0,<1",
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
"opentelemetry-instrumentation-redis>=0.50b0,<1",
"opentelemetry-instrumentation-requests>=0.50b0,<1",
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
"pandas>=3.0.5,<4",
"pdf2image>=1.17.0,<2",
"pgvector>=0.5,<1",
"pillow>=12.3.0,<13",
"praw>=8.0.2,<9",
"psycopg[binary,pool]>=3.1,<4",
"pydantic>=2.13.5,<3",
"pydantic-settings>=2.15.0,<3",
"pypdf>=6.15.0,<7",
"pypdfium2>=5.12.1,<6",
"python-dateutil>=2.9.0,<3",
"python-dotenv>=1.2.3,<2",
"python-jose>=3.5.0,<4",
"python-pptx>=1.0.2,<2",
"PyYAML>=6.0.3,<7",
"qdrant-client>=1.19.0,<2",
"redis>=7.4.0,<8",
"requests>=2.34.2,<3",
"retry>=0.9.2,<0.10",
"sqlalchemy>=2.0,<3",
"starlette>=1.0,<2",
"tiktoken>=0.13.0,<0.14",
"tldextract>=5.3.2,<6",
"tokenizers>=0.22.2,<0.23",
"tqdm>=4.67.3,<5",
"uvicorn[standard]>=0.30,<1",
"uvicorn-worker>=0.4,<1",
"websocket-client>=1.9.0,<2",
"werkzeug>=3.1.0,<4",
]
[project.optional-dependencies]
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
# attachment parsing, and read_document's `structured` output. Pulls torch and
# transformers; with uv on Linux torch resolves from the CPU-only PyTorch index
# (see [tool.uv.sources]) so the extra does not drag the CUDA stack in. pip
# users on Linux install torch and torchvision from that index first
# (--index-url https://download.pytorch.org/whl/cpu), then the extra; pip keeps
# the torch it already has.
docling = [
"docling>=2.119.0,<3",
"rapidocr>=3.9.2,<4",
# docling's model stack. Declared here (not left transitive) so the floors
# hold and the CPU index source below applies. transformers is capped by
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
"torch>=2.11.0,<3",
"torchvision>=0.26.0,<0.27",
"transformers>=5.8.1,<5.9",
]
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
milvus = [
"pymilvus>=3.0.1,<4",
"milvus-lite>=3.2.0,<4; sys_platform != 'win32'",
]
[project.scripts]
docsgpt = "docsgpt.cli:main"
[project.urls]
Homepage = "https://www.docsgpt.cloud/"
Documentation = "https://docs.docsgpt.cloud/"
Repository = "https://github.com/arc53/DocsGPT"
Issues = "https://github.com/arc53/DocsGPT/issues"
Changelog = "https://github.com/arc53/DocsGPT/releases"
[dependency-groups]
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
dev = [
"pytest>=8.0.0",
"pytest-asyncio>=0.23",
"pytest-cov>=4.1.0",
"pytest-xdist>=3.5",
"coverage>=7.4.0",
"pytest-postgresql>=6.0.0",
"jupyter-client>=8.0",
"python-docx>=1.1",
"reportlab>=4.0,<5",
"ruff",
]
[tool.hatch.version]
path = "docsgpt/version.py"
# The wheel is the docsgpt package with the data it reads at runtime: prompts,
# model catalogs, the seed config, alembic.ini and the migrations. Build inputs
# (Dockerfile, exported requirements), the sample index and local runtime data
# stay out. The one-release `application` import alias is checkout-only.
[tool.hatch.build.targets.wheel]
packages = ["docsgpt"]
exclude = [
"docsgpt/Dockerfile",
"docsgpt/requirements*.txt",
"docsgpt/index.faiss",
"docsgpt/index.pkl",
"docsgpt/indexes/",
"docsgpt/inputs/",
"docsgpt/vectors/",
]
[tool.hatch.build.targets.sdist]
include = ["/docsgpt"]
exclude = [
"docsgpt/Dockerfile",
"docsgpt/requirements*.txt",
"docsgpt/index.faiss",
"docsgpt/index.pkl",
"docsgpt/indexes/",
"docsgpt/inputs/",
"docsgpt/vectors/",
]
[[tool.uv.index]]
name = "pytorch-cpu"
url = "https://download.pytorch.org/whl/cpu"
explicit = true
[tool.uv.sources]
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
# wheels). The docling extra runs its models on CPU, so take torch from the
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]