#!/usr/bin/env bash # Regenerate the pip-facing requirements files from uv.lock. # # pyproject.toml declares the direct dependencies and the optional extras; # uv.lock pins everything. pip users, the Dockerfile and CI install from the # exported files, so run this after any change to pyproject.toml or uv.lock: # # uv lock # or: uv lock --upgrade-package # bash scripts/export_requirements.sh # # Each exported file is a complete environment (core plus the named extra), # so `pip install -r docsgpt/requirements-docling.txt` on its own works, # and installing it on top of requirements.txt only adds the extra's packages. set -euo pipefail cd "$(dirname "$0")/.." UV=(uv) if ! uv --version 2>/dev/null | grep -qE '^uv 0\.([89]|[1-9][0-9])\.'; then # `uv export` needs a current uv; run one through uvx without touching the # machine's install. UV=(uv tool run --from 'uv>=0.8' uv) fi export_file() { local out="$1"; shift local header="$1"; shift local index_url="${INDEX_URL:-}" { echo "# GENERATED by scripts/export_requirements.sh from uv.lock -- do not edit." echo "#" # shellcheck disable=SC2001 echo "$header" | sed 's/^/# /' echo # uv export records the wheel's origin in uv.lock but writes no index # directive; pip needs one to find the +cpu torch build. if [ -n "$index_url" ]; then echo "--extra-index-url $index_url"; echo; fi "${UV[@]}" export --frozen --no-hashes --no-dev --no-emit-project --no-header --quiet "$@" } > "$out" echo "wrote $out ($(grep -cE '^[A-Za-z0-9]' "$out") packages)" } export_file docsgpt/requirements.txt \ "Core runtime. Optional extras live in requirements-.txt: docling DOC_PARSER_ENGINE=docling, docling OCR backend, read_document structured output milvus VECTOR_STORE=milvus" INDEX_URL=https://download.pytorch.org/whl/cpu export_file docsgpt/requirements-docling.txt \ "Core runtime plus the docling extra: DOC_PARSER_ENGINE=docling, the docling OCR backend (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml attachment parsing, and read_document's 'structured' output. The default anydoc engine needs none of this, and OCR itself does not either: OCR_ENABLED=true with the tesseract binary (or a DeepSeek-OCR endpoint) runs through docsgpt/parser/file/ocr_parser.py. On Linux torch comes from the CPU-only PyTorch index (no CUDA stack); a GPU deployment can reinstall torch from PyPI on top. pip resolves the extra index as expected. uv only takes a package from the first index that lists it, and the PyTorch index carries stale copies of common packages, so with uv either run 'uv sync --extra docling' (the lock pins the index per package) or set UV_INDEX_STRATEGY=unsafe-best-match. Docker: --build-arg EXTRAS=docling" \ --extra docling export_file docsgpt/requirements-milvus.txt \ "Core runtime plus the milvus extra (VECTOR_STORE=milvus): pymilvus and the embedded milvus-lite server, which pulls pyarrow. Docker: --build-arg EXTRAS=milvus" \ --extra milvus