mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
requirements.txt pinned torch and transformers in core although only docling needs them, and on Linux torch pulls the CUDA 13 stack: 2.7 GB of the 3.0 GB wheel download. Direct dependencies now live in pyproject.toml, uv.lock pins everything, and application/requirements*.txt are exported from the lock by scripts/export_requirements.sh (each file is the core set plus one extra). The docling extra pins torch/torchvision/transformers itself and, on Linux, resolves torch from the CPU-only PyTorch index (no nvidia packages). milvus (pymilvus + milvus-lite, which pulls pyarrow) is the second extra. application/core/optional_deps.py is the one place install hints come from; the milvus store and the docling call sites use it so a missing extra fails with the exact command to run.
150 lines
4.6 KiB
TOML
150 lines
4.6 KiB
TOML
[project]
|
|
name = "docsgpt"
|
|
# Bump together with application/version.py and frontend/package.json.
|
|
version = "0.19.0"
|
|
description = "DocsGPT backend: chat with your documents, agents, and tools."
|
|
readme = "README.md"
|
|
requires-python = ">=3.12"
|
|
license = { file = "LICENSE" }
|
|
|
|
# Direct dependencies only. Transitive pins live in uv.lock; the pip-facing
|
|
# files under application/ (requirements*.txt) are exported from that lock by
|
|
# scripts/export_requirements.sh and must not be edited by hand.
|
|
dependencies = [
|
|
"a2wsgi==1.10.10",
|
|
"alembic>=1.13,<2",
|
|
"anthropic==0.121.0",
|
|
"beautifulsoup4==4.15.0",
|
|
"boto3==1.43.67",
|
|
"cel-python==0.5.0",
|
|
"celery==5.6.3",
|
|
"celery-redbeat==2.4.2",
|
|
"croniter==6.2.4",
|
|
"cryptography==50.0.0",
|
|
"dataclasses-json==0.6.7",
|
|
"daytona==0.205.1",
|
|
"ddgs>=8.0.0",
|
|
"defusedxml==0.7.1",
|
|
"docx2txt==0.9",
|
|
"elevenlabs==2.62.0",
|
|
"faiss-cpu==1.15.0",
|
|
"fast-ebook",
|
|
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
|
|
# with no model downloads. The docling engine is the `docling` extra.
|
|
"firecrawl-anydoc==0.2.3",
|
|
"fastembed==0.8.0",
|
|
"fastmcp==3.4.6",
|
|
"Flask==3.1.3",
|
|
"flask-restx==1.3.2",
|
|
"google-api-python-client==2.198.0",
|
|
"google-auth-oauthlib==1.4.0",
|
|
"google-genai==2.17.0",
|
|
"gTTS==2.5.4",
|
|
"gunicorn==26.0.0",
|
|
"jinja2==3.1.6",
|
|
"kombu==5.6.2",
|
|
"markdownify==1.2.3",
|
|
"msal==1.37.0",
|
|
"networkx==3.6.1",
|
|
"numpy==2.5.1",
|
|
# fastembed's runtime: local embeddings execute on it.
|
|
"onnxruntime==1.28.0",
|
|
"openai==2.53.0",
|
|
"openapi3-parser==1.1.22",
|
|
# pandas reads .xlsx through openpyxl but does not depend on it.
|
|
"openpyxl==3.1.5",
|
|
"opentelemetry-distro>=0.50b0,<1",
|
|
"opentelemetry-exporter-otlp>=1.29.0,<2",
|
|
"opentelemetry-instrumentation-celery>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-flask>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-logging>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-redis>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-requests>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
|
|
"pandas==3.0.5",
|
|
"pdf2image>=1.17.0",
|
|
"pgvector>=0.5,<1",
|
|
"pillow==12.3.0",
|
|
"praw==8.0.2",
|
|
"psycopg[binary,pool]>=3.1,<4",
|
|
"pydantic",
|
|
"pydantic-settings",
|
|
"pypdf==6.15.0",
|
|
"pypdfium2==5.12.1",
|
|
"python-dateutil==2.9.0.post0",
|
|
"python-dotenv",
|
|
"python-jose==3.5.0",
|
|
"python-pptx==1.0.2",
|
|
"PyYAML",
|
|
"qdrant-client==1.19.0",
|
|
"redis==7.4.0",
|
|
"requests==2.34.2",
|
|
"retry==0.9.2",
|
|
"sqlalchemy>=2.0,<3",
|
|
"starlette>=1.0,<2",
|
|
"tiktoken==0.13.0",
|
|
"tldextract==5.3.2",
|
|
"tokenizers==0.22.2",
|
|
"tqdm==4.67.3",
|
|
"uvicorn[standard]>=0.30,<1",
|
|
"uvicorn-worker>=0.4,<1",
|
|
"websocket-client==1.9.0",
|
|
"werkzeug>=3.1.0",
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
|
|
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
|
|
# attachment parsing, and read_document's `structured` output. Pulls torch and
|
|
# transformers; on Linux torch resolves from the CPU-only PyTorch index (see
|
|
# [tool.uv.sources]) so the extra does not drag the CUDA stack in.
|
|
docling = [
|
|
"docling==2.119.0",
|
|
"rapidocr==3.9.2",
|
|
# docling's model stack. Declared here (not left transitive) so the pins
|
|
# hold and the CPU index source below applies. transformers is capped by
|
|
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
|
|
"torch==2.11.0",
|
|
"torchvision==0.26.0",
|
|
"transformers==5.8.1",
|
|
]
|
|
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
|
|
milvus = [
|
|
"pymilvus==3.0.1",
|
|
"milvus-lite==3.2.0; sys_platform != 'win32'",
|
|
]
|
|
|
|
[dependency-groups]
|
|
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
|
|
dev = [
|
|
"pytest>=8.0.0",
|
|
"pytest-asyncio>=0.23",
|
|
"pytest-cov>=4.1.0",
|
|
"pytest-xdist>=3.5",
|
|
"coverage>=7.4.0",
|
|
"pytest-postgresql>=6.0.0",
|
|
"jupyter-client>=8.0",
|
|
"python-docx>=1.1",
|
|
"reportlab>=4.0,<5",
|
|
"ruff",
|
|
]
|
|
|
|
[tool.uv]
|
|
# The repo is run in place (`application.*` imported from the checkout), not
|
|
# installed as a distribution.
|
|
package = false
|
|
|
|
[[tool.uv.index]]
|
|
name = "pytorch-cpu"
|
|
url = "https://download.pytorch.org/whl/cpu"
|
|
explicit = true
|
|
|
|
[tool.uv.sources]
|
|
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
|
|
# wheels). The docling extra runs its models on CPU, so take torch from the
|
|
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
|
|
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
|
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|