Files
DocsGPT/pyproject.toml
T
Alex 574f96341e refactor: rename the application package to docsgpt
The backend import package is now docsgpt, the name it will carry on PyPI;
application was far too generic to install into anyone's site-packages.
git mv plus a mechanical rewrite of every import, dotted string and path
reference: 734 Python files, the compose files, Dockerfile, workflows, docs,
setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage
config, .gitignore. Behaviour is unchanged.

Kept for one release:
- A top-level application package whose meta-path finder resolves
  application.x.y to the already-imported docsgpt.x.y object, so old imports
  and entry points (celery -A application.app.celery,
  uvicorn application.asgi:asgi_app) keep working with a FutureWarning.
- Celery registers every application.* task name as an alias of its
  docsgpt.* task on start-up, so messages queued by the previous release still
  run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries
  the previous release wrote are left unread instead of firing twice.

The backend image builds from the repository root (docker build -f
docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore
allow-lists docsgpt/ and application/ and keeps caches, local data, .env
files, the sample index files and the Dockerfile out. Compose and the image
workflows point at the new context.
2026-09-07 10:20:43 +01:00

150 lines
4.6 KiB
TOML

[project]
name = "docsgpt"
# Bump together with docsgpt/version.py and frontend/package.json.
version = "0.19.0"
description = "DocsGPT backend: chat with your documents, agents, and tools."
readme = "README.md"
requires-python = ">=3.12"
license = { file = "LICENSE" }
# Direct dependencies only. Transitive pins live in uv.lock; the pip-facing
# files under docsgpt/ (requirements*.txt) are exported from that lock by
# scripts/export_requirements.sh and must not be edited by hand.
dependencies = [
"a2wsgi==1.10.10",
"alembic>=1.13,<2",
"anthropic==0.121.0",
"beautifulsoup4==4.15.0",
"boto3==1.43.67",
"cel-python==0.5.0",
"celery==5.6.3",
"celery-redbeat==2.4.2",
"croniter==6.2.4",
"cryptography==50.0.0",
"dataclasses-json==0.6.7",
"daytona==0.205.1",
"ddgs>=8.0.0",
"defusedxml==0.7.1",
"docx2txt==0.9",
"elevenlabs==2.62.0",
"faiss-cpu==1.15.0",
"fast-ebook",
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
# with no model downloads. The docling engine is the `docling` extra.
"firecrawl-anydoc==0.2.3",
"fastembed==0.8.0",
"fastmcp==3.4.6",
"Flask==3.1.3",
"flask-restx==1.3.2",
"google-api-python-client==2.198.0",
"google-auth-oauthlib==1.4.0",
"google-genai==2.17.0",
"gTTS==2.5.4",
"gunicorn==26.0.0",
"jinja2==3.1.6",
"kombu==5.6.2",
"markdownify==1.2.3",
"msal==1.37.0",
"networkx==3.6.1",
"numpy==2.5.1",
# fastembed's runtime: local embeddings execute on it.
"onnxruntime==1.28.0",
"openai==2.53.0",
"openapi3-parser==1.1.22",
# pandas reads .xlsx through openpyxl but does not depend on it.
"openpyxl==3.1.5",
"opentelemetry-distro>=0.50b0,<1",
"opentelemetry-exporter-otlp>=1.29.0,<2",
"opentelemetry-instrumentation-celery>=0.50b0,<1",
"opentelemetry-instrumentation-flask>=0.50b0,<1",
"opentelemetry-instrumentation-logging>=0.50b0,<1",
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
"opentelemetry-instrumentation-redis>=0.50b0,<1",
"opentelemetry-instrumentation-requests>=0.50b0,<1",
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
"pandas==3.0.5",
"pdf2image>=1.17.0",
"pgvector>=0.5,<1",
"pillow==12.3.0",
"praw==8.0.2",
"psycopg[binary,pool]>=3.1,<4",
"pydantic",
"pydantic-settings",
"pypdf==6.15.0",
"pypdfium2==5.12.1",
"python-dateutil==2.9.0.post0",
"python-dotenv",
"python-jose==3.5.0",
"python-pptx==1.0.2",
"PyYAML",
"qdrant-client==1.19.0",
"redis==7.4.0",
"requests==2.34.2",
"retry==0.9.2",
"sqlalchemy>=2.0,<3",
"starlette>=1.0,<2",
"tiktoken==0.13.0",
"tldextract==5.3.2",
"tokenizers==0.22.2",
"tqdm==4.67.3",
"uvicorn[standard]>=0.30,<1",
"uvicorn-worker>=0.4,<1",
"websocket-client==1.9.0",
"werkzeug>=3.1.0",
]
[project.optional-dependencies]
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
# attachment parsing, and read_document's `structured` output. Pulls torch and
# transformers; on Linux torch resolves from the CPU-only PyTorch index (see
# [tool.uv.sources]) so the extra does not drag the CUDA stack in.
docling = [
"docling==2.119.0",
"rapidocr==3.9.2",
# docling's model stack. Declared here (not left transitive) so the pins
# hold and the CPU index source below applies. transformers is capped by
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
"torch==2.11.0",
"torchvision==0.26.0",
"transformers==5.8.1",
]
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
milvus = [
"pymilvus==3.0.1",
"milvus-lite==3.2.0; sys_platform != 'win32'",
]
[dependency-groups]
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
dev = [
"pytest>=8.0.0",
"pytest-asyncio>=0.23",
"pytest-cov>=4.1.0",
"pytest-xdist>=3.5",
"coverage>=7.4.0",
"pytest-postgresql>=6.0.0",
"jupyter-client>=8.0",
"python-docx>=1.1",
"reportlab>=4.0,<5",
"ruff",
]
[tool.uv]
# The repo is run in place (`application.*` imported from the checkout), not
# installed as a distribution.
package = false
[[tool.uv.index]]
name = "pytorch-cpu"
url = "https://download.pytorch.org/whl/cpu"
explicit = true
[tool.uv.sources]
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
# wheels). The docling extra runs its models on CPU, so take torch from the
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]