mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
150 lines
4.6 KiB
TOML
150 lines
4.6 KiB
TOML
[project]
|
|
name = "docsgpt"
|
|
# Bump together with docsgpt/version.py and frontend/package.json.
|
|
version = "0.19.0"
|
|
description = "DocsGPT backend: chat with your documents, agents, and tools."
|
|
readme = "README.md"
|
|
requires-python = ">=3.12"
|
|
license = { file = "LICENSE" }
|
|
|
|
# Direct dependencies only. Transitive pins live in uv.lock; the pip-facing
|
|
# files under docsgpt/ (requirements*.txt) are exported from that lock by
|
|
# scripts/export_requirements.sh and must not be edited by hand.
|
|
dependencies = [
|
|
"a2wsgi==1.10.10",
|
|
"alembic>=1.13,<2",
|
|
"anthropic==0.121.0",
|
|
"beautifulsoup4==4.15.0",
|
|
"boto3==1.43.67",
|
|
"cel-python==0.5.0",
|
|
"celery==5.6.3",
|
|
"celery-redbeat==2.4.2",
|
|
"croniter==6.2.4",
|
|
"cryptography==50.0.0",
|
|
"dataclasses-json==0.6.7",
|
|
"daytona==0.205.1",
|
|
"ddgs>=8.0.0",
|
|
"defusedxml==0.7.1",
|
|
"docx2txt==0.9",
|
|
"elevenlabs==2.62.0",
|
|
"faiss-cpu==1.15.0",
|
|
"fast-ebook",
|
|
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
|
|
# with no model downloads. The docling engine is the `docling` extra.
|
|
"firecrawl-anydoc==0.2.3",
|
|
"fastembed==0.8.0",
|
|
"fastmcp==3.4.6",
|
|
"Flask==3.1.3",
|
|
"flask-restx==1.3.2",
|
|
"google-api-python-client==2.198.0",
|
|
"google-auth-oauthlib==1.4.0",
|
|
"google-genai==2.17.0",
|
|
"gTTS==2.5.4",
|
|
"gunicorn==26.0.0",
|
|
"jinja2==3.1.6",
|
|
"kombu==5.6.2",
|
|
"markdownify==1.2.3",
|
|
"msal==1.37.0",
|
|
"networkx==3.6.1",
|
|
"numpy==2.5.1",
|
|
# fastembed's runtime: local embeddings execute on it.
|
|
"onnxruntime==1.28.0",
|
|
"openai==2.53.0",
|
|
"openapi3-parser==1.1.22",
|
|
# pandas reads .xlsx through openpyxl but does not depend on it.
|
|
"openpyxl==3.1.5",
|
|
"opentelemetry-distro>=0.50b0,<1",
|
|
"opentelemetry-exporter-otlp>=1.29.0,<2",
|
|
"opentelemetry-instrumentation-celery>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-flask>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-logging>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-redis>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-requests>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
|
|
"pandas==3.0.5",
|
|
"pdf2image>=1.17.0",
|
|
"pgvector>=0.5,<1",
|
|
"pillow==12.3.0",
|
|
"praw==8.0.2",
|
|
"psycopg[binary,pool]>=3.1,<4",
|
|
"pydantic",
|
|
"pydantic-settings",
|
|
"pypdf==6.15.0",
|
|
"pypdfium2==5.12.1",
|
|
"python-dateutil==2.9.0.post0",
|
|
"python-dotenv",
|
|
"python-jose==3.5.0",
|
|
"python-pptx==1.0.2",
|
|
"PyYAML",
|
|
"qdrant-client==1.19.0",
|
|
"redis==7.4.0",
|
|
"requests==2.34.2",
|
|
"retry==0.9.2",
|
|
"sqlalchemy>=2.0,<3",
|
|
"starlette>=1.0,<2",
|
|
"tiktoken==0.13.0",
|
|
"tldextract==5.3.2",
|
|
"tokenizers==0.22.2",
|
|
"tqdm==4.67.3",
|
|
"uvicorn[standard]>=0.30,<1",
|
|
"uvicorn-worker>=0.4,<1",
|
|
"websocket-client==1.9.0",
|
|
"werkzeug>=3.1.0",
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
|
|
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
|
|
# attachment parsing, and read_document's `structured` output. Pulls torch and
|
|
# transformers; on Linux torch resolves from the CPU-only PyTorch index (see
|
|
# [tool.uv.sources]) so the extra does not drag the CUDA stack in.
|
|
docling = [
|
|
"docling==2.119.0",
|
|
"rapidocr==3.9.2",
|
|
# docling's model stack. Declared here (not left transitive) so the pins
|
|
# hold and the CPU index source below applies. transformers is capped by
|
|
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
|
|
"torch==2.11.0",
|
|
"torchvision==0.26.0",
|
|
"transformers==5.8.1",
|
|
]
|
|
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
|
|
milvus = [
|
|
"pymilvus==3.0.1",
|
|
"milvus-lite==3.2.0; sys_platform != 'win32'",
|
|
]
|
|
|
|
[dependency-groups]
|
|
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
|
|
dev = [
|
|
"pytest>=8.0.0",
|
|
"pytest-asyncio>=0.23",
|
|
"pytest-cov>=4.1.0",
|
|
"pytest-xdist>=3.5",
|
|
"coverage>=7.4.0",
|
|
"pytest-postgresql>=6.0.0",
|
|
"jupyter-client>=8.0",
|
|
"python-docx>=1.1",
|
|
"reportlab>=4.0,<5",
|
|
"ruff",
|
|
]
|
|
|
|
[tool.uv]
|
|
# The repo is run in place (`application.*` imported from the checkout), not
|
|
# installed as a distribution.
|
|
package = false
|
|
|
|
[[tool.uv.index]]
|
|
name = "pytorch-cpu"
|
|
url = "https://download.pytorch.org/whl/cpu"
|
|
explicit = true
|
|
|
|
[tool.uv.sources]
|
|
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
|
|
# wheels). The docling extra runs its models on CPU, so take torch from the
|
|
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
|
|
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
|
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|