mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
`uv lock --upgrade` plus the two code changes the new versions need.
firecrawl-anydoc 0.2.4 raises a dedicated `NeedsOcrError` where 0.2.3 raised
`UnsupportedError("... OCR is required")`, so the anydoc parser no longer
recognised a scanned PDF: the fallback still ran, but a near-empty result was
stored as an empty document instead of failing with the OCR_ENABLED hint.
`_needs_ocr` now accepts both spellings and looks the class up lazily, so an
older anydoc keeps working. 0.2.4 also refuses the CID-font NDA fixture
outright rather than dropping its Chinese column silently, so the PDF
trust-check tests stub that dropped output against the fixture's real bytes
(the check's own inputs) and a new test pins the refusal path.
ruff 0.16 widened its implicit default rule set, turning the dev-group bump
into 7131 findings across the tree. `.ruff.toml` now states the historical
selection (E4, E7, E9, F) explicitly and the CI pin moves to the locked
0.16.7, so lint no longer drifts with the version.
215 lines
7.0 KiB
TOML
215 lines
7.0 KiB
TOML
[build-system]
|
|
requires = ["hatchling>=1.27,<2"]
|
|
build-backend = "hatchling.build"
|
|
|
|
[project]
|
|
name = "docsgpt"
|
|
# The version is read from docsgpt/version.py (the release workflow reads the
|
|
# same file); bump it there.
|
|
dynamic = ["version"]
|
|
description = "DocsGPT backend: chat with your documents, agents, and tools."
|
|
readme = "README.md"
|
|
requires-python = ">=3.12"
|
|
license = "MIT"
|
|
license-files = ["LICENSE"]
|
|
authors = [{ name = "Arc53" }]
|
|
keywords = ["rag", "llm", "agents", "documents", "chat", "search"]
|
|
classifiers = [
|
|
"Development Status :: 5 - Production/Stable",
|
|
"Environment :: Web Environment",
|
|
"Framework :: Flask",
|
|
"Intended Audience :: Developers",
|
|
"Operating System :: OS Independent",
|
|
"Programming Language :: Python :: 3",
|
|
"Programming Language :: Python :: 3.12",
|
|
"Programming Language :: Python :: 3.13",
|
|
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
]
|
|
|
|
# Direct dependencies only, as compatible ranges: the floor is the version
|
|
# uv.lock pins, the ceiling the next major (next minor below 1.0), so the
|
|
# PyPI package installs next to other packages. The pip-facing files under
|
|
# docsgpt/ (requirements*.txt) are exported from the lock by
|
|
# scripts/export_requirements.sh and must not be edited by hand; Docker, CI
|
|
# and the checkout install those exact versions.
|
|
dependencies = [
|
|
"a2wsgi>=1.10.10,<2",
|
|
"alembic>=1.13,<2",
|
|
"anthropic>=0.121.0,<0.122",
|
|
"beautifulsoup4>=4.15.0,<5",
|
|
"boto3>=1.43.67,<2",
|
|
"cel-python>=0.5.0,<0.6",
|
|
"celery>=5.6.3,<6",
|
|
"celery-redbeat>=2.4.2,<3",
|
|
"croniter>=6.2.4,<7",
|
|
"cryptography>=50.0.0,<51",
|
|
"dataclasses-json>=0.6.7,<0.7",
|
|
"daytona>=0.205.1,<0.206",
|
|
"ddgs>=8.0.0,<10",
|
|
"defusedxml>=0.7.1,<0.8",
|
|
"docx2txt>=0.9,<0.10",
|
|
"elevenlabs>=2.62.0,<3",
|
|
"faiss-cpu>=1.15.0,<2",
|
|
"fast-ebook>=0.2.0,<0.3",
|
|
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
|
|
# with no model downloads. The docling engine is the `docling` extra.
|
|
"firecrawl-anydoc>=0.2.4,<0.3",
|
|
"fastembed>=0.8.0,<0.9",
|
|
"fastmcp>=3.4.6,<4",
|
|
"Flask>=3.1.3,<4",
|
|
"flask-restx>=1.3.2,<2",
|
|
"google-api-python-client>=2.198.0,<3",
|
|
"google-auth-oauthlib>=1.4.0,<2",
|
|
"google-genai>=2.17.0,<3",
|
|
"gTTS>=2.5.4,<3",
|
|
"gunicorn>=26.0.0,<27",
|
|
"jinja2>=3.1.6,<4",
|
|
"kombu>=5.6.2,<6",
|
|
"markdownify>=1.2.3,<2",
|
|
"msal>=1.37.0,<2",
|
|
"networkx>=3.6.1,<4",
|
|
"numpy>=2.5.1,<3",
|
|
# fastembed's runtime: local embeddings execute on it.
|
|
"onnxruntime>=1.28.0,<2",
|
|
"openai>=2.53.0,<3",
|
|
"openapi3-parser>=1.1.22,<2",
|
|
# pandas reads .xlsx through openpyxl but does not depend on it.
|
|
"openpyxl>=3.1.5,<4",
|
|
"opentelemetry-distro>=0.50b0,<1",
|
|
"opentelemetry-exporter-otlp>=1.29.0,<2",
|
|
"opentelemetry-instrumentation-celery>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-flask>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-logging>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-redis>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-requests>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
|
|
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
|
|
"pandas>=3.0.5,<4",
|
|
"pdf2image>=1.17.0,<2",
|
|
"pgvector>=0.5,<1",
|
|
"pillow>=12.3.0,<13",
|
|
"praw>=8.0.2,<9",
|
|
"psycopg[binary,pool]>=3.1,<4",
|
|
"pydantic>=2.13.5,<3",
|
|
"pydantic-settings>=2.15.0,<3",
|
|
"pypdf>=6.15.0,<7",
|
|
"pypdfium2>=5.12.1,<6",
|
|
"python-dateutil>=2.9.0,<3",
|
|
"python-dotenv>=1.2.3,<2",
|
|
"python-jose>=3.5.0,<4",
|
|
"python-pptx>=1.0.2,<2",
|
|
"PyYAML>=6.0.3,<7",
|
|
"qdrant-client>=1.19.0,<2",
|
|
"redis>=7.4.0,<8",
|
|
"requests>=2.34.2,<3",
|
|
"retry>=0.9.2,<0.10",
|
|
"sqlalchemy>=2.0,<3",
|
|
"starlette>=1.0,<2",
|
|
"tiktoken>=0.13.0,<0.14",
|
|
"tldextract>=5.3.2,<6",
|
|
"tokenizers>=0.22.2,<0.23",
|
|
"tqdm>=4.67.3,<5",
|
|
"uvicorn[standard]>=0.30,<1",
|
|
"uvicorn-worker>=0.4,<1",
|
|
"websocket-client>=1.9.0,<2",
|
|
"werkzeug>=3.1.0,<4",
|
|
]
|
|
|
|
[project.optional-dependencies]
|
|
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
|
|
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
|
|
# attachment parsing, and read_document's `structured` output. Pulls torch and
|
|
# transformers; with uv on Linux torch resolves from the CPU-only PyTorch index
|
|
# (see [tool.uv.sources]) so the extra does not drag the CUDA stack in. pip
|
|
# users on Linux install torch and torchvision from that index first
|
|
# (--index-url https://download.pytorch.org/whl/cpu), then the extra; pip keeps
|
|
# the torch it already has.
|
|
docling = [
|
|
"docling>=2.119.0,<3",
|
|
"rapidocr>=3.9.2,<4",
|
|
# docling's model stack. Declared here (not left transitive) so the floors
|
|
# hold and the CPU index source below applies. transformers is capped by
|
|
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
|
|
"torch>=2.11.0,<3",
|
|
"torchvision>=0.26.0,<0.27",
|
|
"transformers>=5.8.1,<5.9",
|
|
]
|
|
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
|
|
milvus = [
|
|
"pymilvus>=3.0.1,<4",
|
|
"milvus-lite>=3.2.0,<4; sys_platform != 'win32'",
|
|
]
|
|
|
|
[project.scripts]
|
|
docsgpt = "docsgpt.cli:main"
|
|
|
|
[project.urls]
|
|
Homepage = "https://www.docsgpt.cloud/"
|
|
Documentation = "https://docs.docsgpt.cloud/"
|
|
Repository = "https://github.com/arc53/DocsGPT"
|
|
Issues = "https://github.com/arc53/DocsGPT/issues"
|
|
Changelog = "https://github.com/arc53/DocsGPT/releases"
|
|
|
|
[dependency-groups]
|
|
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
|
|
dev = [
|
|
"pytest>=8.0.0",
|
|
"pytest-asyncio>=0.23",
|
|
"pytest-cov>=4.1.0",
|
|
"pytest-xdist>=3.5",
|
|
"coverage>=7.4.0",
|
|
"pytest-postgresql>=6.0.0",
|
|
"jupyter-client>=8.0",
|
|
"python-docx>=1.1",
|
|
"reportlab>=4.0,<5",
|
|
"ruff",
|
|
]
|
|
|
|
[tool.hatch.version]
|
|
path = "docsgpt/version.py"
|
|
|
|
# The wheel is the docsgpt package with the data it reads at runtime: prompts,
|
|
# model catalogs, the seed config, alembic.ini and the migrations, plus the web
|
|
# UI build under docsgpt/static (gitignored; scripts/build_frontend.sh makes
|
|
# it, and `artifacts` admits it). Build inputs (Dockerfile, exported
|
|
# requirements), the sample index and local runtime data stay out. The
|
|
# one-release `application` import alias is checkout-only.
|
|
[tool.hatch.build.targets.wheel]
|
|
packages = ["docsgpt"]
|
|
artifacts = ["docsgpt/static/**"]
|
|
exclude = [
|
|
"docsgpt/Dockerfile",
|
|
"docsgpt/requirements*.txt",
|
|
"docsgpt/index.faiss",
|
|
"docsgpt/index.pkl",
|
|
"docsgpt/indexes/",
|
|
"docsgpt/inputs/",
|
|
"docsgpt/vectors/",
|
|
]
|
|
|
|
[tool.hatch.build.targets.sdist]
|
|
include = ["/docsgpt"]
|
|
artifacts = ["docsgpt/static/**"]
|
|
exclude = [
|
|
"docsgpt/Dockerfile",
|
|
"docsgpt/requirements*.txt",
|
|
"docsgpt/index.faiss",
|
|
"docsgpt/index.pkl",
|
|
"docsgpt/indexes/",
|
|
"docsgpt/inputs/",
|
|
"docsgpt/vectors/",
|
|
]
|
|
|
|
[[tool.uv.index]]
|
|
name = "pytorch-cpu"
|
|
url = "https://download.pytorch.org/whl/cpu"
|
|
explicit = true
|
|
|
|
[tool.uv.sources]
|
|
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
|
|
# wheels). The docling extra runs its models on CPU, so take torch from the
|
|
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
|
|
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
|
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|