mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
Merge pull request #2729 from arc53/feat/slim-image-optional-deps
Slim default install and images: extras for docling/milvus, self-contained images, -docling variant, static frontend
This commit is contained in:
55 files changed
+11769
-411
No files matched your search
+80
-42
@@ -7,106 +7,144 @@ on:
|
||||
jobs:
|
||||
build:
|
||||
if: github.repository == 'arc53/DocsGPT'
|
||||
# Publishing jobs run in a GitHub Actions environment so the registry
|
||||
# credentials can be scoped to it and protection rules (required reviewers,
|
||||
# branch restrictions) applied in the repository settings.
|
||||
environment: docker-hub
|
||||
env:
|
||||
# Public namespace the compose files pull from; the login secret only
|
||||
# authenticates the push.
|
||||
DOCKERHUB_NAMESPACE: arc53
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- platform: linux/amd64
|
||||
runner: ubuntu-latest
|
||||
suffix: amd64
|
||||
- platform: linux/arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
suffix: arm64
|
||||
runs-on: ${{ matrix.runner }}
|
||||
platform: [linux/amd64, linux/arm64]
|
||||
# "" is the slim default image; "-docling" bakes the docling parser
|
||||
# engine, its models and tesseract in (OCR-ready).
|
||||
variant: ["", "-docling"]
|
||||
runs-on: ${{ matrix.platform == 'linux/arm64' && 'ubuntu-24.04-arm' || 'ubuntu-latest' }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up QEMU # Only needed for emulation, not for native arm64 builds
|
||||
if: matrix.platform == 'linux/arm64'
|
||||
uses: docker/setup-qemu-action@v3
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
with:
|
||||
driver: docker-container
|
||||
install: true
|
||||
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to ghcr.io
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Image metadata (OCI labels)
|
||||
id: meta
|
||||
uses: docker/metadata-action@c299e40c65443455700f0fdfc63efafe5b349051 # v5.10.0
|
||||
with:
|
||||
images: |
|
||||
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt
|
||||
ghcr.io/${{ github.repository_owner }}/docsgpt
|
||||
labels: |
|
||||
org.opencontainers.image.title=DocsGPT${{ matrix.variant }}
|
||||
org.opencontainers.image.version=${{ github.event.release.tag_name }}
|
||||
|
||||
- name: Build and push platform-specific images
|
||||
uses: docker/build-push-action@v6
|
||||
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
|
||||
with:
|
||||
file: './application/Dockerfile'
|
||||
platforms: ${{ matrix.platform }}
|
||||
context: ./application
|
||||
push: true
|
||||
build-args: |
|
||||
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
|
||||
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
|
||||
tags: |
|
||||
${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-${{ matrix.suffix }}
|
||||
ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-${{ matrix.suffix }}
|
||||
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:${{ github.event.release.tag_name }}${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
|
||||
ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
provenance: false
|
||||
sbom: false
|
||||
cache-from: type=registry,ref=${{ secrets.DOCKER_USERNAME }}/docsgpt:latest
|
||||
cache-from: type=registry,ref=${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:latest${{ matrix.variant }}
|
||||
cache-to: type=inline
|
||||
|
||||
manifest:
|
||||
if: github.repository == 'arc53/DocsGPT'
|
||||
# Publishing jobs run in a GitHub Actions environment so the registry
|
||||
# credentials can be scoped to it and protection rules (required reviewers,
|
||||
# branch restrictions) applied in the repository settings.
|
||||
environment: docker-hub
|
||||
env:
|
||||
# Public namespace the compose files pull from; the login secret only
|
||||
# authenticates the push.
|
||||
DOCKERHUB_NAMESPACE: arc53
|
||||
needs: build
|
||||
strategy:
|
||||
matrix:
|
||||
variant: ["", "-docling"]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
packages: write
|
||||
steps:
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
with:
|
||||
driver: docker-container
|
||||
install: true
|
||||
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to ghcr.io
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Create and push manifest for DockerHub
|
||||
- name: Create and push multi-arch manifests
|
||||
env:
|
||||
TAG: ${{ github.event.release.tag_name }}${{ matrix.variant }}
|
||||
LATEST: latest${{ matrix.variant }}
|
||||
run: |
|
||||
set -e
|
||||
docker manifest create ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }} \
|
||||
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
|
||||
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-arm64
|
||||
docker manifest push ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}
|
||||
docker manifest create ${{ secrets.DOCKER_USERNAME }}/docsgpt:latest \
|
||||
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
|
||||
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-arm64
|
||||
docker manifest push ${{ secrets.DOCKER_USERNAME }}/docsgpt:latest
|
||||
for repo in "$DOCKERHUB_NAMESPACE/docsgpt" "ghcr.io/${{ github.repository_owner }}/docsgpt"; do
|
||||
for name in "$TAG" "$LATEST"; do
|
||||
docker manifest create "$repo:$name" \
|
||||
--amend "$repo:$TAG-amd64" \
|
||||
--amend "$repo:$TAG-arm64"
|
||||
docker manifest push "$repo:$name"
|
||||
done
|
||||
done
|
||||
|
||||
- name: Create and push manifest for ghcr.io
|
||||
release-assets:
|
||||
if: github.repository == 'arc53/DocsGPT'
|
||||
needs: manifest
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Attach the standalone compose file to the release
|
||||
env:
|
||||
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
TAG: ${{ github.event.release.tag_name }}
|
||||
run: |
|
||||
set -e
|
||||
docker manifest create ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }} \
|
||||
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
|
||||
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-arm64
|
||||
docker manifest push ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}
|
||||
docker manifest create ghcr.io/${{ github.repository_owner }}/docsgpt:latest \
|
||||
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
|
||||
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-arm64
|
||||
docker manifest push ghcr.io/${{ github.repository_owner }}/docsgpt:latest
|
||||
gh release upload "$TAG" deployment/docker-compose-standalone.yaml --clobber
|
||||
@@ -9,92 +9,123 @@ on:
|
||||
jobs:
|
||||
build:
|
||||
if: github.repository == 'arc53/DocsGPT'
|
||||
# Publishing jobs run in a GitHub Actions environment so the registry
|
||||
# credentials can be scoped to it and protection rules (required reviewers,
|
||||
# branch restrictions) applied in the repository settings.
|
||||
environment: docker-hub
|
||||
env:
|
||||
# Public namespace the compose files pull from; the login secret only
|
||||
# authenticates the push.
|
||||
DOCKERHUB_NAMESPACE: arc53
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- platform: linux/amd64
|
||||
runner: ubuntu-latest
|
||||
suffix: amd64
|
||||
- platform: linux/arm64
|
||||
runner: ubuntu-24.04-arm
|
||||
suffix: arm64
|
||||
runs-on: ${{ matrix.runner }}
|
||||
platform: [linux/amd64, linux/arm64]
|
||||
# "" is the slim default image; "-docling" bakes the docling parser
|
||||
# engine, its models and tesseract in (OCR-ready).
|
||||
variant: ["", "-docling"]
|
||||
runs-on: ${{ matrix.platform == 'linux/arm64' && 'ubuntu-24.04-arm' || 'ubuntu-latest' }}
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
with:
|
||||
driver: docker-container
|
||||
install: true
|
||||
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
|
||||
- name: Login to ghcr.io
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Image metadata (OCI labels)
|
||||
id: meta
|
||||
uses: docker/metadata-action@c299e40c65443455700f0fdfc63efafe5b349051 # v5.10.0
|
||||
with:
|
||||
images: |
|
||||
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt
|
||||
ghcr.io/${{ github.repository_owner }}/docsgpt
|
||||
labels: |
|
||||
org.opencontainers.image.title=DocsGPT${{ matrix.variant }}
|
||||
org.opencontainers.image.version=develop
|
||||
|
||||
- name: Build and push platform-specific images
|
||||
uses: docker/build-push-action@v6
|
||||
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
|
||||
with:
|
||||
file: './application/Dockerfile'
|
||||
platforms: ${{ matrix.platform }}
|
||||
context: ./application
|
||||
push: true
|
||||
build-args: |
|
||||
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
|
||||
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
|
||||
tags: |
|
||||
${{ secrets.DOCKER_USERNAME }}/docsgpt:develop-${{ matrix.suffix }}
|
||||
ghcr.io/${{ github.repository_owner }}/docsgpt:develop-${{ matrix.suffix }}
|
||||
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:develop${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
|
||||
ghcr.io/${{ github.repository_owner }}/docsgpt:develop${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
|
||||
labels: ${{ steps.meta.outputs.labels }}
|
||||
provenance: false
|
||||
sbom: false
|
||||
cache-from: type=registry,ref=${{ secrets.DOCKER_USERNAME }}/docsgpt:develop
|
||||
cache-from: type=registry,ref=${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:develop${{ matrix.variant }}
|
||||
cache-to: type=inline
|
||||
|
||||
manifest:
|
||||
if: github.repository == 'arc53/DocsGPT'
|
||||
# Publishing jobs run in a GitHub Actions environment so the registry
|
||||
# credentials can be scoped to it and protection rules (required reviewers,
|
||||
# branch restrictions) applied in the repository settings.
|
||||
environment: docker-hub
|
||||
env:
|
||||
# Public namespace the compose files pull from; the login secret only
|
||||
# authenticates the push.
|
||||
DOCKERHUB_NAMESPACE: arc53
|
||||
needs: build
|
||||
strategy:
|
||||
matrix:
|
||||
variant: ["", "-docling"]
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
packages: write
|
||||
steps:
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
with:
|
||||
driver: docker-container
|
||||
install: true
|
||||
|
||||
- name: Login to DockerHub
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
|
||||
- name: Login to ghcr.io
|
||||
uses: docker/login-action@v3
|
||||
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ github.repository_owner }}
|
||||
password: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Create and push manifest for DockerHub
|
||||
run: |
|
||||
docker manifest create ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop \
|
||||
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop-amd64 \
|
||||
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop-arm64
|
||||
docker manifest push ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop
|
||||
|
||||
- name: Create and push manifest for ghcr.io
|
||||
- name: Create and push multi-arch manifests
|
||||
env:
|
||||
TAG: develop${{ matrix.variant }}
|
||||
run: |
|
||||
docker manifest create ghcr.io/${{ github.repository_owner }}/docsgpt:develop \
|
||||
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:develop-amd64 \
|
||||
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:develop-arm64
|
||||
docker manifest push ghcr.io/${{ github.repository_owner }}/docsgpt:develop
|
||||
set -e
|
||||
for repo in "$DOCKERHUB_NAMESPACE/docsgpt" "ghcr.io/${{ github.repository_owner }}/docsgpt"; do
|
||||
docker manifest create "$repo:$TAG" \
|
||||
--amend "$repo:$TAG-amd64" \
|
||||
--amend "$repo:$TAG-arm64"
|
||||
docker manifest push "$repo:$TAG"
|
||||
done
|
||||
@@ -0,0 +1,66 @@
|
||||
name: Verify the Docker image works offline
|
||||
|
||||
# Builds the backend image and runs its offline check with networking off, so
|
||||
# a change that reintroduces a first-request download (a tokenizer, tiktoken's
|
||||
# encoding, an embedding model) fails here instead of in an air-gapped install.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
paths:
|
||||
- 'application/Dockerfile'
|
||||
- 'application/.dockerignore'
|
||||
- 'application/requirements*.txt'
|
||||
- 'application/scripts/prefetch_models.py'
|
||||
- 'application/scripts/verify_offline.py'
|
||||
- 'application/vectorstore/model_registry.py'
|
||||
- 'application/parser/tokenization.py'
|
||||
- 'application/vectorstore/embeddings_local.py'
|
||||
- '.github/workflows/docker-image-verify.yml'
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
verify:
|
||||
strategy:
|
||||
matrix:
|
||||
# "" is the slim default; "-docling" bakes docling, its models and
|
||||
# tesseract in, so the conversion check in verify_offline runs too.
|
||||
variant: ["", "-docling"]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
|
||||
|
||||
- name: Build the image
|
||||
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
|
||||
with:
|
||||
file: ./application/Dockerfile
|
||||
context: ./application
|
||||
platforms: linux/amd64
|
||||
load: true
|
||||
tags: docsgpt:verify${{ matrix.variant }}
|
||||
build-args: |
|
||||
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
|
||||
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
|
||||
cache-from: type=gha,scope=verify${{ matrix.variant }}
|
||||
cache-to: type=gha,mode=max,scope=verify${{ matrix.variant }}
|
||||
|
||||
- name: Image size
|
||||
env:
|
||||
IMAGE: docsgpt:verify${{ matrix.variant }}
|
||||
run: |
|
||||
docker image inspect "$IMAGE" --format '{{.Size}}' | awk '{printf "uncompressed: %.2f GB\n", $1/1e9}'
|
||||
docker history "$IMAGE" --format '{{.Size}}\t{{.CreatedBy}}' | head -20
|
||||
|
||||
- name: Offline verification (no network)
|
||||
env:
|
||||
IMAGE: docsgpt:verify${{ matrix.variant }}
|
||||
run: |
|
||||
docker run --rm --network none "$IMAGE" \
|
||||
python -m application.scripts.verify_offline
|
||||
@@ -20,3 +20,20 @@ jobs:
|
||||
uses: chartboost/ruff-action@v1
|
||||
with:
|
||||
version: 0.14.10
|
||||
|
||||
requirements-in-sync:
|
||||
# application/requirements*.txt are exported from uv.lock; fail when a
|
||||
# change to pyproject.toml or uv.lock was not re-exported.
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e # v6.8.0
|
||||
|
||||
- name: Re-export and diff
|
||||
run: |
|
||||
uv lock --check
|
||||
bash scripts/export_requirements.sh
|
||||
git diff --exit-code -- application/requirements.txt application/requirements-docling.txt application/requirements-milvus.txt
|
||||
@@ -29,10 +29,20 @@ Use these commands once the dev prerequisites above are satisfied.
|
||||
```bash
|
||||
source .venv/bin/activate # macOS/Linux
|
||||
uv pip install -r application/requirements.txt # or: pip install -r application/requirements.txt
|
||||
# Optional docling parser engine (OCR / read_document structured output):
|
||||
# uv pip install -r application/requirements-docling.txt
|
||||
# Optional extras (not installed by default; each file = core + the extra):
|
||||
# uv pip install -r application/requirements-docling.txt # docling parser engine (OCR backend, structured output)
|
||||
# uv pip install -r application/requirements-milvus.txt # VECTOR_STORE=milvus
|
||||
# With uv alone: `uv sync --extra docling` (pyproject.toml + uv.lock are the source of truth).
|
||||
# `uv pip install -r application/requirements-docling.txt` needs UV_INDEX_STRATEGY=unsafe-best-match
|
||||
# (the file adds the PyTorch CPU index; prefer `uv sync --extra docling`).
|
||||
```
|
||||
|
||||
Dependencies are declared in `pyproject.toml` and locked in `uv.lock`; the
|
||||
`application/requirements*.txt` files are exported from the lock. To add or
|
||||
bump a package: edit `pyproject.toml`, run `uv lock`, then
|
||||
`bash scripts/export_requirements.sh` (CI fails if the exports are stale).
|
||||
Never edit the requirements files by hand.
|
||||
|
||||
Run the API. For local dev, prefer the ASGI entrypoint under uvicorn — it
|
||||
serves the **whole** app, matches production, and hot-reloads:
|
||||
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
# Build context is application/. Keep local state and caches out of the image.
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
.pytest_cache/
|
||||
.ruff_cache/
|
||||
.coverage
|
||||
htmlcov/
|
||||
*.log
|
||||
|
||||
# Runtime data: bind-mounted or created at run time, never baked in.
|
||||
indexes/
|
||||
inputs/
|
||||
vectors/
|
||||
*.faiss
|
||||
*.pkl
|
||||
|
||||
# Secrets and local config.
|
||||
.env
|
||||
.env.*
|
||||
|
||||
# Not needed inside the image.
|
||||
Dockerfile
|
||||
.dockerignore
|
||||
+126
-103
@@ -1,74 +1,81 @@
|
||||
# Builder Stage
|
||||
FROM ubuntu:24.04 as builder
|
||||
# DocsGPT backend image.
|
||||
#
|
||||
# Build args:
|
||||
# EXTRAS comma-separated optional extras to bake in, matching the
|
||||
# pyproject extras / requirements-<extra>.txt files:
|
||||
# docling (layout-model parser + OCR backend), milvus.
|
||||
# INSTALL_DOCLING legacy alias for EXTRAS=docling (setup.sh writes it).
|
||||
# INSTALL_TESSERACT bake the tesseract binary for OCR_ENGINE=tesseract.
|
||||
# EMBEDDINGS_PREFETCH registry names of the embedding models to bake; empty
|
||||
# bakes both defaults (mpnet for upgrades, granite for
|
||||
# new installs).
|
||||
#
|
||||
# Everything the default configuration needs is inside the image: embedding
|
||||
# models, their tokenizers, tiktoken's encoding and, with the docling extra,
|
||||
# docling's layout/table/OCR models. `python -m application.scripts.verify_offline`
|
||||
# under `docker run --network none` proves it.
|
||||
|
||||
FROM ubuntu:24.04 AS builder
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
# Ubuntu 24.04 ships Python 3.12 in its main archive: no PPA needed. Every pin
|
||||
# resolves to a wheel, so no compiler toolchain either.
|
||||
RUN apt-get update && \
|
||||
apt-get install -y --no-install-recommends python3.12 python3.12-venv ca-certificates && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY requirements.txt requirements-docling.txt requirements-milvus.txt ./
|
||||
|
||||
RUN python3.12 -m venv /venv
|
||||
ENV PATH="/venv/bin:$PATH"
|
||||
|
||||
RUN pip install --no-cache-dir --upgrade pip && \
|
||||
pip install --no-cache-dir --only-binary=:all: -r requirements.txt
|
||||
|
||||
# Optional extras. Each requirements-<extra>.txt is exported from the same
|
||||
# lock as requirements.txt, so installing it on top only adds the extra's
|
||||
# packages. The docling file takes torch from the CPU-only PyTorch index.
|
||||
# Not wheels-only: docling's antlr4 runtime ships as a pure-Python sdist.
|
||||
ARG EXTRAS=""
|
||||
ARG INSTALL_DOCLING=false
|
||||
RUN set -e; \
|
||||
extras="$EXTRAS"; \
|
||||
if [ "$INSTALL_DOCLING" = "true" ]; then extras="$extras,docling"; fi; \
|
||||
for extra in $(echo "$extras" | tr ',' ' '); do \
|
||||
echo "Installing extra: $extra"; \
|
||||
pip install --no-cache-dir -r "requirements-$extra.txt"; \
|
||||
done
|
||||
|
||||
# google-api-python-client bundles discovery documents for ~600 Google APIs
|
||||
# (99 MB). The application builds one client, Drive v3; keep only its document.
|
||||
# Building another API's client needs its file back, or static_discovery=False.
|
||||
RUN find /venv/lib/python3.12/site-packages/googleapiclient/discovery_cache/documents \
|
||||
-type f ! -name 'drive.v3.json' -delete
|
||||
|
||||
|
||||
FROM ubuntu:24.04 AS final
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y software-properties-common && \
|
||||
add-apt-repository ppa:deadsnakes/ppa && \
|
||||
apt-get update && \
|
||||
apt-get install -y --no-install-recommends gcc g++ wget unzip libc6-dev python3.12 python3.12-venv python3.12-dev && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Verify Python installation and setup symlink
|
||||
RUN if [ -f /usr/bin/python3.12 ]; then \
|
||||
ln -s /usr/bin/python3.12 /usr/bin/python; \
|
||||
else \
|
||||
echo "Python 3.12 not found"; exit 1; \
|
||||
fi
|
||||
|
||||
# Install Rust
|
||||
RUN wget -q -O - https://sh.rustup.rs | sh -s -- -y
|
||||
|
||||
# Clean up to reduce container size
|
||||
RUN apt-get remove --purge -y wget unzip && apt-get autoremove -y && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Copy requirements manifests
|
||||
COPY requirements.txt requirements-docling.txt ./
|
||||
|
||||
# Setup Python virtual environment
|
||||
RUN python3.12 -m venv /venv
|
||||
|
||||
# Activate virtual environment and install Python packages
|
||||
ENV PATH="/venv/bin:$PATH"
|
||||
|
||||
# Install Python packages
|
||||
RUN pip install --no-cache-dir --upgrade pip && \
|
||||
pip install --no-cache-dir tiktoken && \
|
||||
pip install --no-cache-dir -r requirements.txt
|
||||
|
||||
# Optional docling parser engine (DOC_PARSER_ENGINE=docling, the docling OCR
|
||||
# backend, read_document's structured output) — OFF by default: it pulls the
|
||||
# layout/OCR model stack and adds gigabytes to the image. anydoc (in
|
||||
# requirements.txt) is the default parser and needs none of it, and OCR runs
|
||||
# natively on tesseract (below, also opt-in) without it.
|
||||
ARG INSTALL_DOCLING=false
|
||||
RUN if [ "$INSTALL_DOCLING" = "true" ]; then \
|
||||
pip install --no-cache-dir -r requirements-docling.txt; \
|
||||
fi
|
||||
|
||||
# Final Stage
|
||||
FROM ubuntu:24.04 as final
|
||||
|
||||
RUN apt-get update && \
|
||||
apt-get install -y software-properties-common && \
|
||||
add-apt-repository ppa:deadsnakes/ppa && \
|
||||
apt-get update && apt-get install -y --no-install-recommends \
|
||||
python3.12 \
|
||||
libgl1 \
|
||||
libglib2.0-0 \
|
||||
poppler-utils \
|
||||
&& \
|
||||
apt-get install -y --no-install-recommends python3.12 poppler-utils ca-certificates && \
|
||||
ln -s /usr/bin/python3.12 /usr/bin/python && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# opencv (rapidocr, part of the docling extra) needs libGL at import time.
|
||||
ARG EXTRAS=""
|
||||
ARG INSTALL_DOCLING=false
|
||||
RUN if [ "$INSTALL_DOCLING" = "true" ] || echo ",$EXTRAS," | grep -q ",docling,"; then \
|
||||
apt-get update && \
|
||||
apt-get install -y --no-install-recommends libgl1 libglib2.0-0 && \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
fi
|
||||
|
||||
# Optional tesseract OCR engine (OCR_ENABLED=true with OCR_ENGINE=tesseract,
|
||||
# the default engine) — OFF by default like every other OCR dependency; OCR
|
||||
# itself is off unless configured. Opt in with --build-arg
|
||||
# INSTALL_TESSERACT=true (setup.sh writes it to .env when OCR is enabled);
|
||||
# ~35 MB of system packages. Extra language packs are a deployment concern
|
||||
# (apt: tesseract-ocr-<lang>, then list them in OCR_LANGS). A DeepSeek-OCR
|
||||
# endpoint (OCR_ENGINE=deepseek) needs none of this.
|
||||
# the default engine); ~35 MB of system packages. Extra language packs are a
|
||||
# deployment concern (apt: tesseract-ocr-<lang>, then list them in OCR_LANGS).
|
||||
# A DeepSeek-OCR endpoint (OCR_ENGINE=deepseek) needs none of this.
|
||||
ARG INSTALL_TESSERACT=false
|
||||
RUN if [ "$INSTALL_TESSERACT" = "true" ]; then \
|
||||
apt-get update && \
|
||||
@@ -76,61 +83,77 @@ RUN if [ "$INSTALL_TESSERACT" = "true" ]; then \
|
||||
rm -rf /var/lib/apt/lists/*; \
|
||||
fi
|
||||
|
||||
# Set working directory
|
||||
LABEL org.opencontainers.image.source="https://github.com/arc53/DocsGPT" \
|
||||
org.opencontainers.image.title="DocsGPT" \
|
||||
org.opencontainers.image.description="DocsGPT backend: API and Celery worker" \
|
||||
org.opencontainers.image.licenses="MIT"
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
# Create a non-root user: `appuser` (Feel free to choose a name)
|
||||
# The process user owns /app so the model prefetch below can run as it: an
|
||||
# unprivileged prefetch writes the model files with the right owner up front,
|
||||
# instead of a trailing chown -R that rewrites every model file into a second
|
||||
# layer.
|
||||
RUN groupadd -r appuser && \
|
||||
useradd -r -g appuser -d /app -s /sbin/nologin -c "Docker image user" appuser
|
||||
useradd -r -g appuser -d /app -s /sbin/nologin -c "Docker image user" appuser && \
|
||||
chown appuser:appuser /app && \
|
||||
install -d -o appuser -g appuser /app/models /app/application
|
||||
|
||||
# Copy the virtual environment and model from the builder stage
|
||||
COPY --from=builder /venv /venv
|
||||
|
||||
# Pre-fetch the embedding models into FastEmbed's cache so a fresh container
|
||||
# does not download on first ingest and an air-gapped install works at all.
|
||||
# Both defaults are baked: an upgraded deployment keeps using mpnet until it
|
||||
# runs the re-embed script, while a new one starts on granite.
|
||||
# The prefetch writes hub-layout snapshots (including tokenizer.json) here, so
|
||||
# HF_HUB_CACHE has to point at the same directory: chunking loads the tokenizer
|
||||
# through ``tokenizers``, which reads the hub cache and would otherwise fetch
|
||||
# over the network on first ingest -- and fall back to cl100k when offline.
|
||||
# Every cache the application reads at run time lives under /app/models and is
|
||||
# filled at build time:
|
||||
# EMBEDDINGS_CACHE_DIR / HF_HUB_CACHE FastEmbed models and their tokenizers
|
||||
# (chunking reads tokenizer.json from
|
||||
# the same hub-layout snapshot)
|
||||
# TIKTOKEN_CACHE_DIR cl100k_base for token accounting
|
||||
# DOCLING_ARTIFACTS_PATH docling's models (docling extra only)
|
||||
ENV EMBEDDINGS_CACHE_DIR=/app/models \
|
||||
HF_HUB_CACHE=/app/models
|
||||
|
||||
# Only the modules the prefetch imports are copied first. It reaches nothing
|
||||
# beyond model_registry, which is stdlib-only, so keeping the full source copy
|
||||
# below this layer stops an unrelated edit from re-downloading ~780 MB of model
|
||||
# artifacts on every build.
|
||||
COPY __init__.py /app/application/__init__.py
|
||||
COPY scripts/__init__.py scripts/prefetch_models.py /app/application/scripts/
|
||||
COPY vectorstore/__init__.py vectorstore/model_registry.py /app/application/vectorstore/
|
||||
ARG EMBEDDINGS_PREFETCH=""
|
||||
RUN PYTHONPATH=/app /venv/bin/python -m application.scripts.prefetch_models ${EMBEDDINGS_PREFETCH}
|
||||
|
||||
# Copy your application code
|
||||
COPY . /app/application
|
||||
|
||||
# Change the ownership of the /app directory to the appuser
|
||||
|
||||
RUN mkdir -p /app/application/inputs/local
|
||||
RUN chown -R appuser:appuser /app
|
||||
|
||||
# Set environment variables
|
||||
ENV FLASK_APP=app.py \
|
||||
FLASK_DEBUG=true \
|
||||
HF_HUB_CACHE=/app/models \
|
||||
TIKTOKEN_CACHE_DIR=/app/models/tiktoken \
|
||||
DOCLING_ARTIFACTS_PATH=/app/models/docling \
|
||||
HF_HUB_DISABLE_TELEMETRY=1 \
|
||||
PATH="/venv/bin:$PATH"
|
||||
|
||||
# Only the modules the prefetch imports are copied first, so an unrelated
|
||||
# source edit does not invalidate the model layer.
|
||||
COPY --chown=appuser:appuser __init__.py /app/application/__init__.py
|
||||
COPY --chown=appuser:appuser scripts/__init__.py scripts/prefetch_models.py /app/application/scripts/
|
||||
COPY --chown=appuser:appuser vectorstore/__init__.py vectorstore/model_registry.py /app/application/vectorstore/
|
||||
|
||||
USER appuser
|
||||
|
||||
ARG EMBEDDINGS_PREFETCH=""
|
||||
RUN PYTHONPATH=/app python -m application.scripts.prefetch_models ${EMBEDDINGS_PREFETCH} && \
|
||||
rm -rf /app/models/.locks /app/.cache
|
||||
|
||||
# docling downloads its layout, table-structure and OCR models on first parse;
|
||||
# bake them so the docling variant is as self-contained as the default image.
|
||||
RUN if python -c "import docling" 2>/dev/null; then \
|
||||
docling-tools models download --output-dir /app/models/docling layout tableformer rapidocr && \
|
||||
rm -rf /app/.cache; \
|
||||
fi
|
||||
|
||||
COPY --chown=appuser:appuser . /app/application
|
||||
|
||||
# Runtime data directories, owned by the process user so a named volume
|
||||
# mounted on them (docker-compose-standalone.yaml) inherits that ownership
|
||||
# and uploads work without running the container as root.
|
||||
RUN mkdir -p /app/application/inputs/local /app/inputs /app/indexes /app/vectors
|
||||
|
||||
ENV FLASK_APP=app.py
|
||||
|
||||
# Thread caps. onnxruntime (FastEmbed) ignores OMP_NUM_THREADS and sizes its
|
||||
# pool to the host's core count, which a CPU-limited container still reports;
|
||||
# EMBEDDINGS_THREADS pins it the way OMP_NUM_THREADS pinned torch before.
|
||||
ENV MALLOC_ARENA_MAX=2 \
|
||||
OMP_NUM_THREADS=4 \
|
||||
MKL_NUM_THREADS=4 \
|
||||
OPENBLAS_NUM_THREADS=4
|
||||
OPENBLAS_NUM_THREADS=4 \
|
||||
EMBEDDINGS_THREADS=4
|
||||
|
||||
# Expose the port the app runs on
|
||||
EXPOSE 7091
|
||||
|
||||
# Switch to non-root user
|
||||
USER appuser
|
||||
|
||||
# BoundedDrainUvicornWorker makes max_requests recycles safe with held-open SSE
|
||||
# connections (see application/gunicorn_worker.py); with recycles now safe,
|
||||
# --max-requests is raised (kept for memory hygiene) to cut churn.
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
"""Optional dependency extras and the one place their install hints come from.
|
||||
|
||||
Heavy or niche packages are not installed by default. Each extra maps to a
|
||||
pyproject extra and to an exported ``application/requirements-<extra>.txt``,
|
||||
so a missing module can always be explained with the exact command to run.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib
|
||||
import importlib.util
|
||||
import sys
|
||||
from types import ModuleType
|
||||
from typing import Dict, Tuple
|
||||
|
||||
#: Extra name -> top-level modules it provides. Keep in sync with
|
||||
#: ``[project.optional-dependencies]`` in pyproject.toml.
|
||||
EXTRAS: Dict[str, Tuple[str, ...]] = {
|
||||
"docling": ("docling", "rapidocr"),
|
||||
"milvus": ("pymilvus",),
|
||||
}
|
||||
|
||||
_MODULE_TO_EXTRA: Dict[str, str] = {
|
||||
module: extra for extra, modules in EXTRAS.items() for module in modules
|
||||
}
|
||||
|
||||
|
||||
def install_hint(extra: str) -> str:
|
||||
"""Install command for ``extra``, for error messages and logs."""
|
||||
return (
|
||||
f"pip install -r application/requirements-{extra}.txt "
|
||||
f"(or: uv sync --extra {extra}; Docker: --build-arg EXTRAS={extra})"
|
||||
)
|
||||
|
||||
|
||||
def extra_for(module: str) -> str | None:
|
||||
"""Extra that provides top-level module ``module``, if any."""
|
||||
return _MODULE_TO_EXTRA.get(module.split(".")[0])
|
||||
|
||||
|
||||
def is_available(module: str) -> bool:
|
||||
"""Whether ``module`` can be imported, without importing it."""
|
||||
root = module.split(".")[0]
|
||||
if root in sys.modules:
|
||||
return sys.modules[root] is not None
|
||||
try:
|
||||
return importlib.util.find_spec(root) is not None
|
||||
except (ImportError, ValueError):
|
||||
return False
|
||||
|
||||
|
||||
def missing_message(module: str, purpose: str | None = None) -> str:
|
||||
"""Human-readable explanation that ``module`` is absent and how to add it."""
|
||||
extra = extra_for(module)
|
||||
what = f"{module} is not installed"
|
||||
if purpose:
|
||||
what += f" ({purpose})"
|
||||
if extra:
|
||||
return f"{what}. It is part of the optional '{extra}' extra: {install_hint(extra)}"
|
||||
return f"{what}. Install it with: pip install {module}"
|
||||
|
||||
|
||||
def require(module: str, purpose: str | None = None) -> ModuleType:
|
||||
"""Import ``module`` or raise ``ImportError`` naming the extra to install.
|
||||
|
||||
Args:
|
||||
module: Importable module path, e.g. ``"pymilvus"``.
|
||||
purpose: Short note on what needed it, included in the error.
|
||||
|
||||
Returns:
|
||||
The imported module.
|
||||
|
||||
Raises:
|
||||
ImportError: With the install hint when the module is absent.
|
||||
"""
|
||||
try:
|
||||
return importlib.import_module(module)
|
||||
except ImportError as exc:
|
||||
raise ImportError(missing_message(module, purpose)) from exc
|
||||
@@ -19,6 +19,7 @@ from application.parser.schema.base import Document
|
||||
from application.stt.constants import SUPPORTED_AUDIO_EXTENSIONS
|
||||
from application.utils import num_tokens_from_string
|
||||
from application.core.settings import settings
|
||||
from application.core.optional_deps import install_hint
|
||||
|
||||
|
||||
def _build_audio_parser_mapping() -> Dict[str, BaseParser]:
|
||||
@@ -199,8 +200,9 @@ def _docling_file_extractor(
|
||||
logging.log(
|
||||
missing_log_level,
|
||||
"docling is not installed. Using standard parsers%s. For layout-model "
|
||||
"parsing, install with: pip install -r application/requirements-docling.txt",
|
||||
"parsing, install the docling extra: %s",
|
||||
" with native OCR" if ocr_enabled else "",
|
||||
install_hint("docling"),
|
||||
)
|
||||
return _legacy_file_extractor(pdf_text_fast_path, ocr_enabled=ocr_enabled)
|
||||
|
||||
|
||||
@@ -26,6 +26,8 @@ from application.parser.file.ocr_parser import VALID_OCR_ENGINES as _VALID_OCR_E
|
||||
from application.parser.file.ocr_parser import collapse_cjk_spaces
|
||||
from application.utils import truncate_to_line_boundary
|
||||
|
||||
from application.core.optional_deps import install_hint
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -590,11 +592,14 @@ class DoclingParser(BaseParser):
|
||||
logger.info(f" force_full_page_ocr={self.force_full_page_ocr}")
|
||||
logger.info(f" ocr_engine={self.ocr_engine or settings.OCR_ENGINE}")
|
||||
|
||||
if importlib.util.find_spec("docling.document_converter") is None:
|
||||
raise ImportError(
|
||||
"docling is required for DoclingParser. "
|
||||
"Install it with: pip install -r application/requirements-docling.txt"
|
||||
)
|
||||
# find_spec raises when the parent package is absent, so the hint has
|
||||
# to cover both a missing docling and a docling without the submodule.
|
||||
try:
|
||||
converter_spec = importlib.util.find_spec("docling.document_converter")
|
||||
except ModuleNotFoundError:
|
||||
converter_spec = None
|
||||
if converter_spec is None:
|
||||
raise ImportError(f"docling is required for DoclingParser. {install_hint('docling')}")
|
||||
|
||||
# Create converter with hybrid OCR (smart: text direct, bitmaps OCR'd)
|
||||
self._converter = self._create_converter()
|
||||
|
||||
@@ -39,6 +39,8 @@ from application.parser.file.base_parser import (
|
||||
module_available,
|
||||
)
|
||||
|
||||
from application.core.optional_deps import install_hint
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Every engine ``OCR_ENGINE`` accepts. ``auto``, ``ocrmac`` and ``rapidocr``
|
||||
@@ -129,8 +131,8 @@ def resolve_ocr_backend(requested: Optional[str] = None) -> str:
|
||||
docling_installed = module_available("docling")
|
||||
if backend == "docling" and not docling_installed:
|
||||
logger.warning(
|
||||
"OCR_BACKEND=docling but docling is not installed (pip install -r "
|
||||
"application/requirements-docling.txt); using the native OCR backend"
|
||||
"OCR_BACKEND=docling but docling is not installed (%s); using the native OCR backend",
|
||||
install_hint("docling"),
|
||||
)
|
||||
return "native"
|
||||
if backend == "auto":
|
||||
|
||||
@@ -9,6 +9,11 @@ from application.parser.schema.base import Document
|
||||
import tldextract
|
||||
import os
|
||||
|
||||
# The bundled public-suffix snapshot is enough for domain matching; the
|
||||
# default extractor would fetch the live list on first use and cache it on
|
||||
# disk, which is a network round trip the ingest worker should not depend on.
|
||||
_extract = tldextract.TLDExtract(suffix_list_urls=(), cache_dir=None)
|
||||
|
||||
class CrawlerLoader(BaseRemote):
|
||||
def __init__(self, limit=10, allow_subdomains=False):
|
||||
"""
|
||||
@@ -124,7 +129,7 @@ class CrawlerLoader(BaseRemote):
|
||||
return links
|
||||
|
||||
def _get_base_domain(self, url):
|
||||
extracted = tldextract.extract(url)
|
||||
extracted = _extract(url)
|
||||
# Reconstruct the domain as domain.suffix
|
||||
base_domain = f"{extracted.domain}.{extracted.suffix}"
|
||||
return base_domain
|
||||
@@ -141,7 +146,7 @@ class CrawlerLoader(BaseRemote):
|
||||
if not parsed_link.netloc:
|
||||
continue
|
||||
|
||||
extracted = tldextract.extract(parsed_link.netloc)
|
||||
extracted = _extract(parsed_link.netloc)
|
||||
link_base = f"{extracted.domain}.{extracted.suffix}"
|
||||
|
||||
if self.allow_subdomains:
|
||||
|
||||
@@ -216,12 +216,28 @@ class HuggingFaceCounter(TokenCounter):
|
||||
return _cap_piece_chars(pieces, first, rest)
|
||||
|
||||
|
||||
def _tokenizer_file(repo: str) -> str:
|
||||
"""Path to ``repo``'s ``tokenizer.json``, from the hub cache when present.
|
||||
|
||||
A warmed cache (the Docker image bakes the default models) answers without
|
||||
touching the network. ``hf_hub_download`` would otherwise revalidate the
|
||||
revision with a HEAD request on every process start, and stall for the
|
||||
etag timeout on a host that cannot reach huggingface.co.
|
||||
"""
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
try:
|
||||
return hf_hub_download(repo, "tokenizer.json", local_files_only=True)
|
||||
except Exception: # noqa: BLE001 -- not cached: fetch it
|
||||
return hf_hub_download(repo, "tokenizer.json")
|
||||
|
||||
|
||||
def _load_hf_counter(repo: str) -> Optional[HuggingFaceCounter]:
|
||||
"""Load ``repo``'s tokenizer, or ``None`` if it is not reachable."""
|
||||
try:
|
||||
from tokenizers import Tokenizer
|
||||
|
||||
tokenizer = Tokenizer.from_pretrained(repo)
|
||||
tokenizer = Tokenizer.from_file(_tokenizer_file(repo))
|
||||
# Repos ship padding and truncation defaults meant for inference
|
||||
# batches. Left on, every count returns the padded width (128 for
|
||||
# mpnet) and the offsets carry (0, 0) entries for the padding, so both
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,977 @@
|
||||
# GENERATED by scripts/export_requirements.sh from uv.lock -- do not edit.
|
||||
#
|
||||
# Core runtime plus the milvus extra (VECTOR_STORE=milvus): pymilvus and the
|
||||
# embedded milvus-lite server, which pulls pyarrow.
|
||||
# Docker: --build-arg EXTRAS=milvus
|
||||
|
||||
a2wsgi==1.10.10
|
||||
# via docsgpt
|
||||
aiofile==3.12.3
|
||||
# via py-key-value-aio
|
||||
aiofiles==25.1.0
|
||||
# via daytona
|
||||
aiohappyeyeballs==2.7.1
|
||||
# via aiohttp
|
||||
aiohttp==3.14.3
|
||||
# via
|
||||
# aiohttp-retry
|
||||
# daytona
|
||||
# daytona-analytics-api-client-async
|
||||
# daytona-api-client-async
|
||||
# daytona-toolbox-api-client-async
|
||||
# python-socketio
|
||||
aiohttp-retry==2.9.1
|
||||
# via
|
||||
# daytona-analytics-api-client-async
|
||||
# daytona-api-client-async
|
||||
# daytona-toolbox-api-client-async
|
||||
aiosignal==1.4.0
|
||||
# via aiohttp
|
||||
alembic==1.19.2
|
||||
# via docsgpt
|
||||
amqp==5.3.1
|
||||
# via kombu
|
||||
aniso8601==10.0.1
|
||||
# via flask-restx
|
||||
annotated-doc==0.0.5
|
||||
# via typer
|
||||
annotated-types==0.8.0
|
||||
# via pydantic
|
||||
anthropic==0.121.0
|
||||
# via docsgpt
|
||||
anyio==4.15.1
|
||||
# via
|
||||
# anthropic
|
||||
# google-genai
|
||||
# httpx
|
||||
# httpx-ws
|
||||
# mcp
|
||||
# openai
|
||||
# py-key-value-aio
|
||||
# sse-starlette
|
||||
# starlette
|
||||
# watchfiles
|
||||
asgiref==3.12.1
|
||||
# via opentelemetry-instrumentation-asgi
|
||||
attrs==26.1.0
|
||||
# via
|
||||
# aiohttp
|
||||
# cyclopts
|
||||
# jsonschema
|
||||
# jsonschema-path
|
||||
# referencing
|
||||
authlib==1.8.0
|
||||
# via fastmcp-slim
|
||||
beartype==0.22.9
|
||||
# via py-key-value-aio
|
||||
beautifulsoup4==4.15.0
|
||||
# via
|
||||
# docsgpt
|
||||
# markdownify
|
||||
bidict==0.24.1
|
||||
# via python-socketio
|
||||
billiard==4.2.4
|
||||
# via celery
|
||||
blinker==1.9.0
|
||||
# via flask
|
||||
boto3==1.43.67
|
||||
# via docsgpt
|
||||
botocore==1.43.89
|
||||
# via
|
||||
# boto3
|
||||
# s3transfer
|
||||
cachetools==7.1.8
|
||||
# via
|
||||
# py-key-value-aio
|
||||
# pymilvus
|
||||
caio==0.12.2
|
||||
# via aiofile
|
||||
cel-python==0.5.0
|
||||
# via docsgpt
|
||||
celery==5.6.3
|
||||
# via
|
||||
# celery-redbeat
|
||||
# docsgpt
|
||||
celery-redbeat==2.4.2
|
||||
# via docsgpt
|
||||
certifi==2026.7.22
|
||||
# via
|
||||
# httpcore
|
||||
# httpx
|
||||
# requests
|
||||
cffi==2.1.1 ; platform_python_implementation != 'PyPy'
|
||||
# via cryptography
|
||||
chardet==7.6.0
|
||||
# via prance
|
||||
charset-normalizer==3.5.1
|
||||
# via requests
|
||||
click==8.1.8
|
||||
# via
|
||||
# celery
|
||||
# click-didyoumean
|
||||
# click-plugins
|
||||
# click-repl
|
||||
# ddgs
|
||||
# flask
|
||||
# gtts
|
||||
# uvicorn
|
||||
click-didyoumean==0.3.1
|
||||
# via celery
|
||||
click-plugins==1.1.1.2
|
||||
# via celery
|
||||
click-repl==0.3.0
|
||||
# via celery
|
||||
colorama==0.4.6 ; sys_platform == 'win32'
|
||||
# via
|
||||
# click
|
||||
# loguru
|
||||
# tqdm
|
||||
# typer
|
||||
croniter==6.2.4
|
||||
# via docsgpt
|
||||
cryptography==50.0.0
|
||||
# via
|
||||
# authlib
|
||||
# docsgpt
|
||||
# google-auth
|
||||
# joserfc
|
||||
# msal
|
||||
# pyjwt
|
||||
# secretstorage
|
||||
cyclopts==4.24.0
|
||||
# via fastmcp-slim
|
||||
dataclasses-json==0.6.7
|
||||
# via docsgpt
|
||||
daytona==0.205.1
|
||||
# via docsgpt
|
||||
daytona-analytics-api-client==0.205.1
|
||||
# via daytona
|
||||
daytona-analytics-api-client-async==0.205.1
|
||||
# via daytona
|
||||
daytona-api-client==0.205.1
|
||||
# via daytona
|
||||
daytona-api-client-async==0.205.1
|
||||
# via daytona
|
||||
daytona-toolbox-api-client==0.205.1
|
||||
# via daytona
|
||||
daytona-toolbox-api-client-async==0.205.1
|
||||
# via daytona
|
||||
ddgs==9.16.0
|
||||
# via docsgpt
|
||||
decorator==5.3.1
|
||||
# via retry
|
||||
defusedxml==0.7.1
|
||||
# via
|
||||
# docsgpt
|
||||
# praw
|
||||
deprecated==1.3.1
|
||||
# via daytona
|
||||
distro==1.9.0
|
||||
# via
|
||||
# anthropic
|
||||
# google-genai
|
||||
# openai
|
||||
dnspython==2.8.0
|
||||
# via email-validator
|
||||
docstring-parser==0.18.0
|
||||
# via
|
||||
# anthropic
|
||||
# cyclopts
|
||||
docx2txt==0.9
|
||||
# via docsgpt
|
||||
ecdsa==0.19.2
|
||||
# via python-jose
|
||||
elevenlabs==2.62.0
|
||||
# via docsgpt
|
||||
email-validator==2.3.0
|
||||
# via pydantic
|
||||
et-xmlfile==2.0.0
|
||||
# via openpyxl
|
||||
exceptiongroup==1.3.1
|
||||
# via fastmcp-slim
|
||||
faiss-cpu==1.15.0
|
||||
# via
|
||||
# docsgpt
|
||||
# milvus-lite
|
||||
fast-ebook==0.2.0
|
||||
# via docsgpt
|
||||
fastembed==0.8.0
|
||||
# via docsgpt
|
||||
fastmcp==3.4.6
|
||||
# via docsgpt
|
||||
fastmcp-slim==3.4.6
|
||||
# via fastmcp
|
||||
filelock==3.32.5
|
||||
# via
|
||||
# huggingface-hub
|
||||
# tldextract
|
||||
firecrawl-anydoc==0.2.3
|
||||
# via docsgpt
|
||||
flask==3.1.3
|
||||
# via
|
||||
# docsgpt
|
||||
# flask-restx
|
||||
flask-restx==1.3.2
|
||||
# via docsgpt
|
||||
flatbuffers==25.12.19
|
||||
# via onnxruntime
|
||||
frozenlist==1.8.0
|
||||
# via
|
||||
# aiohttp
|
||||
# aiosignal
|
||||
fsspec==2026.7.0
|
||||
# via huggingface-hub
|
||||
google-api-core==2.36.0
|
||||
# via google-api-python-client
|
||||
google-api-python-client==2.198.0
|
||||
# via docsgpt
|
||||
google-auth==2.57.1
|
||||
# via
|
||||
# google-api-core
|
||||
# google-api-python-client
|
||||
# google-auth-httplib2
|
||||
# google-auth-oauthlib
|
||||
# google-genai
|
||||
google-auth-httplib2==0.4.2
|
||||
# via google-api-python-client
|
||||
google-auth-oauthlib==1.4.0
|
||||
# via docsgpt
|
||||
google-genai==2.17.0
|
||||
# via docsgpt
|
||||
google-re2==1.1.20251105
|
||||
# via cel-python
|
||||
googleapis-common-protos==1.75.3
|
||||
# via
|
||||
# google-api-core
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
greenlet==3.5.5 ; platform_machine == 'AMD64' or platform_machine == 'WIN32' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'ppc64le' or platform_machine == 'win32' or platform_machine == 'x86_64'
|
||||
# via sqlalchemy
|
||||
griffelib==2.3.0
|
||||
# via fastmcp-slim
|
||||
grpcio==1.83.1
|
||||
# via
|
||||
# milvus-lite
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# pymilvus
|
||||
# qdrant-client
|
||||
gtts==2.5.4
|
||||
# via docsgpt
|
||||
gunicorn==26.0.0
|
||||
# via
|
||||
# docsgpt
|
||||
# uvicorn-worker
|
||||
h11==0.16.0
|
||||
# via
|
||||
# httpcore
|
||||
# uvicorn
|
||||
# wsproto
|
||||
h2==4.4.1
|
||||
# via httpx
|
||||
hf-xet==1.6.0 ; platform_machine == 'AMD64' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'arm64' or platform_machine == 'x86_64'
|
||||
# via huggingface-hub
|
||||
hpack==4.2.0
|
||||
# via h2
|
||||
httpcore==1.0.9
|
||||
# via
|
||||
# httpx
|
||||
# httpx-ws
|
||||
httplib2==0.32.0
|
||||
# via
|
||||
# google-api-python-client
|
||||
# google-auth-httplib2
|
||||
httptools==0.8.0
|
||||
# via uvicorn
|
||||
httpx==0.28.1
|
||||
# via
|
||||
# anthropic
|
||||
# daytona
|
||||
# elevenlabs
|
||||
# fastmcp-slim
|
||||
# google-genai
|
||||
# httpx-ws
|
||||
# huggingface-hub
|
||||
# mcp
|
||||
# openai
|
||||
# qdrant-client
|
||||
httpx-sse==0.4.3
|
||||
# via mcp
|
||||
httpx-ws==0.9.0
|
||||
# via daytona
|
||||
huggingface-hub==1.16.1
|
||||
# via
|
||||
# fastembed
|
||||
# tokenizers
|
||||
hyperframe==6.1.0
|
||||
# via h2
|
||||
idna==3.19
|
||||
# via
|
||||
# anyio
|
||||
# email-validator
|
||||
# httpx
|
||||
# requests
|
||||
# tldextract
|
||||
# yarl
|
||||
importlib-resources==7.1.0
|
||||
# via flask-restx
|
||||
itsdangerous==2.2.0
|
||||
# via flask
|
||||
jaraco-classes==3.4.0
|
||||
# via keyring
|
||||
jaraco-context==6.1.2
|
||||
# via keyring
|
||||
jaraco-functools==4.6.0
|
||||
# via keyring
|
||||
jeepney==0.9.0 ; sys_platform == 'linux'
|
||||
# via
|
||||
# keyring
|
||||
# secretstorage
|
||||
jinja2==3.1.6
|
||||
# via
|
||||
# docsgpt
|
||||
# flask
|
||||
jiter==0.16.0
|
||||
# via
|
||||
# anthropic
|
||||
# openai
|
||||
jmespath==1.1.0
|
||||
# via
|
||||
# boto3
|
||||
# botocore
|
||||
# cel-python
|
||||
joserfc==1.7.5
|
||||
# via
|
||||
# authlib
|
||||
# fastmcp-slim
|
||||
jsonref==1.1.0
|
||||
# via fastmcp-slim
|
||||
jsonschema==4.26.0
|
||||
# via
|
||||
# flask-restx
|
||||
# mcp
|
||||
# openapi-schema-validator
|
||||
# openapi-spec-validator
|
||||
jsonschema-path==0.5.0
|
||||
# via
|
||||
# fastmcp-slim
|
||||
# openapi-spec-validator
|
||||
jsonschema-specifications==2025.9.1
|
||||
# via
|
||||
# jsonschema
|
||||
# openapi-schema-validator
|
||||
keyring==25.7.0
|
||||
# via py-key-value-aio
|
||||
kombu==5.6.2
|
||||
# via
|
||||
# celery
|
||||
# docsgpt
|
||||
lark==1.3.1
|
||||
# via cel-python
|
||||
lazy-object-proxy==1.12.0
|
||||
# via openapi-spec-validator
|
||||
loguru==0.7.3
|
||||
# via fastembed
|
||||
lxml==6.1.3
|
||||
# via
|
||||
# ddgs
|
||||
# python-pptx
|
||||
mako==1.4.1
|
||||
# via alembic
|
||||
markdown-it-py==4.2.0
|
||||
# via rich
|
||||
markdownify==1.2.3
|
||||
# via docsgpt
|
||||
markupsafe==3.0.3
|
||||
# via
|
||||
# flask
|
||||
# jinja2
|
||||
# mako
|
||||
# werkzeug
|
||||
marshmallow==3.26.2
|
||||
# via dataclasses-json
|
||||
mcp==1.29.1
|
||||
# via fastmcp-slim
|
||||
mdurl==0.1.2
|
||||
# via markdown-it-py
|
||||
milvus-lite==3.2.0 ; sys_platform != 'win32'
|
||||
# via docsgpt
|
||||
mmh3==5.3.0
|
||||
# via fastembed
|
||||
more-itertools==11.1.0
|
||||
# via
|
||||
# jaraco-classes
|
||||
# jaraco-functools
|
||||
msal==1.37.0
|
||||
# via docsgpt
|
||||
multidict==6.7.1
|
||||
# via
|
||||
# aiohttp
|
||||
# yarl
|
||||
mypy-extensions==1.1.0
|
||||
# via typing-inspect
|
||||
networkx==3.6.1
|
||||
# via docsgpt
|
||||
numpy==2.5.1
|
||||
# via
|
||||
# docsgpt
|
||||
# faiss-cpu
|
||||
# fastembed
|
||||
# milvus-lite
|
||||
# onnxruntime
|
||||
# pandas
|
||||
# qdrant-client
|
||||
oauthlib==3.3.1
|
||||
# via requests-oauthlib
|
||||
obstore==0.11.1
|
||||
# via daytona
|
||||
onnxruntime==1.28.0
|
||||
# via
|
||||
# docsgpt
|
||||
# fastembed
|
||||
openai==2.53.0
|
||||
# via docsgpt
|
||||
openapi-pydantic==0.5.1
|
||||
# via fastmcp-slim
|
||||
openapi-schema-validator==0.9.0
|
||||
# via openapi-spec-validator
|
||||
openapi-spec-validator==0.9.0
|
||||
# via openapi3-parser
|
||||
openapi3-parser==1.1.22
|
||||
# via docsgpt
|
||||
openpyxl==3.1.5
|
||||
# via docsgpt
|
||||
opentelemetry-api==1.44.0
|
||||
# via
|
||||
# daytona
|
||||
# fastmcp-slim
|
||||
# google-api-core
|
||||
# opentelemetry-distro
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
# opentelemetry-instrumentation
|
||||
# opentelemetry-instrumentation-aiohttp-client
|
||||
# opentelemetry-instrumentation-asgi
|
||||
# opentelemetry-instrumentation-celery
|
||||
# opentelemetry-instrumentation-dbapi
|
||||
# opentelemetry-instrumentation-flask
|
||||
# opentelemetry-instrumentation-logging
|
||||
# opentelemetry-instrumentation-psycopg
|
||||
# opentelemetry-instrumentation-redis
|
||||
# opentelemetry-instrumentation-requests
|
||||
# opentelemetry-instrumentation-sqlalchemy
|
||||
# opentelemetry-instrumentation-starlette
|
||||
# opentelemetry-instrumentation-wsgi
|
||||
# opentelemetry-sdk
|
||||
# opentelemetry-semantic-conventions
|
||||
opentelemetry-distro==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-exporter-otlp==1.44.0
|
||||
# via docsgpt
|
||||
opentelemetry-exporter-otlp-proto-common==1.44.0
|
||||
# via
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
opentelemetry-exporter-otlp-proto-grpc==1.44.0
|
||||
# via opentelemetry-exporter-otlp
|
||||
opentelemetry-exporter-otlp-proto-http==1.44.0
|
||||
# via
|
||||
# daytona
|
||||
# opentelemetry-exporter-otlp
|
||||
opentelemetry-instrumentation==0.65b0
|
||||
# via
|
||||
# opentelemetry-distro
|
||||
# opentelemetry-instrumentation-aiohttp-client
|
||||
# opentelemetry-instrumentation-asgi
|
||||
# opentelemetry-instrumentation-celery
|
||||
# opentelemetry-instrumentation-dbapi
|
||||
# opentelemetry-instrumentation-flask
|
||||
# opentelemetry-instrumentation-logging
|
||||
# opentelemetry-instrumentation-psycopg
|
||||
# opentelemetry-instrumentation-redis
|
||||
# opentelemetry-instrumentation-requests
|
||||
# opentelemetry-instrumentation-sqlalchemy
|
||||
# opentelemetry-instrumentation-starlette
|
||||
# opentelemetry-instrumentation-wsgi
|
||||
opentelemetry-instrumentation-aiohttp-client==0.65b0
|
||||
# via daytona
|
||||
opentelemetry-instrumentation-asgi==0.65b0
|
||||
# via opentelemetry-instrumentation-starlette
|
||||
opentelemetry-instrumentation-celery==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-dbapi==0.65b0
|
||||
# via opentelemetry-instrumentation-psycopg
|
||||
opentelemetry-instrumentation-flask==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-logging==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-psycopg==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-redis==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-requests==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-sqlalchemy==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-starlette==0.65b0
|
||||
# via docsgpt
|
||||
opentelemetry-instrumentation-wsgi==0.65b0
|
||||
# via opentelemetry-instrumentation-flask
|
||||
opentelemetry-proto==1.44.0
|
||||
# via
|
||||
# opentelemetry-exporter-otlp-proto-common
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
opentelemetry-sdk==1.44.0
|
||||
# via
|
||||
# daytona
|
||||
# opentelemetry-distro
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
opentelemetry-semantic-conventions==0.65b0
|
||||
# via
|
||||
# opentelemetry-instrumentation
|
||||
# opentelemetry-instrumentation-aiohttp-client
|
||||
# opentelemetry-instrumentation-asgi
|
||||
# opentelemetry-instrumentation-celery
|
||||
# opentelemetry-instrumentation-dbapi
|
||||
# opentelemetry-instrumentation-flask
|
||||
# opentelemetry-instrumentation-logging
|
||||
# opentelemetry-instrumentation-redis
|
||||
# opentelemetry-instrumentation-requests
|
||||
# opentelemetry-instrumentation-sqlalchemy
|
||||
# opentelemetry-instrumentation-starlette
|
||||
# opentelemetry-instrumentation-wsgi
|
||||
# opentelemetry-sdk
|
||||
opentelemetry-util-http==0.65b0
|
||||
# via
|
||||
# opentelemetry-instrumentation-aiohttp-client
|
||||
# opentelemetry-instrumentation-asgi
|
||||
# opentelemetry-instrumentation-flask
|
||||
# opentelemetry-instrumentation-requests
|
||||
# opentelemetry-instrumentation-starlette
|
||||
# opentelemetry-instrumentation-wsgi
|
||||
orjson==3.12.0
|
||||
# via pymilvus
|
||||
packaging==26.3
|
||||
# via
|
||||
# faiss-cpu
|
||||
# fastmcp-slim
|
||||
# gunicorn
|
||||
# huggingface-hub
|
||||
# kombu
|
||||
# marshmallow
|
||||
# onnxruntime
|
||||
# opentelemetry-instrumentation
|
||||
# opentelemetry-instrumentation-flask
|
||||
# opentelemetry-instrumentation-sqlalchemy
|
||||
# prance
|
||||
pandas==3.0.5
|
||||
# via
|
||||
# docsgpt
|
||||
# pymilvus
|
||||
pathable==0.6.0
|
||||
# via jsonschema-path
|
||||
pdf2image==1.17.0
|
||||
# via docsgpt
|
||||
pendulum==3.2.0
|
||||
# via cel-python
|
||||
pgvector==0.5.0
|
||||
# via docsgpt
|
||||
pillow==12.3.0
|
||||
# via
|
||||
# docsgpt
|
||||
# fastembed
|
||||
# pdf2image
|
||||
# python-pptx
|
||||
platformdirs==4.11.7
|
||||
# via fastmcp-slim
|
||||
portalocker==3.2.0
|
||||
# via qdrant-client
|
||||
prance==26.7.19.0
|
||||
# via openapi3-parser
|
||||
praw==8.0.2
|
||||
# via docsgpt
|
||||
prawcore==4.0.0
|
||||
# via praw
|
||||
primp==2.0.0
|
||||
# via ddgs
|
||||
prompt-toolkit==3.0.53
|
||||
# via click-repl
|
||||
propcache==0.5.2
|
||||
# via
|
||||
# aiohttp
|
||||
# yarl
|
||||
proto-plus==1.28.4
|
||||
# via google-api-core
|
||||
protobuf==7.36.1
|
||||
# via
|
||||
# google-api-core
|
||||
# googleapis-common-protos
|
||||
# onnxruntime
|
||||
# opentelemetry-proto
|
||||
# proto-plus
|
||||
# pymilvus
|
||||
# qdrant-client
|
||||
psycopg==3.3.5
|
||||
# via docsgpt
|
||||
psycopg-binary==3.3.5 ; implementation_name != 'pypy'
|
||||
# via psycopg
|
||||
psycopg-pool==3.3.1
|
||||
# via psycopg
|
||||
py==1.11.0
|
||||
# via retry
|
||||
py-key-value-aio==0.4.5
|
||||
# via fastmcp-slim
|
||||
py-rust-stemmers==0.1.8
|
||||
# via fastembed
|
||||
pyarrow==25.0.1 ; sys_platform != 'win32'
|
||||
# via milvus-lite
|
||||
pyasn1==0.6.4
|
||||
# via
|
||||
# pyasn1-modules
|
||||
# python-jose
|
||||
# rsa
|
||||
pyasn1-modules==0.4.2
|
||||
# via google-auth
|
||||
pycparser==3.0 ; implementation_name != 'PyPy' and platform_python_implementation != 'PyPy'
|
||||
# via cffi
|
||||
pydantic==2.13.5
|
||||
# via
|
||||
# anthropic
|
||||
# daytona
|
||||
# daytona-analytics-api-client
|
||||
# daytona-analytics-api-client-async
|
||||
# daytona-api-client
|
||||
# daytona-api-client-async
|
||||
# daytona-toolbox-api-client
|
||||
# daytona-toolbox-api-client-async
|
||||
# docsgpt
|
||||
# elevenlabs
|
||||
# fastmcp-slim
|
||||
# google-genai
|
||||
# mcp
|
||||
# openai
|
||||
# openapi-pydantic
|
||||
# openapi-schema-validator
|
||||
# openapi-spec-validator
|
||||
# pydantic-settings
|
||||
# qdrant-client
|
||||
pydantic-core==2.46.5
|
||||
# via
|
||||
# elevenlabs
|
||||
# pydantic
|
||||
pydantic-settings==2.15.0
|
||||
# via
|
||||
# docsgpt
|
||||
# fastmcp-slim
|
||||
# mcp
|
||||
# openapi-schema-validator
|
||||
# openapi-spec-validator
|
||||
pygments==2.21.0
|
||||
# via
|
||||
# rich
|
||||
# rich-rst
|
||||
pyjwt==2.13.0
|
||||
# via
|
||||
# mcp
|
||||
# msal
|
||||
pymilvus==3.0.1
|
||||
# via docsgpt
|
||||
pyparsing==3.3.2
|
||||
# via httplib2
|
||||
pypdf==6.15.0
|
||||
# via docsgpt
|
||||
pypdfium2==5.12.1
|
||||
# via docsgpt
|
||||
pyperclip==1.11.0
|
||||
# via fastmcp-slim
|
||||
python-dateutil==2.9.0.post0
|
||||
# via
|
||||
# botocore
|
||||
# celery
|
||||
# celery-redbeat
|
||||
# croniter
|
||||
# daytona-analytics-api-client
|
||||
# daytona-analytics-api-client-async
|
||||
# daytona-api-client
|
||||
# daytona-api-client-async
|
||||
# daytona-toolbox-api-client
|
||||
# daytona-toolbox-api-client-async
|
||||
# docsgpt
|
||||
# pandas
|
||||
# pendulum
|
||||
python-dotenv==1.2.3
|
||||
# via
|
||||
# daytona
|
||||
# docsgpt
|
||||
# fastmcp-slim
|
||||
# pydantic-settings
|
||||
# pymilvus
|
||||
# uvicorn
|
||||
python-engineio==4.14.0
|
||||
# via python-socketio
|
||||
python-jose==3.5.0
|
||||
# via docsgpt
|
||||
python-multipart==0.0.32
|
||||
# via
|
||||
# daytona
|
||||
# fastmcp-slim
|
||||
# mcp
|
||||
python-pptx==1.0.2
|
||||
# via docsgpt
|
||||
python-socketio==5.16.4
|
||||
# via daytona
|
||||
pywin32==312 ; sys_platform == 'win32'
|
||||
# via
|
||||
# mcp
|
||||
# portalocker
|
||||
pywin32-ctypes==0.2.3 ; sys_platform == 'win32'
|
||||
# via keyring
|
||||
pyyaml==6.0.3
|
||||
# via
|
||||
# cel-python
|
||||
# docsgpt
|
||||
# fastmcp-slim
|
||||
# huggingface-hub
|
||||
# jsonschema-path
|
||||
# uvicorn
|
||||
qdrant-client==1.19.0
|
||||
# via docsgpt
|
||||
redis==7.4.0
|
||||
# via
|
||||
# celery-redbeat
|
||||
# docsgpt
|
||||
referencing==0.37.0
|
||||
# via
|
||||
# flask-restx
|
||||
# jsonschema
|
||||
# jsonschema-path
|
||||
# jsonschema-specifications
|
||||
# openapi-schema-validator
|
||||
regex==2026.9.3
|
||||
# via tiktoken
|
||||
requests==2.34.2
|
||||
# via
|
||||
# docsgpt
|
||||
# elevenlabs
|
||||
# fastembed
|
||||
# google-api-core
|
||||
# google-auth
|
||||
# google-genai
|
||||
# gtts
|
||||
# msal
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
# prance
|
||||
# prawcore
|
||||
# pymilvus
|
||||
# python-socketio
|
||||
# requests-file
|
||||
# requests-oauthlib
|
||||
# tiktoken
|
||||
# tldextract
|
||||
requests-file==3.0.1
|
||||
# via tldextract
|
||||
requests-oauthlib==2.0.0
|
||||
# via google-auth-oauthlib
|
||||
retry==0.9.2
|
||||
# via docsgpt
|
||||
rfc3339-validator==0.1.4
|
||||
# via openapi-schema-validator
|
||||
rich==15.0.0
|
||||
# via
|
||||
# cyclopts
|
||||
# fastmcp-slim
|
||||
# rich-rst
|
||||
# typer
|
||||
rich-rst==2.1.0
|
||||
# via cyclopts
|
||||
rpds-py==2026.6.3
|
||||
# via
|
||||
# jsonschema
|
||||
# referencing
|
||||
rsa==4.9.1
|
||||
# via python-jose
|
||||
ruamel-yaml==0.19.1
|
||||
# via prance
|
||||
s3transfer==0.19.2
|
||||
# via boto3
|
||||
secretstorage==3.5.0 ; sys_platform == 'linux'
|
||||
# via keyring
|
||||
shellingham==1.5.4
|
||||
# via typer
|
||||
simple-websocket==1.1.0
|
||||
# via python-engineio
|
||||
six==1.17.0
|
||||
# via
|
||||
# ecdsa
|
||||
# markdownify
|
||||
# python-dateutil
|
||||
# rfc3339-validator
|
||||
sniffio==1.3.1
|
||||
# via
|
||||
# anthropic
|
||||
# google-genai
|
||||
# openai
|
||||
soupsieve==2.9.2
|
||||
# via beautifulsoup4
|
||||
sqlalchemy==2.0.52
|
||||
# via
|
||||
# alembic
|
||||
# docsgpt
|
||||
sse-starlette==3.4.11
|
||||
# via mcp
|
||||
starlette==1.6.0
|
||||
# via
|
||||
# docsgpt
|
||||
# fastmcp-slim
|
||||
# mcp
|
||||
# sse-starlette
|
||||
tenacity==9.1.4
|
||||
# via
|
||||
# celery-redbeat
|
||||
# google-genai
|
||||
tiktoken==0.13.0
|
||||
# via docsgpt
|
||||
tldextract==5.3.2
|
||||
# via docsgpt
|
||||
tokenizers==0.22.2
|
||||
# via
|
||||
# docsgpt
|
||||
# fastembed
|
||||
toml==0.10.2
|
||||
# via daytona
|
||||
tqdm==4.67.3
|
||||
# via
|
||||
# docsgpt
|
||||
# fastembed
|
||||
# huggingface-hub
|
||||
# openai
|
||||
typer==0.26.8
|
||||
# via huggingface-hub
|
||||
typing-extensions==4.16.0
|
||||
# via
|
||||
# aiohttp
|
||||
# aiosignal
|
||||
# alembic
|
||||
# anthropic
|
||||
# anyio
|
||||
# beautifulsoup4
|
||||
# daytona
|
||||
# daytona-analytics-api-client
|
||||
# daytona-analytics-api-client-async
|
||||
# daytona-api-client
|
||||
# daytona-api-client-async
|
||||
# daytona-toolbox-api-client
|
||||
# daytona-toolbox-api-client-async
|
||||
# elevenlabs
|
||||
# exceptiongroup
|
||||
# fastmcp-slim
|
||||
# google-genai
|
||||
# grpcio
|
||||
# huggingface-hub
|
||||
# mcp
|
||||
# obstore
|
||||
# openai
|
||||
# opentelemetry-api
|
||||
# opentelemetry-exporter-otlp-proto-grpc
|
||||
# opentelemetry-exporter-otlp-proto-http
|
||||
# opentelemetry-sdk
|
||||
# opentelemetry-semantic-conventions
|
||||
# psycopg
|
||||
# psycopg-pool
|
||||
# py-key-value-aio
|
||||
# pydantic
|
||||
# pydantic-core
|
||||
# python-pptx
|
||||
# referencing
|
||||
# sqlalchemy
|
||||
# starlette
|
||||
# typing-inspect
|
||||
# typing-inspection
|
||||
typing-inspect==0.9.0
|
||||
# via dataclasses-json
|
||||
typing-inspection==0.4.4
|
||||
# via
|
||||
# mcp
|
||||
# pydantic
|
||||
# pydantic-settings
|
||||
tzdata==2026.3
|
||||
# via
|
||||
# kombu
|
||||
# pandas
|
||||
# pendulum
|
||||
# psycopg
|
||||
# tzlocal
|
||||
tzlocal==5.4.4
|
||||
# via celery
|
||||
uncalled-for==0.4.0
|
||||
# via fastmcp-slim
|
||||
update-checker==1.0.0
|
||||
# via praw
|
||||
uritemplate==4.2.0
|
||||
# via google-api-python-client
|
||||
urllib3==2.7.0
|
||||
# via
|
||||
# botocore
|
||||
# daytona
|
||||
# daytona-analytics-api-client
|
||||
# daytona-api-client
|
||||
# daytona-toolbox-api-client
|
||||
# qdrant-client
|
||||
# requests
|
||||
uvicorn==0.52.4
|
||||
# via
|
||||
# docsgpt
|
||||
# fastmcp-slim
|
||||
# mcp
|
||||
# uvicorn-worker
|
||||
uvicorn-worker==0.4.0
|
||||
# via docsgpt
|
||||
uvloop==0.22.1 ; platform_python_implementation != 'PyPy' and sys_platform != 'cygwin' and sys_platform != 'win32'
|
||||
# via uvicorn
|
||||
vine==5.1.0
|
||||
# via
|
||||
# amqp
|
||||
# celery
|
||||
# kombu
|
||||
watchfiles==1.2.0
|
||||
# via
|
||||
# fastmcp-slim
|
||||
# uvicorn
|
||||
wcwidth==0.8.3
|
||||
# via prompt-toolkit
|
||||
websocket-client==1.9.0
|
||||
# via
|
||||
# docsgpt
|
||||
# praw
|
||||
# python-socketio
|
||||
websockets==16.1.1
|
||||
# via
|
||||
# elevenlabs
|
||||
# fastmcp-slim
|
||||
# google-genai
|
||||
# uvicorn
|
||||
werkzeug==3.1.8
|
||||
# via
|
||||
# docsgpt
|
||||
# flask
|
||||
# flask-restx
|
||||
win32-setctime==1.2.0 ; sys_platform == 'win32'
|
||||
# via loguru
|
||||
wrapt==2.4.0
|
||||
# via
|
||||
# deprecated
|
||||
# opentelemetry-instrumentation
|
||||
# opentelemetry-instrumentation-aiohttp-client
|
||||
# opentelemetry-instrumentation-dbapi
|
||||
# opentelemetry-instrumentation-redis
|
||||
# opentelemetry-instrumentation-sqlalchemy
|
||||
wsproto==1.3.2
|
||||
# via
|
||||
# daytona
|
||||
# httpx-ws
|
||||
# simple-websocket
|
||||
xlsxwriter==3.2.9
|
||||
# via python-pptx
|
||||
yarl==1.24.5
|
||||
# via aiohttp
|
||||
+930
-98
File diff suppressed because it is too large.
Load diff
@@ -1,9 +1,14 @@
|
||||
"""Download embedding model artifacts into FastEmbed's cache.
|
||||
"""Download the model artifacts a fresh container would otherwise fetch.
|
||||
|
||||
Run at image build time so a fresh container does not download a model on its
|
||||
first ingest, and an air-gapped install works at all. Both the legacy and the
|
||||
current default are baked: an upgraded deployment keeps using mpnet until it
|
||||
runs ``reembed``, while a new one starts on granite.
|
||||
Run at image build time so a fresh container does not download on its first
|
||||
request, and an air-gapped install works at all. Two things are warmed:
|
||||
|
||||
* Embedding models, into FastEmbed's cache. Both the legacy and the current
|
||||
default are baked: an upgraded deployment keeps using mpnet until it runs
|
||||
``reembed``, while a new one starts on granite.
|
||||
* tiktoken's ``cl100k_base`` encoding, which token accounting uses on every
|
||||
chat. tiktoken caches it under ``TIKTOKEN_CACHE_DIR`` (a temp dir when
|
||||
unset), so the image sets that variable and this warms it.
|
||||
|
||||
Usage::
|
||||
|
||||
@@ -29,6 +34,23 @@ logger = logging.getLogger("prefetch_models")
|
||||
#: Fetched when no names are given.
|
||||
DEFAULT_MODELS = (DEFAULT_LEGACY, DEFAULT_NEW_INSTALL)
|
||||
|
||||
#: tiktoken encodings the application loads (``application.utils.get_encoding``).
|
||||
TIKTOKEN_ENCODINGS = ("cl100k_base",)
|
||||
|
||||
|
||||
def prefetch_tiktoken(names: Sequence[str] = TIKTOKEN_ENCODINGS) -> List[str]:
|
||||
"""Warm tiktoken's cache for each encoding in ``names``.
|
||||
|
||||
Returns:
|
||||
The encodings fetched.
|
||||
"""
|
||||
import tiktoken
|
||||
|
||||
for name in names:
|
||||
logger.info("Fetching tiktoken encoding %s", name)
|
||||
tiktoken.get_encoding(name)
|
||||
return list(names)
|
||||
|
||||
|
||||
def prefetch(names: Sequence[str], cache_dir: Optional[str] = None) -> List[str]:
|
||||
"""Fetch each named model's artifacts.
|
||||
@@ -82,6 +104,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
|
||||
names = list(argv) if argv else list(DEFAULT_MODELS)
|
||||
fetched = prefetch(names, os.environ.get("EMBEDDINGS_CACHE_DIR"))
|
||||
logger.info("Cached %d model(s): %s", len(fetched), ", ".join(fetched))
|
||||
encodings = prefetch_tiktoken()
|
||||
logger.info("Cached tiktoken encoding(s): %s", ", ".join(encodings))
|
||||
return 0
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,153 @@
|
||||
"""Check that an image can serve its defaults without any network access.
|
||||
|
||||
Exercises the code paths a fresh container hits first, the way the
|
||||
application does: tiktoken token counting, the chunker's tokenizer for each
|
||||
baked embedding model, and a FastEmbed embed with each. Run it inside the
|
||||
image with networking disabled; every check must pass with zero requests::
|
||||
|
||||
docker run --rm --network none arc53/docsgpt:latest \\
|
||||
python -m application.scripts.verify_offline
|
||||
|
||||
Exit status is non-zero on the first failure. Models to check default to
|
||||
the prefetch defaults; pass registry names to check a different set.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import socket
|
||||
import sys
|
||||
import time
|
||||
from typing import Callable, List, Optional, Sequence
|
||||
|
||||
from application.core.optional_deps import is_available
|
||||
from application.scripts.prefetch_models import DEFAULT_MODELS, TIKTOKEN_ENCODINGS
|
||||
from application.vectorstore.model_registry import resolve
|
||||
|
||||
logger = logging.getLogger("verify_offline")
|
||||
|
||||
|
||||
def _network_reachable(host: str = "huggingface.co", port: int = 443) -> bool:
|
||||
try:
|
||||
socket.create_connection((host, port), timeout=2).close()
|
||||
return True
|
||||
except OSError:
|
||||
return False
|
||||
|
||||
|
||||
def _check(name: str, fn: Callable[[], object]) -> bool:
|
||||
started = time.time()
|
||||
try:
|
||||
detail = fn()
|
||||
except Exception as exc: # noqa: BLE001 -- report every failure the same way
|
||||
print(f"FAIL {name}: {type(exc).__name__}: {exc}")
|
||||
return False
|
||||
print(f"ok {name}: {detail} ({time.time() - started:.2f}s)")
|
||||
return True
|
||||
|
||||
|
||||
def verify(models: Sequence[str]) -> bool:
|
||||
"""Run every check; return whether all passed."""
|
||||
ok = True
|
||||
|
||||
def tiktoken_check(encoding: str) -> Callable[[], object]:
|
||||
def run() -> object:
|
||||
import tiktoken
|
||||
|
||||
return f"{len(tiktoken.get_encoding(encoding).encode('hello world'))} tokens"
|
||||
|
||||
return run
|
||||
|
||||
for encoding in TIKTOKEN_ENCODINGS:
|
||||
ok &= _check(f"tiktoken {encoding}", tiktoken_check(encoding))
|
||||
|
||||
for name in models:
|
||||
spec = resolve(name)
|
||||
if spec is None or spec.provider != "fastembed":
|
||||
print(f"skip {name}: not a local model")
|
||||
continue
|
||||
|
||||
def tokenizer_check(model_name: str = name) -> object:
|
||||
from application.parser.tokenization import get_token_counter
|
||||
|
||||
counter = get_token_counter(model_name)
|
||||
if counter.name == "cl100k_base":
|
||||
raise RuntimeError("tokenizer missing from the cache; chunking fell back to cl100k")
|
||||
return f"{counter.name}, {counter.count('The quick brown fox')} tokens"
|
||||
|
||||
def embed_check(model_name: str = name) -> object:
|
||||
from application.vectorstore.embeddings_local import EmbeddingsWrapper
|
||||
|
||||
vector = EmbeddingsWrapper(model_name).embed_query("hello")
|
||||
return f"dimension {len(vector)}"
|
||||
|
||||
ok &= _check(f"tokenizer {name}", tokenizer_check)
|
||||
ok &= _check(f"embeddings {name}", embed_check)
|
||||
|
||||
if is_available("docling"):
|
||||
ok &= _check("docling PDF conversion", _docling_check)
|
||||
else:
|
||||
print("skip docling: not installed (slim image)")
|
||||
|
||||
return ok
|
||||
|
||||
|
||||
# A one-page PDF with a single text run; enough for the layout model to have
|
||||
# something to look at.
|
||||
_TINY_PDF = (
|
||||
b"%PDF-1.4\n"
|
||||
b"1 0 obj<</Type/Catalog/Pages 2 0 R>>endobj\n"
|
||||
b"2 0 obj<</Type/Pages/Kids[3 0 R]/Count 1>>endobj\n"
|
||||
b"3 0 obj<</Type/Page/Parent 2 0 R/MediaBox[0 0 300 144]/Contents 4 0 R"
|
||||
b"/Resources<</Font<</F1 5 0 R>>>>>>endobj\n"
|
||||
b"4 0 obj<</Length 58>>stream\n"
|
||||
b"BT /F1 18 Tf 20 100 Td (Offline verification page) Tj ET\n"
|
||||
b"endstream\nendobj\n"
|
||||
b"5 0 obj<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>endobj\n"
|
||||
b"trailer<</Root 1 0 R>>\n%%EOF\n"
|
||||
)
|
||||
|
||||
|
||||
def _docling_check() -> object:
|
||||
"""Convert a tiny PDF through docling; its models must come from DOCLING_ARTIFACTS_PATH."""
|
||||
import os
|
||||
import tempfile
|
||||
|
||||
from docling.datamodel.base_models import InputFormat
|
||||
from docling.datamodel.pipeline_options import PdfPipelineOptions
|
||||
from docling.document_converter import DocumentConverter, PdfFormatOption
|
||||
|
||||
from application.parser.file.docling_parser import _apply_inference_settings
|
||||
|
||||
# Same global docling settings the parser applies: torch.compile stays off
|
||||
# unless DOCLING_COMPILE_TORCH_MODELS asks for it (it needs a C++ toolchain).
|
||||
_apply_inference_settings()
|
||||
artifacts = os.environ.get("DOCLING_ARTIFACTS_PATH")
|
||||
options = PdfPipelineOptions(artifacts_path=artifacts, do_ocr=False, do_table_structure=True)
|
||||
converter = DocumentConverter(format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=options)})
|
||||
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as handle:
|
||||
handle.write(_TINY_PDF)
|
||||
path = handle.name
|
||||
try:
|
||||
text = converter.convert(path).document.export_to_markdown()
|
||||
finally:
|
||||
os.unlink(path)
|
||||
if "Offline verification" not in text:
|
||||
raise RuntimeError(f"unexpected conversion output: {text[:80]!r}")
|
||||
return f"models from {artifacts or 'default cache'}, {len(text)} chars"
|
||||
|
||||
|
||||
def main(argv: Optional[Sequence[str]] = None) -> int:
|
||||
logging.basicConfig(level=logging.WARNING, format="%(levelname)s %(message)s")
|
||||
models: List[str] = list(argv) if argv else list(DEFAULT_MODELS)
|
||||
if _network_reachable():
|
||||
print("note network is reachable; run with --network none to prove the offline path")
|
||||
else:
|
||||
print("note network unreachable, as intended")
|
||||
passed = verify(models)
|
||||
print("VERIFY OFFLINE: " + ("PASS" if passed else "FAIL"))
|
||||
return 0 if passed else 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv[1:]))
|
||||
@@ -104,7 +104,11 @@ def _read_repo_json(repo: str, filename: str) -> Optional[dict]:
|
||||
try:
|
||||
from huggingface_hub import hf_hub_download
|
||||
|
||||
with open(hf_hub_download(repo_id=repo, filename=filename), encoding="utf-8") as handle:
|
||||
try:
|
||||
path = hf_hub_download(repo_id=repo, filename=filename, local_files_only=True)
|
||||
except Exception: # noqa: BLE001 -- not cached: fetch it
|
||||
path = hf_hub_download(repo_id=repo, filename=filename)
|
||||
with open(path, encoding="utf-8") as handle:
|
||||
return json.load(handle)
|
||||
except Exception as exc:
|
||||
logger.debug("No %s for %s (%s)", filename, repo, exc)
|
||||
|
||||
@@ -4,6 +4,7 @@ import uuid
|
||||
from contextlib import contextmanager
|
||||
from typing import Any, Dict, List, Optional, Tuple
|
||||
|
||||
from application.core.optional_deps import require
|
||||
from application.core.settings import settings
|
||||
from application.vectorstore.base import BaseVectorStore
|
||||
from application.vectorstore.document_class import Document
|
||||
@@ -41,7 +42,8 @@ class MilvusStore(BaseVectorStore):
|
||||
def __init__(self, source_id: str = "", embeddings_key: str = "embeddings"):
|
||||
super().__init__()
|
||||
with _without_milvus_uri_env():
|
||||
from pymilvus import DataType, MilvusClient
|
||||
pymilvus = require("pymilvus", "VECTOR_STORE=milvus")
|
||||
DataType, MilvusClient = pymilvus.DataType, pymilvus.MilvusClient
|
||||
|
||||
self._DataType = DataType
|
||||
self._source_id = str(source_id).replace("application/indexes/", "").rstrip("/")
|
||||
|
||||
@@ -1,9 +1,24 @@
|
||||
services:
|
||||
frontend:
|
||||
build: ../frontend
|
||||
build:
|
||||
context: ../frontend
|
||||
target: dev
|
||||
environment:
|
||||
# Every VITE_* the app reads. A bare name is passed through only when it is
|
||||
# set in the shell or the --env-file, so an unset one does not reach the
|
||||
# container as an empty string and override the image's own default.
|
||||
- VITE_API_HOST=http://localhost:7091
|
||||
- VITE_API_STREAMING=$VITE_API_STREAMING
|
||||
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
|
||||
- VITE_BASE_URL
|
||||
- VITE_GOOGLE_CLIENT_ID
|
||||
- VITE_GOOGLE_PICKER_API_KEY
|
||||
- VITE_SHARE_POINT_CLIENT_ID
|
||||
- VITE_CONFLUENCE_CLIENT_ID
|
||||
- VITE_NOTIFICATION_TEXT
|
||||
- VITE_NOTIFICATION_LINK
|
||||
- VITE_ENABLE_VOICE_INPUT
|
||||
- VITE_DISABLE_SOURCE_FE
|
||||
- VITE_USE_V
|
||||
ports:
|
||||
- "5173:5173"
|
||||
depends_on:
|
||||
@@ -13,6 +28,7 @@ services:
|
||||
build:
|
||||
context: ../application
|
||||
args:
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
# Off by default; deployments running OCR_ENABLED=true with tesseract
|
||||
# must set INSTALL_TESSERACT=true before rebuilding (see docker-compose.yaml).
|
||||
@@ -40,6 +56,7 @@ services:
|
||||
build:
|
||||
context: ../application
|
||||
args:
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
# Off by default; deployments running OCR_ENABLED=true with tesseract
|
||||
# must set INSTALL_TESSERACT=true before rebuilding (see docker-compose.yaml).
|
||||
|
||||
@@ -1,12 +1,30 @@
|
||||
# Pre-built images from Docker Hub (mirrored at ghcr.io/arc53).
|
||||
# DOCSGPT_IMAGE_TAG develop (default, follows main) or a release, e.g. 0.20.0
|
||||
# DOCSGPT_IMAGE_VARIANT empty (default, slim) or -docling: docling parser engine,
|
||||
# its models, and tesseract baked in (OCR-ready)
|
||||
# Set them in ../.env or the shell. deployment/docker-compose-standalone.yaml is
|
||||
# the same stack without a git checkout.
|
||||
name: docsgpt-oss
|
||||
services:
|
||||
|
||||
frontend:
|
||||
image: arc53/docsgpt-fe:develop
|
||||
image: arc53/docsgpt-fe:${DOCSGPT_IMAGE_TAG:-develop}
|
||||
environment:
|
||||
# Every VITE_* the app reads. A bare name is passed through only when it is
|
||||
# set in the shell or the --env-file, so an unset one does not reach the
|
||||
# container as an empty string and override the image's own default.
|
||||
- VITE_API_HOST=http://localhost:7091
|
||||
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
|
||||
- VITE_GOOGLE_CLIENT_ID=${VITE_GOOGLE_CLIENT_ID:-}
|
||||
- VITE_BASE_URL
|
||||
- VITE_GOOGLE_CLIENT_ID
|
||||
- VITE_GOOGLE_PICKER_API_KEY
|
||||
- VITE_SHARE_POINT_CLIENT_ID
|
||||
- VITE_CONFLUENCE_CLIENT_ID
|
||||
- VITE_NOTIFICATION_TEXT
|
||||
- VITE_NOTIFICATION_LINK
|
||||
- VITE_ENABLE_VOICE_INPUT
|
||||
- VITE_DISABLE_SOURCE_FE
|
||||
- VITE_USE_V
|
||||
ports:
|
||||
- "5173:5173"
|
||||
depends_on:
|
||||
@@ -15,7 +33,7 @@ services:
|
||||
|
||||
backend:
|
||||
user: root
|
||||
image: arc53/docsgpt:develop
|
||||
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-develop}${DOCSGPT_IMAGE_VARIANT:-}
|
||||
env_file:
|
||||
- ../.env
|
||||
environment:
|
||||
@@ -38,7 +56,7 @@ services:
|
||||
|
||||
worker:
|
||||
user: root
|
||||
image: arc53/docsgpt:develop
|
||||
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-develop}${DOCSGPT_IMAGE_VARIANT:-}
|
||||
# `parsing` queue carries read_document/parse_document; required for its await to resolve.
|
||||
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
env_file:
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
# DocsGPT from pre-built images, with no git checkout.
|
||||
#
|
||||
# curl -fsSLO https://raw.githubusercontent.com/arc53/DocsGPT/main/deployment/docker-compose-standalone.yaml
|
||||
# printf 'LLM_PROVIDER=docsgpt\nVITE_API_STREAMING=true\nINTERNAL_KEY=%s\n' "$(openssl rand -hex 16)" > .env
|
||||
# docker compose -f docker-compose-standalone.yaml up -d
|
||||
#
|
||||
# INTERNAL_KEY is the shared secret the worker uses to hand finished indexes to
|
||||
# the API; without it every ingest fails with a 401 (setup.sh generates one).
|
||||
# open http://localhost:5173
|
||||
#
|
||||
# Every release also attaches this file as an asset. Settings come from .env
|
||||
# next to this file (any DocsGPT setting; the compose-internal service URLs
|
||||
# below take precedence). Data lives in named volumes, so `docker compose
|
||||
# down` keeps it and `docker compose down -v` removes it.
|
||||
#
|
||||
# DOCSGPT_IMAGE_TAG release to run, e.g. 0.20.0 (default: latest release);
|
||||
# develop follows the main branch
|
||||
# DOCSGPT_IMAGE_VARIANT empty (slim, default) or -docling: docling parser
|
||||
# engine, its models, and tesseract baked in (OCR-ready)
|
||||
# EMBEDDINGS_NAME defaults to granite here (this stack always starts on
|
||||
# fresh volumes, so there is no older index to keep
|
||||
# compatible); the code default stays mpnet for upgrades.
|
||||
name: docsgpt
|
||||
|
||||
services:
|
||||
frontend:
|
||||
image: arc53/docsgpt-fe:${DOCSGPT_IMAGE_TAG:-latest}
|
||||
env_file:
|
||||
- path: .env
|
||||
required: false
|
||||
environment:
|
||||
# Every VITE_* the app reads. A bare name is passed through only when it is
|
||||
# set in the shell or the --env-file, so an unset one does not reach the
|
||||
# container as an empty string and override the image's own default.
|
||||
- VITE_API_HOST=${VITE_API_HOST:-http://localhost:7091}
|
||||
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
|
||||
- VITE_BASE_URL
|
||||
- VITE_GOOGLE_CLIENT_ID
|
||||
- VITE_GOOGLE_PICKER_API_KEY
|
||||
- VITE_SHARE_POINT_CLIENT_ID
|
||||
- VITE_CONFLUENCE_CLIENT_ID
|
||||
- VITE_NOTIFICATION_TEXT
|
||||
- VITE_NOTIFICATION_LINK
|
||||
- VITE_ENABLE_VOICE_INPUT
|
||||
- VITE_DISABLE_SOURCE_FE
|
||||
- VITE_USE_V
|
||||
ports:
|
||||
- "5173:5173"
|
||||
depends_on:
|
||||
- backend
|
||||
|
||||
backend:
|
||||
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-latest}${DOCSGPT_IMAGE_VARIANT:-}
|
||||
# Same as docker-compose-hub.yaml: the data volumes are written by root so
|
||||
# any image tag works, including releases that predate the appuser-owned
|
||||
# /app/inputs, /app/indexes and /app/vectors directories.
|
||||
user: root
|
||||
env_file:
|
||||
- path: .env
|
||||
required: false
|
||||
environment:
|
||||
- CELERY_BROKER_URL=redis://redis:6379/0
|
||||
- CELERY_RESULT_BACKEND=redis://redis:6379/1
|
||||
- CACHE_REDIS_URL=redis://redis:6379/2
|
||||
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
||||
- EMBEDDINGS_NAME=${EMBEDDINGS_NAME:-ibm-granite/granite-embedding-311m-multilingual-r2}
|
||||
ports:
|
||||
- "7091:7091"
|
||||
volumes:
|
||||
- indexes:/app/indexes
|
||||
- inputs:/app/inputs
|
||||
- vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
restart: unless-stopped
|
||||
|
||||
worker:
|
||||
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-latest}${DOCSGPT_IMAGE_VARIANT:-}
|
||||
user: root
|
||||
# Consumes the default queue plus `parsing` (read_document) and `embeddings`
|
||||
# (query embedding); without the latter every search times out.
|
||||
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
|
||||
env_file:
|
||||
- path: .env
|
||||
required: false
|
||||
environment:
|
||||
- CELERY_BROKER_URL=redis://redis:6379/0
|
||||
- CELERY_RESULT_BACKEND=redis://redis:6379/1
|
||||
- CACHE_REDIS_URL=redis://redis:6379/2
|
||||
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
|
||||
- API_URL=http://backend:7091
|
||||
- EMBEDDINGS_NAME=${EMBEDDINGS_NAME:-ibm-granite/granite-embedding-311m-multilingual-r2}
|
||||
volumes:
|
||||
- indexes:/app/indexes
|
||||
- inputs:/app/inputs
|
||||
- vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
postgres:
|
||||
condition: service_healthy
|
||||
restart: unless-stopped
|
||||
|
||||
redis:
|
||||
image: redis:6-alpine
|
||||
restart: unless-stopped
|
||||
|
||||
postgres:
|
||||
image: postgres:16-alpine
|
||||
environment:
|
||||
- POSTGRES_USER=docsgpt
|
||||
- POSTGRES_PASSWORD=docsgpt
|
||||
- POSTGRES_DB=docsgpt
|
||||
volumes:
|
||||
- postgres_data:/var/lib/postgresql/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "pg_isready -U docsgpt -d docsgpt"]
|
||||
interval: 5s
|
||||
timeout: 5s
|
||||
retries: 10
|
||||
restart: unless-stopped
|
||||
|
||||
volumes:
|
||||
indexes:
|
||||
inputs:
|
||||
vectors:
|
||||
postgres_data:
|
||||
@@ -1,13 +1,29 @@
|
||||
name: docsgpt-oss
|
||||
services:
|
||||
frontend:
|
||||
build: ../frontend
|
||||
build:
|
||||
context: ../frontend
|
||||
# Vite dev server with hot reload over the bind mount below. The default
|
||||
# target (what Docker Hub publishes) is a static build behind nginx.
|
||||
target: dev
|
||||
volumes:
|
||||
- ../frontend/src:/app/src
|
||||
environment:
|
||||
# Every VITE_* the app reads. A bare name is passed through only when it is
|
||||
# set in the shell or the --env-file, so an unset one does not reach the
|
||||
# container as an empty string and override the image's own default.
|
||||
- VITE_API_HOST=http://localhost:7091
|
||||
- VITE_API_STREAMING=$VITE_API_STREAMING
|
||||
- VITE_GOOGLE_CLIENT_ID=$VITE_GOOGLE_CLIENT_ID
|
||||
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
|
||||
- VITE_BASE_URL
|
||||
- VITE_GOOGLE_CLIENT_ID
|
||||
- VITE_GOOGLE_PICKER_API_KEY
|
||||
- VITE_SHARE_POINT_CLIENT_ID
|
||||
- VITE_CONFLUENCE_CLIENT_ID
|
||||
- VITE_NOTIFICATION_TEXT
|
||||
- VITE_NOTIFICATION_LINK
|
||||
- VITE_ENABLE_VOICE_INPUT
|
||||
- VITE_DISABLE_SOURCE_FE
|
||||
- VITE_USE_V
|
||||
ports:
|
||||
- "5173:5173"
|
||||
depends_on:
|
||||
@@ -18,8 +34,10 @@ services:
|
||||
build:
|
||||
context: ../application
|
||||
args:
|
||||
# Bake the optional docling engine (layout-model OCR backend, structured
|
||||
# output) into the image: set INSTALL_DOCLING=true in ../.env or the shell.
|
||||
# Optional extras to bake in (comma-separated): docling, milvus. The
|
||||
# docling extra brings the layout-model parser/OCR backend and its
|
||||
# models. INSTALL_DOCLING=true is the older spelling of EXTRAS=docling.
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
# Bake the tesseract binary behind OCR_ENABLED=true (~35 MB): set
|
||||
# INSTALL_TESSERACT=true in ../.env or the shell (setup.sh does this
|
||||
@@ -53,6 +71,7 @@ services:
|
||||
build:
|
||||
context: ../application
|
||||
args:
|
||||
EXTRAS: ${EXTRAS:-}
|
||||
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
|
||||
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
|
||||
# Consumes the default queue AND the dedicated `parsing` (read_document /
|
||||
|
||||
@@ -94,13 +94,29 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
|
||||
pip install -r application/requirements.txt
|
||||
```
|
||||
|
||||
Optionally add the docling parser engine (OCR, `structured` output for
|
||||
`read_document`; the default anydoc engine does not need it):
|
||||
Dependencies are declared in `pyproject.toml` and locked in `uv.lock`; the
|
||||
`requirements*.txt` files are exported from that lock, so with
|
||||
[uv](https://docs.astral.sh/uv/) installed `uv sync` sets up the same
|
||||
environment (plus the test tools) in one step.
|
||||
|
||||
Optional extras are not installed by default. Add them when you need the
|
||||
feature (each file is the core set plus the extra):
|
||||
|
||||
```bash
|
||||
pip install -r application/requirements-docling.txt
|
||||
pip install -r application/requirements-docling.txt # docling parser engine: OCR backend, read_document structured output
|
||||
pip install -r application/requirements-milvus.txt # VECTOR_STORE=milvus
|
||||
# or with uv: uv sync --extra docling --extra milvus
|
||||
```
|
||||
|
||||
The docling file adds the PyTorch CPU index; pip handles that as-is, while
|
||||
`uv pip install -r` needs `UV_INDEX_STRATEGY=unsafe-best-match` (or use
|
||||
`uv sync --extra docling`, which reads the lock).
|
||||
|
||||
A feature whose extra is missing fails with the exact command to run.
|
||||
After changing `pyproject.toml`, run `uv lock` and
|
||||
`bash scripts/export_requirements.sh` so the exported files stay in sync
|
||||
(CI checks this).
|
||||
|
||||
5. **Run the Backend:**
|
||||
|
||||
For local development, run the ASGI composition under uvicorn. It serves the **whole** application, hot-reloads on source changes, and matches the production runtime:
|
||||
|
||||
@@ -17,9 +17,62 @@ Docker is the recommended method for deploying DocsGPT, providing a consistent a
|
||||
|
||||
**Important Note for Windows Users:** Docker Desktop on Windows generally requires the WSL 2 backend to function correctly, especially when using features like host networking which are utilized in DocsGPT's Docker Compose setup. Ensure WSL 2 is enabled and configured in Docker Desktop settings.
|
||||
|
||||
## Quickest Setup: Using DocsGPT Public API
|
||||
## Quickest Setup: Pre-built Images, No Checkout
|
||||
|
||||
The fastest way to try out DocsGPT is by using the public API endpoint. This requires minimal configuration and no local LLM setup.
|
||||
Every release publishes ready-to-run images to Docker Hub (`arc53/docsgpt`,
|
||||
`arc53/docsgpt-fe`) and GitHub Container Registry (`ghcr.io/arc53/docsgpt`,
|
||||
`ghcr.io/arc53/docsgpt-fe`) for `linux/amd64` and `linux/arm64`. The images
|
||||
contain everything the default configuration needs (embedding models,
|
||||
tokenizers, tiktoken's encoding), so a fresh container makes no downloads on
|
||||
first use. You do not need the source tree to run them:
|
||||
|
||||
1. **Download the standalone Compose file** (also attached to every
|
||||
[release](https://github.com/arc53/DocsGPT/releases)):
|
||||
|
||||
```bash
|
||||
mkdir docsgpt && cd docsgpt
|
||||
curl -fsSLO https://raw.githubusercontent.com/arc53/DocsGPT/main/deployment/docker-compose-standalone.yaml
|
||||
```
|
||||
|
||||
2. **Create a `.env` next to it** with your settings, for example the public API:
|
||||
|
||||
```bash
|
||||
printf 'LLM_PROVIDER=docsgpt\nVITE_API_STREAMING=true\nINTERNAL_KEY=%s\n' "$(openssl rand -hex 16)" > .env
|
||||
```
|
||||
|
||||
`INTERNAL_KEY` is the secret the worker uses to hand finished indexes to
|
||||
the API; without it every upload fails with a 401. `setup.sh` generates
|
||||
one for you, a hand-written `.env` has to include it. This stack runs the
|
||||
granite embedding model unless `.env` sets `EMBEDDINGS_NAME`; both granite
|
||||
and mpnet are baked into the image.
|
||||
|
||||
3. **Start it:**
|
||||
|
||||
```bash
|
||||
docker compose -f docker-compose-standalone.yaml up -d
|
||||
```
|
||||
|
||||
Then open [http://localhost:5173/](http://localhost:5173/). Data lives in
|
||||
named Docker volumes; `docker compose -f docker-compose-standalone.yaml down`
|
||||
keeps it and `down -v` removes it.
|
||||
|
||||
**Tags and variants.** `DOCSGPT_IMAGE_TAG` picks the version: a release such
|
||||
as `0.20.0`, `latest` (the newest release, the default) or `develop` (follows
|
||||
the `main` branch). `DOCSGPT_IMAGE_VARIANT` picks the flavour: empty for the
|
||||
slim default image, or `-docling` for the image with the docling parser
|
||||
engine, its models and tesseract baked in (needed for OCR of scanned
|
||||
documents, see the [OCR guide](/Guides/ocr)). Both are read from `.env` or
|
||||
the shell, e.g. `DOCSGPT_IMAGE_TAG=0.20.0 DOCSGPT_IMAGE_VARIANT=-docling`.
|
||||
The same two variables drive `deployment/docker-compose-hub.yaml` in a
|
||||
checkout.
|
||||
|
||||
## Using the Source Checkout
|
||||
|
||||
With a clone of the repository, `deployment/docker-compose-hub.yaml` runs the
|
||||
same pre-built images while keeping your data in `application/indexes`,
|
||||
`application/inputs` and `application/vectors`, and `deployment/docker-compose.yaml`
|
||||
builds the images from your working tree (for local changes, or a build with
|
||||
extra packages: `EXTRAS=docling` in `.env`).
|
||||
|
||||
1. **Clone the DocsGPT Repository (if you haven't already):**
|
||||
|
||||
@@ -39,19 +92,26 @@ The fastest way to try out DocsGPT is by using the public API endpoint. This req
|
||||
```
|
||||
LLM_PROVIDER=docsgpt
|
||||
VITE_API_STREAMING=true
|
||||
INTERNAL_KEY=<any random string, e.g. openssl rand -hex 16>
|
||||
EMBEDDINGS_NAME=ibm-granite/granite-embedding-311m-multilingual-r2
|
||||
```
|
||||
|
||||
This minimal configuration tells DocsGPT to use the public API. For more advanced settings and other LLM options, refer to the [DocsGPT Settings Guide](/Deploying/DocsGPT-Settings).
|
||||
This minimal configuration tells DocsGPT to use the public API. The
|
||||
`EMBEDDINGS_NAME` line is what `setup.sh` writes for a new install; without
|
||||
it the code falls back to mpnet, the model earlier releases indexed with,
|
||||
so that an upgraded deployment keeps its existing index working. For more advanced settings and other LLM options, refer to the [DocsGPT Settings Guide](/Deploying/DocsGPT-Settings).
|
||||
|
||||
4. **Launch DocsGPT with Docker Compose:**
|
||||
|
||||
Navigate to the root directory of the DocsGPT repository in your terminal and run:
|
||||
|
||||
```bash
|
||||
docker compose --env-file .env -f deployment/docker-compose.yaml up -d
|
||||
docker compose --env-file .env -f deployment/docker-compose-hub.yaml up -d
|
||||
```
|
||||
|
||||
The `-d` flag runs Docker Compose in detached mode (in the background).
|
||||
To build the images from your working tree instead of pulling them, use
|
||||
`deployment/docker-compose.yaml` with `up --build -d`.
|
||||
|
||||
5. **Access DocsGPT in your browser:**
|
||||
|
||||
@@ -62,7 +122,7 @@ The fastest way to try out DocsGPT is by using the public API endpoint. This req
|
||||
To stop the application, navigate to the same directory in your terminal and run:
|
||||
|
||||
```bash
|
||||
docker compose -f deployment/docker-compose.yaml down
|
||||
docker compose -f deployment/docker-compose-hub.yaml down
|
||||
```
|
||||
|
||||
## Optional Ollama Setup (Local Models)
|
||||
@@ -84,11 +144,11 @@ There are two Ollama optional files:
|
||||
|
||||
**CPU:**
|
||||
```bash
|
||||
docker compose --env-file .env -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml up -d
|
||||
docker compose --env-file .env -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml up -d
|
||||
```
|
||||
**GPU:**
|
||||
```bash
|
||||
docker compose --env-file .env -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml up -d
|
||||
docker compose --env-file .env -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml up -d
|
||||
```
|
||||
|
||||
3. **Pull the Ollama Model:**
|
||||
@@ -96,11 +156,11 @@ There are two Ollama optional files:
|
||||
**Crucially, after launching with Ollama, you need to pull the desired model into the Ollama container.** Find the `LLM_NAME` you configured in your `.env` file (e.g., `llama3.2:1b`). Then execute the following command to pull the model *inside* the running Ollama container:
|
||||
|
||||
```bash
|
||||
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml exec -it ollama ollama pull <LLM_NAME>
|
||||
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml exec -it ollama ollama pull <LLM_NAME>
|
||||
```
|
||||
or (for GPU):
|
||||
```bash
|
||||
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml exec -it ollama ollama pull <LLM_NAME>
|
||||
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml exec -it ollama ollama pull <LLM_NAME>
|
||||
```
|
||||
Replace `<LLM_NAME>` with the actual model name from your `.env` file.
|
||||
|
||||
@@ -113,12 +173,12 @@ There are two Ollama optional files:
|
||||
To stop a DocsGPT setup launched with Ollama optional files, use `docker compose down` and include all the compose files used during the `up` command:
|
||||
|
||||
```bash
|
||||
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml down
|
||||
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml down
|
||||
```
|
||||
or
|
||||
|
||||
```bash
|
||||
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml down
|
||||
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml down
|
||||
```
|
||||
|
||||
**Important for GPU Usage:**
|
||||
|
||||
@@ -471,6 +471,7 @@ See [Embeddings](/Models/embeddings) for full guidance.
|
||||
| --- | --- | --- |
|
||||
| `EMBEDDINGS_NAME` | `huggingface_sentence-transformers/all-mpnet-base-v2` | The embedding model. New installs use `ibm-granite/granite-embedding-311m-multilingual-r2`. Changing it on a populated index requires `application.scripts.reembed`. |
|
||||
| `EMBEDDINGS_BASE_URL` | unset | Base URL of a remote OpenAI-compatible embeddings server. Setting it routes all embedding calls there. |
|
||||
| `EMBEDDINGS_THREADS` | unset (Docker image: `4`) | Threads one local FastEmbed/onnxruntime session may use. onnxruntime otherwise sizes its pool to the host's core count, which a CPU-limited container still reports, so the image pins it like `OMP_NUM_THREADS`. Raise it on a large dedicated worker. |
|
||||
| `EMBEDDINGS_KEY` | unset | Optional bearer token for the remote embeddings server. |
|
||||
| `EMBEDDINGS_MAX_INPUT_TOKENS` | unset | Truncate each remote embedding input to N tokens (guards servers that reject oversized inputs). |
|
||||
| `EMBEDDINGS_DELEGATE_TO_WORKER` | `true` | Embed queries on the Celery worker instead of loading a model in the API. Requires a worker consuming `EMBEDDINGS_QUEUE`; set `false` to run the API standalone. Ignored when `EMBEDDINGS_BASE_URL` is set. |
|
||||
|
||||
+27
-13
@@ -64,11 +64,17 @@ docker build --build-arg INSTALL_TESSERACT=true ./application
|
||||
`INSTALL_TESSERACT=true` in `.env` (or the shell) bakes tesseract plus the
|
||||
English pack into locally built backend and worker images; `setup.sh` writes
|
||||
it when you answer yes to the OCR question after choosing to build images
|
||||
locally. Pre-built Docker Hub images (`docker-compose-hub.yaml`) do not
|
||||
include it, so `setup.sh` leaves OCR at its default (off) for them; to OCR
|
||||
there, point `OCR_ENGINE=deepseek` at a DeepSeek-OCR endpoint or install
|
||||
`tesseract-ocr` in a derived image. With `OCR_ENABLED=true` and no binary on
|
||||
`PATH`, scanned pages fail with an install hint (text-layer documents are
|
||||
locally.
|
||||
|
||||
With pre-built images the switch is the image variant: every tag is
|
||||
published twice, slim (`arc53/docsgpt:<tag>`) and `-docling`
|
||||
(`arc53/docsgpt:<tag>-docling`), and the latter bakes tesseract, the docling
|
||||
engine and its models in. Set `DOCSGPT_IMAGE_VARIANT=-docling` in `.env` for
|
||||
`docker-compose-hub.yaml` or `docker-compose-standalone.yaml`; `setup.sh`
|
||||
writes it when you answer yes to the OCR question with Docker Hub images.
|
||||
Alternatively point `OCR_ENGINE=deepseek` at a DeepSeek-OCR endpoint, which
|
||||
needs no system package. With `OCR_ENABLED=true` and no binary on `PATH`,
|
||||
scanned pages fail with an install hint (text-layer documents are
|
||||
unaffected).
|
||||
|
||||
<Callout type="warning" emoji="⚠️">
|
||||
@@ -88,23 +94,31 @@ docling is not part of the base install, and OCR does not need it (see
|
||||
output:
|
||||
|
||||
```bash
|
||||
pip install -r application/requirements-docling.txt
|
||||
pip install -r application/requirements-docling.txt # or: uv sync --extra docling
|
||||
```
|
||||
|
||||
Docker images build without it by default; opt in with the build argument:
|
||||
That file is the core set plus the `docling` extra, exported from the same
|
||||
lock. On Linux it takes torch from the CPU-only PyTorch index, so the extra
|
||||
costs about 1.5 GB rather than the 2.7 GB the CUDA build of torch would; a
|
||||
GPU deployment can reinstall torch from PyPI on top.
|
||||
|
||||
Pre-built images: use the `-docling` variant (`arc53/docsgpt:<tag>-docling`,
|
||||
`DOCSGPT_IMAGE_VARIANT=-docling` in `.env`), which also bakes docling's
|
||||
layout, table-structure and RapidOCR models in so the first parse does not
|
||||
download them. Local builds opt in with the build argument:
|
||||
|
||||
```bash
|
||||
docker build --build-arg INSTALL_DOCLING=true ./application
|
||||
docker build --build-arg EXTRAS=docling ./application
|
||||
```
|
||||
|
||||
`deployment/docker-compose.yaml` forwards the same switch, so setting
|
||||
`INSTALL_DOCLING=true` in `.env` (or the shell) bakes docling into locally
|
||||
built backend and worker images; `setup.sh` offers it as a follow-up to the
|
||||
OCR question. Compose reads build arguments from the shell or from the
|
||||
`EXTRAS=docling` (or the older `INSTALL_DOCLING=true`) in `.env` (or the
|
||||
shell) bakes docling into locally built backend and worker images; `setup.sh`
|
||||
offers it as a follow-up to the OCR question. Compose reads build arguments from the shell or from the
|
||||
`.env` you pass with `--env-file .env` (not from the containers' `env_file`),
|
||||
so build with `docker compose --env-file .env -f deployment/docker-compose.yaml build`
|
||||
as `setup.sh` does. Pre-built Docker Hub images (`docker-compose-hub.yaml`)
|
||||
never include docling; install it in a derived image instead. Either way no
|
||||
as `setup.sh` does. Of the pre-built Docker Hub images only the slim default
|
||||
excludes docling; the `-docling` variant ships it with its models. Either way no
|
||||
code changes are needed — docling is picked up as
|
||||
the fallback engine (and, under `OCR_BACKEND=auto`, as the OCR backend) as
|
||||
soon as it is importable, and `DOC_PARSER_ENGINE=docling` makes it the
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
node_modules/
|
||||
dist/
|
||||
# Local overrides never belong in the image; .env.development and .env.production do.
|
||||
.env.local
|
||||
.env.*.local
|
||||
Dockerfile
|
||||
.dockerignore
|
||||
*.log
|
||||
+48
-4
@@ -1,11 +1,55 @@
|
||||
FROM node:22-bullseye-slim
|
||||
# DocsGPT frontend image: a static production build served by nginx.
|
||||
#
|
||||
# Vite inlines VITE_* settings at build time, so the old image ran the Vite
|
||||
# dev server just to read VITE_API_HOST from the container environment. This
|
||||
# image builds once and injects the container's VITE_* variables at start-up
|
||||
# instead (docker/40-runtime-env.sh writes them to /config.js, which the app
|
||||
# reads before its own bundle; see src/env.ts). Same env vars, same port.
|
||||
#
|
||||
# Targets:
|
||||
# (default) nginx serving the built bundle -- what Docker Hub publishes
|
||||
# dev the Vite dev server with hot reload, for docker-compose.yaml's
|
||||
# bind-mounted frontend (build: target: dev)
|
||||
|
||||
FROM node:22-alpine AS deps
|
||||
|
||||
WORKDIR /app
|
||||
COPY package*.json ./
|
||||
RUN npm install
|
||||
RUN npm ci --no-audit --no-fund
|
||||
|
||||
|
||||
FROM deps AS dev
|
||||
|
||||
COPY . .
|
||||
|
||||
# vite.config.ts polls the filesystem when DOCKER is set: native fs events do
|
||||
# not cross a Windows host into a Linux container.
|
||||
ENV DOCKER=1
|
||||
EXPOSE 5173
|
||||
CMD ["npm", "run", "dev", "--", "--host"]
|
||||
|
||||
CMD [ "npm", "run", "dev", "--" , "--host"]
|
||||
|
||||
FROM deps AS build
|
||||
|
||||
COPY . .
|
||||
# The image used to run the dev server, so .env.development was its set of
|
||||
# defaults (notification banner, Google client id, local API host). Keep them
|
||||
# as the production build's baseline; the container's VITE_* values override
|
||||
# any of them at start-up.
|
||||
RUN cp .env.development .env.production.local && \
|
||||
npm run build && \
|
||||
sed -i 's|<head>|<head><script src="/config.js"></script>|' dist/index.html
|
||||
|
||||
|
||||
FROM nginx:1.27-alpine
|
||||
|
||||
LABEL org.opencontainers.image.source="https://github.com/arc53/DocsGPT" \
|
||||
org.opencontainers.image.title="DocsGPT frontend" \
|
||||
org.opencontainers.image.licenses="MIT"
|
||||
|
||||
COPY docker/nginx.conf /etc/nginx/conf.d/default.conf
|
||||
COPY docker/40-runtime-env.sh /docker-entrypoint.d/40-runtime-env.sh
|
||||
RUN chmod +x /docker-entrypoint.d/40-runtime-env.sh
|
||||
COPY --from=build /app/dist /usr/share/nginx/html
|
||||
|
||||
# Same port the dev server used, so compose files and docs keep working.
|
||||
EXPOSE 5173
|
||||
@@ -0,0 +1,25 @@
|
||||
#!/bin/sh
|
||||
# Expose the container's VITE_* environment to the static bundle.
|
||||
#
|
||||
# nginx's entrypoint runs everything in /docker-entrypoint.d before serving.
|
||||
# The generated /config.js is loaded by index.html ahead of the app bundle and
|
||||
# read by src/env.ts, so VITE_API_HOST and friends can differ per deployment
|
||||
# without rebuilding the image.
|
||||
set -eu
|
||||
|
||||
out=/usr/share/nginx/html/config.js
|
||||
{
|
||||
printf 'window.__DOCSGPT_ENV__ = {'
|
||||
first=1
|
||||
env | grep -E '^VITE_[A-Za-z0-9_]+=.' | while IFS='=' read -r key value; do
|
||||
# Empty values are skipped above (=.) so a compose passthrough like
|
||||
# ${VITE_X:-} leaves the build-time default in place.
|
||||
# JSON-escape backslashes and double quotes; values are plain URLs/ids.
|
||||
escaped=$(printf '%s' "$value" | sed 's/\\/\\\\/g; s/"/\\"/g')
|
||||
if [ "$first" -eq 1 ]; then first=0; else printf ','; fi
|
||||
printf '"%s":"%s"' "$key" "$escaped"
|
||||
done
|
||||
printf '};\n'
|
||||
} > "$out"
|
||||
|
||||
echo "runtime-env: wrote $(grep -o 'VITE_[A-Za-z0-9_]*' "$out" | wc -l | tr -d ' ') VITE_* values to /config.js"
|
||||
@@ -0,0 +1,25 @@
|
||||
server {
|
||||
listen 5173;
|
||||
server_name _;
|
||||
root /usr/share/nginx/html;
|
||||
index index.html;
|
||||
|
||||
gzip on;
|
||||
gzip_types text/plain text/css application/javascript application/json image/svg+xml;
|
||||
|
||||
# Hashed bundle assets are immutable; index.html and config.js are not.
|
||||
location /assets/ {
|
||||
add_header Cache-Control "public, max-age=31536000, immutable";
|
||||
try_files $uri =404;
|
||||
}
|
||||
|
||||
location = /config.js {
|
||||
add_header Cache-Control "no-store";
|
||||
}
|
||||
|
||||
# Single-page app: every unknown path renders index.html.
|
||||
location / {
|
||||
add_header Cache-Control "no-cache";
|
||||
try_files $uri $uri/ /index.html;
|
||||
}
|
||||
}
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import './locale/i18n';
|
||||
|
||||
import { useState } from 'react';
|
||||
@@ -108,8 +109,8 @@ export default function App() {
|
||||
const saved = localStorage.getItem('showNotification');
|
||||
return saved ? JSON.parse(saved) : true;
|
||||
});
|
||||
const notificationText = import.meta.env.VITE_NOTIFICATION_TEXT;
|
||||
const notificationLink = import.meta.env.VITE_NOTIFICATION_LINK;
|
||||
const notificationText = envVar('VITE_NOTIFICATION_TEXT');
|
||||
const notificationLink = envVar('VITE_NOTIFICATION_LINK');
|
||||
// Hide the changelog banner on public share routes — those pages are
|
||||
// embedded / shared externally and shouldn't carry product chrome.
|
||||
const isPublicShareRoute =
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import { createAsyncThunk, createSlice, PayloadAction } from '@reduxjs/toolkit';
|
||||
|
||||
import {
|
||||
@@ -26,7 +27,7 @@ const initialState: ConversationState = {
|
||||
conversationId: null,
|
||||
};
|
||||
|
||||
const API_STREAMING = import.meta.env.VITE_API_STREAMING === 'true';
|
||||
const API_STREAMING = envVar('VITE_API_STREAMING') === 'true';
|
||||
|
||||
let abortController: AbortController | null = null;
|
||||
export function handlePreviewAbort() {
|
||||
|
||||
@@ -1,7 +1,8 @@
|
||||
import { envVar } from '@/env';
|
||||
import { withThrottle, type FetchLike } from './throttle';
|
||||
|
||||
export const baseURL =
|
||||
import.meta.env.VITE_API_HOST || 'https://docsapi.arc53.com';
|
||||
envVar('VITE_API_HOST') || 'https://docsapi.arc53.com';
|
||||
|
||||
const getHeaders = (
|
||||
token: string | null,
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import React, { useState, useEffect } from 'react';
|
||||
import { useTranslation } from 'react-i18next';
|
||||
import drivePickerImport from 'react-google-drive-picker';
|
||||
@@ -121,9 +122,9 @@ const GoogleDrivePicker: React.FC<GoogleDrivePickerProps> = ({
|
||||
}
|
||||
|
||||
try {
|
||||
const clientId: string = import.meta.env.VITE_GOOGLE_CLIENT_ID;
|
||||
const clientId: string = envVar('VITE_GOOGLE_CLIENT_ID');
|
||||
const developerKey: string =
|
||||
import.meta.env.VITE_GOOGLE_PICKER_API_KEY ?? '';
|
||||
envVar('VITE_GOOGLE_PICKER_API_KEY') ?? '';
|
||||
|
||||
// Derive appId from clientId (extract numeric part before first dash)
|
||||
const appId = clientId ? clientId.split('-')[0] : null;
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import {
|
||||
useCallback,
|
||||
useEffect,
|
||||
@@ -63,7 +64,7 @@ const LIVE_TRANSCRIPTION_TIMESLICE_MS = 1000;
|
||||
const LIVE_CAPTURE_SAMPLE_RATE = 16000;
|
||||
const LIVE_CAPTURE_MAX_BUFFER_SECONDS = 20;
|
||||
const LIVE_SILENCE_RMS_THRESHOLD = 0.015;
|
||||
const ENABLE_VOICE_INPUT = import.meta.env.VITE_ENABLE_VOICE_INPUT === 'true';
|
||||
const ENABLE_VOICE_INPUT = envVar('VITE_ENABLE_VOICE_INPUT') === 'true';
|
||||
|
||||
type AudioContextWindow = Window &
|
||||
typeof globalThis & {
|
||||
@@ -526,7 +527,7 @@ export default function MessageInput({
|
||||
if (supported.length === 0) return;
|
||||
const files = supported;
|
||||
|
||||
const apiHost = import.meta.env.VITE_API_HOST;
|
||||
const apiHost = envVar('VITE_API_HOST');
|
||||
|
||||
if (files.length > 1) {
|
||||
const formData = new FormData();
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import 'katex/dist/katex.min.css';
|
||||
|
||||
import { Pencil } from 'lucide-react';
|
||||
@@ -37,7 +38,7 @@ import ResearchProgress from './ResearchProgress';
|
||||
import { ToolCallsType } from './types';
|
||||
import { wikiWriteActionKey, wikiWritePath } from './wikiToolCall';
|
||||
|
||||
const DisableSourceFE = import.meta.env.VITE_DISABLE_SOURCE_FE || false;
|
||||
const DisableSourceFE = envVar('VITE_DISABLE_SOURCE_FE') === 'true';
|
||||
|
||||
const ConversationBubble = forwardRef<
|
||||
HTMLDivElement,
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import {
|
||||
createAsyncThunk,
|
||||
createListenerMiddleware,
|
||||
@@ -90,8 +91,8 @@ const initialState: ConversationState = {
|
||||
conversationId: null,
|
||||
};
|
||||
|
||||
const API_STREAMING = import.meta.env.VITE_API_STREAMING === 'true';
|
||||
const USE_V1_API = import.meta.env.VITE_USE_V1_API === 'true';
|
||||
const API_STREAMING = envVar('VITE_API_STREAMING') === 'true';
|
||||
const USE_V1_API = envVar('VITE_USE_V1_API') === 'true';
|
||||
|
||||
let abortController: AbortController | null = null;
|
||||
export function handleAbort() {
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import { createSlice } from '@reduxjs/toolkit';
|
||||
import type { PayloadAction } from '@reduxjs/toolkit';
|
||||
import store from '../store';
|
||||
@@ -12,7 +13,7 @@ import {
|
||||
clearAttachments,
|
||||
} from '../upload/uploadSlice';
|
||||
|
||||
const API_STREAMING = import.meta.env.VITE_API_STREAMING === 'true';
|
||||
const API_STREAMING = envVar('VITE_API_STREAMING') === 'true';
|
||||
interface SharedConversationsType {
|
||||
queries: Query[];
|
||||
apiKey?: string;
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
/**
|
||||
* Runtime-overridable build settings.
|
||||
*
|
||||
* Vite inlines `import.meta.env.VITE_*` at build time, which forced the
|
||||
* Docker image to run the dev server so `VITE_API_HOST` could change per
|
||||
* deployment. The production image serves a static build instead and writes
|
||||
* the container's `VITE_*` environment into `window.__DOCSGPT_ENV__` (see
|
||||
* frontend/docker/40-runtime-env.sh). That object wins over the build-time
|
||||
* value; outside Docker nothing sets it and the build-time value applies.
|
||||
*/
|
||||
|
||||
declare global {
|
||||
interface Window {
|
||||
__DOCSGPT_ENV__?: Record<string, string | undefined>;
|
||||
}
|
||||
}
|
||||
|
||||
export function envVar(name: string): string {
|
||||
const runtime =
|
||||
typeof window !== 'undefined' ? window.__DOCSGPT_ENV__?.[name] : undefined;
|
||||
if (runtime !== undefined && runtime !== '') return runtime;
|
||||
const buildTime = (import.meta.env as Record<string, unknown>)[name];
|
||||
return typeof buildTime === 'string' ? buildTime : '';
|
||||
}
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import { useEffect, useState } from 'react';
|
||||
import { useTranslation } from 'react-i18next';
|
||||
import { useSelector } from 'react-redux';
|
||||
@@ -13,7 +14,7 @@ import { ActiveState } from '../models/misc';
|
||||
import { selectToken } from '../preferences/preferenceSlice';
|
||||
import ConfirmationModal from './ConfirmationModal';
|
||||
|
||||
const baseURL = import.meta.env.VITE_BASE_URL;
|
||||
const baseURL = envVar('VITE_BASE_URL');
|
||||
|
||||
type AgentDetailsModalProps = {
|
||||
agent: Agent;
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import { useCallback, useEffect, useState } from 'react';
|
||||
import { nanoid } from '@reduxjs/toolkit';
|
||||
import { useDropzone } from 'react-dropzone';
|
||||
@@ -575,7 +576,7 @@ function Upload({
|
||||
JSON.stringify(optionsToConfig(retrievalOptions)),
|
||||
);
|
||||
|
||||
const apiHost = import.meta.env.VITE_API_HOST;
|
||||
const apiHost = envVar('VITE_API_HOST');
|
||||
const xhr = new XMLHttpRequest();
|
||||
|
||||
dispatch(
|
||||
@@ -706,7 +707,7 @@ function Upload({
|
||||
|
||||
formData.append('data', JSON.stringify(configData));
|
||||
|
||||
const apiHost: string = import.meta.env.VITE_API_HOST;
|
||||
const apiHost: string = envVar('VITE_API_HOST');
|
||||
const endpoint =
|
||||
ingestor.type === 'local_file'
|
||||
? `${apiHost}/api/upload`
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { envVar } from '@/env';
|
||||
import CrawlerIcon from '../../assets/crawler.svg';
|
||||
import FileUploadIcon from '../../assets/file_upload.svg';
|
||||
import UrlIcon from '../../assets/url.svg';
|
||||
@@ -146,7 +147,7 @@ export const IngestorFormSchemas: IngestorSchema[] = [
|
||||
icon: DriveIcon,
|
||||
heading: 'Upload from Google Drive',
|
||||
validate: () => {
|
||||
const googleClientId = import.meta.env.VITE_GOOGLE_CLIENT_ID;
|
||||
const googleClientId = envVar('VITE_GOOGLE_CLIENT_ID');
|
||||
return !!googleClientId;
|
||||
},
|
||||
fields: [
|
||||
@@ -208,7 +209,7 @@ export const IngestorFormSchemas: IngestorSchema[] = [
|
||||
icon: SharePoint,
|
||||
heading: 'Upload from Share Point',
|
||||
validate: () => {
|
||||
const sharePointClientId = import.meta.env.VITE_SHARE_POINT_CLIENT_ID;
|
||||
const sharePointClientId = envVar('VITE_SHARE_POINT_CLIENT_ID');
|
||||
return !!sharePointClientId;
|
||||
},
|
||||
fields: [
|
||||
@@ -226,7 +227,7 @@ export const IngestorFormSchemas: IngestorSchema[] = [
|
||||
icon: ConfluenceIcon,
|
||||
heading: 'Upload from Confluence',
|
||||
validate: () => {
|
||||
const confluenceClientId = import.meta.env.VITE_CONFLUENCE_CLIENT_ID;
|
||||
const confluenceClientId = envVar('VITE_CONFLUENCE_CLIENT_ID');
|
||||
return !!confluenceClientId;
|
||||
},
|
||||
fields: [
|
||||
|
||||
+149
@@ -0,0 +1,149 @@
|
||||
[project]
|
||||
name = "docsgpt"
|
||||
# Bump together with application/version.py and frontend/package.json.
|
||||
version = "0.19.0"
|
||||
description = "DocsGPT backend: chat with your documents, agents, and tools."
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.12"
|
||||
license = { file = "LICENSE" }
|
||||
|
||||
# Direct dependencies only. Transitive pins live in uv.lock; the pip-facing
|
||||
# files under application/ (requirements*.txt) are exported from that lock by
|
||||
# scripts/export_requirements.sh and must not be edited by hand.
|
||||
dependencies = [
|
||||
"a2wsgi==1.10.10",
|
||||
"alembic>=1.13,<2",
|
||||
"anthropic==0.121.0",
|
||||
"beautifulsoup4==4.15.0",
|
||||
"boto3==1.43.67",
|
||||
"cel-python==0.5.0",
|
||||
"celery==5.6.3",
|
||||
"celery-redbeat==2.4.2",
|
||||
"croniter==6.2.4",
|
||||
"cryptography==50.0.0",
|
||||
"dataclasses-json==0.6.7",
|
||||
"daytona==0.205.1",
|
||||
"ddgs>=8.0.0",
|
||||
"defusedxml==0.7.1",
|
||||
"docx2txt==0.9",
|
||||
"elevenlabs==2.62.0",
|
||||
"faiss-cpu==1.15.0",
|
||||
"fast-ebook",
|
||||
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
|
||||
# with no model downloads. The docling engine is the `docling` extra.
|
||||
"firecrawl-anydoc==0.2.3",
|
||||
"fastembed==0.8.0",
|
||||
"fastmcp==3.4.6",
|
||||
"Flask==3.1.3",
|
||||
"flask-restx==1.3.2",
|
||||
"google-api-python-client==2.198.0",
|
||||
"google-auth-oauthlib==1.4.0",
|
||||
"google-genai==2.17.0",
|
||||
"gTTS==2.5.4",
|
||||
"gunicorn==26.0.0",
|
||||
"jinja2==3.1.6",
|
||||
"kombu==5.6.2",
|
||||
"markdownify==1.2.3",
|
||||
"msal==1.37.0",
|
||||
"networkx==3.6.1",
|
||||
"numpy==2.5.1",
|
||||
# fastembed's runtime: local embeddings execute on it.
|
||||
"onnxruntime==1.28.0",
|
||||
"openai==2.53.0",
|
||||
"openapi3-parser==1.1.22",
|
||||
# pandas reads .xlsx through openpyxl but does not depend on it.
|
||||
"openpyxl==3.1.5",
|
||||
"opentelemetry-distro>=0.50b0,<1",
|
||||
"opentelemetry-exporter-otlp>=1.29.0,<2",
|
||||
"opentelemetry-instrumentation-celery>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-flask>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-logging>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-redis>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-requests>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
|
||||
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
|
||||
"pandas==3.0.5",
|
||||
"pdf2image>=1.17.0",
|
||||
"pgvector>=0.5,<1",
|
||||
"pillow==12.3.0",
|
||||
"praw==8.0.2",
|
||||
"psycopg[binary,pool]>=3.1,<4",
|
||||
"pydantic",
|
||||
"pydantic-settings",
|
||||
"pypdf==6.15.0",
|
||||
"pypdfium2==5.12.1",
|
||||
"python-dateutil==2.9.0.post0",
|
||||
"python-dotenv",
|
||||
"python-jose==3.5.0",
|
||||
"python-pptx==1.0.2",
|
||||
"PyYAML",
|
||||
"qdrant-client==1.19.0",
|
||||
"redis==7.4.0",
|
||||
"requests==2.34.2",
|
||||
"retry==0.9.2",
|
||||
"sqlalchemy>=2.0,<3",
|
||||
"starlette>=1.0,<2",
|
||||
"tiktoken==0.13.0",
|
||||
"tldextract==5.3.2",
|
||||
"tokenizers==0.22.2",
|
||||
"tqdm==4.67.3",
|
||||
"uvicorn[standard]>=0.30,<1",
|
||||
"uvicorn-worker>=0.4,<1",
|
||||
"websocket-client==1.9.0",
|
||||
"werkzeug>=3.1.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
|
||||
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
|
||||
# attachment parsing, and read_document's `structured` output. Pulls torch and
|
||||
# transformers; on Linux torch resolves from the CPU-only PyTorch index (see
|
||||
# [tool.uv.sources]) so the extra does not drag the CUDA stack in.
|
||||
docling = [
|
||||
"docling==2.119.0",
|
||||
"rapidocr==3.9.2",
|
||||
# docling's model stack. Declared here (not left transitive) so the pins
|
||||
# hold and the CPU index source below applies. transformers is capped by
|
||||
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
|
||||
"torch==2.11.0",
|
||||
"torchvision==0.26.0",
|
||||
"transformers==5.8.1",
|
||||
]
|
||||
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
|
||||
milvus = [
|
||||
"pymilvus==3.0.1",
|
||||
"milvus-lite==3.2.0; sys_platform != 'win32'",
|
||||
]
|
||||
|
||||
[dependency-groups]
|
||||
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
|
||||
dev = [
|
||||
"pytest>=8.0.0",
|
||||
"pytest-asyncio>=0.23",
|
||||
"pytest-cov>=4.1.0",
|
||||
"pytest-xdist>=3.5",
|
||||
"coverage>=7.4.0",
|
||||
"pytest-postgresql>=6.0.0",
|
||||
"jupyter-client>=8.0",
|
||||
"python-docx>=1.1",
|
||||
"reportlab>=4.0,<5",
|
||||
"ruff",
|
||||
]
|
||||
|
||||
[tool.uv]
|
||||
# The repo is run in place (`application.*` imported from the checkout), not
|
||||
# installed as a distribution.
|
||||
package = false
|
||||
|
||||
[[tool.uv.index]]
|
||||
name = "pytorch-cpu"
|
||||
url = "https://download.pytorch.org/whl/cpu"
|
||||
explicit = true
|
||||
|
||||
[tool.uv.sources]
|
||||
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
|
||||
# wheels). The docling extra runs its models on CPU, so take torch from the
|
||||
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
|
||||
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
||||
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
|
||||
Executable
+68
@@ -0,0 +1,68 @@
|
||||
#!/usr/bin/env bash
|
||||
# Regenerate the pip-facing requirements files from uv.lock.
|
||||
#
|
||||
# pyproject.toml declares the direct dependencies and the optional extras;
|
||||
# uv.lock pins everything. pip users, the Dockerfile and CI install from the
|
||||
# exported files, so run this after any change to pyproject.toml or uv.lock:
|
||||
#
|
||||
# uv lock # or: uv lock --upgrade-package <name>
|
||||
# bash scripts/export_requirements.sh
|
||||
#
|
||||
# Each exported file is a complete environment (core plus the named extra),
|
||||
# so `pip install -r application/requirements-docling.txt` on its own works,
|
||||
# and installing it on top of requirements.txt only adds the extra's packages.
|
||||
set -euo pipefail
|
||||
|
||||
cd "$(dirname "$0")/.."
|
||||
|
||||
UV=(uv)
|
||||
if ! uv --version 2>/dev/null | grep -qE '^uv 0\.([89]|[1-9][0-9])\.'; then
|
||||
# `uv export` needs a current uv; run one through uvx without touching the
|
||||
# machine's install.
|
||||
UV=(uv tool run --from 'uv>=0.8' uv)
|
||||
fi
|
||||
|
||||
export_file() {
|
||||
local out="$1"; shift
|
||||
local header="$1"; shift
|
||||
local index_url="${INDEX_URL:-}"
|
||||
{
|
||||
echo "# GENERATED by scripts/export_requirements.sh from uv.lock -- do not edit."
|
||||
echo "#"
|
||||
# shellcheck disable=SC2001
|
||||
echo "$header" | sed 's/^/# /'
|
||||
echo
|
||||
# uv export records the wheel's origin in uv.lock but writes no index
|
||||
# directive; pip needs one to find the +cpu torch build.
|
||||
if [ -n "$index_url" ]; then echo "--extra-index-url $index_url"; echo; fi
|
||||
"${UV[@]}" export --frozen --no-hashes --no-dev --no-emit-project --no-header --quiet "$@"
|
||||
} > "$out"
|
||||
echo "wrote $out ($(grep -cE '^[A-Za-z0-9]' "$out") packages)"
|
||||
}
|
||||
|
||||
export_file application/requirements.txt \
|
||||
"Core runtime. Optional extras live in requirements-<extra>.txt:
|
||||
docling DOC_PARSER_ENGINE=docling, docling OCR backend, read_document structured output
|
||||
milvus VECTOR_STORE=milvus"
|
||||
|
||||
INDEX_URL=https://download.pytorch.org/whl/cpu export_file application/requirements-docling.txt \
|
||||
"Core runtime plus the docling extra: DOC_PARSER_ENGINE=docling, the docling
|
||||
OCR backend (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
|
||||
attachment parsing, and read_document's 'structured' output. The default
|
||||
anydoc engine needs none of this, and OCR itself does not either:
|
||||
OCR_ENABLED=true with the tesseract binary (or a DeepSeek-OCR endpoint) runs
|
||||
through application/parser/file/ocr_parser.py.
|
||||
On Linux torch comes from the CPU-only PyTorch index (no CUDA stack); a GPU
|
||||
deployment can reinstall torch from PyPI on top.
|
||||
pip resolves the extra index as expected. uv only takes a package from the
|
||||
first index that lists it, and the PyTorch index carries stale copies of
|
||||
common packages, so with uv either run 'uv sync --extra docling' (the lock
|
||||
pins the index per package) or set UV_INDEX_STRATEGY=unsafe-best-match.
|
||||
Docker: --build-arg EXTRAS=docling" \
|
||||
--extra docling
|
||||
|
||||
export_file application/requirements-milvus.txt \
|
||||
"Core runtime plus the milvus extra (VECTOR_STORE=milvus): pymilvus and the
|
||||
embedded milvus-lite server, which pulls pyarrow.
|
||||
Docker: --build-arg EXTRAS=milvus" \
|
||||
--extra milvus
|
||||
@@ -513,29 +513,32 @@ function Configure-DocProcessing {
|
||||
Write-ColorText "PDF-as-image parsing enabled." -ForegroundColor "Green"
|
||||
}
|
||||
|
||||
# OCR needs the tesseract binary, an optional system package that only
|
||||
# locally built images can include (INSTALL_TESSERACT build arg). The
|
||||
# pre-built Docker Hub images ship without it, so there OCR stays off
|
||||
# (its default) rather than being switched on to fail on every scan.
|
||||
if ($COMPOSE_FILE -ne $COMPOSE_FILE_LOCAL) {
|
||||
Write-ColorText "OCR for scanned PDFs and images stays off: the pre-built Docker Hub images do not include tesseract. To use OCR, rerun setup and choose option 5 (build images locally), or use a DeepSeek-OCR endpoint by adding OCR_ENABLED=true, OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> to .env." -ForegroundColor "Yellow"
|
||||
# OCR needs the tesseract binary. The default (slim) images ship without
|
||||
# it; the pre-built "-docling" image variant bakes tesseract, the docling
|
||||
# layout engine and its models in, so with Docker Hub images OCR means
|
||||
# switching the variant. Locally built images get it via build args.
|
||||
$ocr_enabled = Read-Host "Enable OCR for scanned PDFs and images? (y/N)"
|
||||
if (-not ($ocr_enabled -eq "y" -or $ocr_enabled -eq "Y")) {
|
||||
return
|
||||
}
|
||||
|
||||
$ocr_enabled = Read-Host "Enable OCR for scanned PDFs and images? (y/N)"
|
||||
if ($ocr_enabled -eq "y" -or $ocr_enabled -eq "Y") {
|
||||
"OCR_ENABLED=true" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
# Bakes tesseract into the locally built images (docker compose
|
||||
# --env-file .env build).
|
||||
"INSTALL_TESSERACT=true" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
Write-ColorText "OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env." -ForegroundColor "Green"
|
||||
$docling_ocr = Read-Host "Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N)"
|
||||
if ($docling_ocr -eq "y" -or $docling_ocr -eq "Y") {
|
||||
# Locally built images include docling via this build arg; it becomes
|
||||
# the OCR backend automatically (OCR_BACKEND=auto).
|
||||
"INSTALL_DOCLING=true" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
Write-ColorText "Docling will be built into locally built images (docker compose --env-file .env build). Pre-built Docker Hub images do not include it." -ForegroundColor "Green"
|
||||
}
|
||||
"OCR_ENABLED=true" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
if ($COMPOSE_FILE -ne $COMPOSE_FILE_LOCAL) {
|
||||
# Pre-built images: pull arc53/docsgpt:<tag>-docling instead of the
|
||||
# slim default (about 1.5 GB more to download).
|
||||
"DOCSGPT_IMAGE_VARIANT=-docling" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
Write-ColorText "OCR enabled. The -docling image variant will be pulled (tesseract, docling layout engine and its models included). For a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env." -ForegroundColor "Green"
|
||||
return
|
||||
}
|
||||
# Bakes tesseract into the locally built images (docker compose
|
||||
# --env-file .env build).
|
||||
"INSTALL_TESSERACT=true" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
Write-ColorText "OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env." -ForegroundColor "Green"
|
||||
$docling_ocr = Read-Host "Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N)"
|
||||
if ($docling_ocr -eq "y" -or $docling_ocr -eq "Y") {
|
||||
# Locally built images include docling via this build arg; it becomes
|
||||
# the OCR backend automatically (OCR_BACKEND=auto).
|
||||
"INSTALL_DOCLING=true" | Add-Content -Path $ENV_FILE -Encoding utf8
|
||||
Write-ColorText "Docling will be built into locally built images (docker compose --env-file .env build)." -ForegroundColor "Green"
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -367,29 +367,32 @@ configure_doc_processing() {
|
||||
echo -e "${GREEN}PDF-as-image parsing enabled.${NC}"
|
||||
fi
|
||||
|
||||
# OCR needs the tesseract binary, an optional system package that only
|
||||
# locally built images can include (INSTALL_TESSERACT build arg). The
|
||||
# pre-built Docker Hub images ship without it, so there OCR stays off
|
||||
# (its default) rather than being switched on to fail on every scan.
|
||||
if [[ "$COMPOSE_FILE" != "$COMPOSE_FILE_LOCAL" ]]; then
|
||||
echo -e "${YELLOW}OCR for scanned PDFs and images stays off: the pre-built Docker Hub images do not include tesseract. To use OCR, rerun setup and choose option 5 (build images locally), or use a DeepSeek-OCR endpoint by adding OCR_ENABLED=true, OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> to .env.${NC}"
|
||||
# OCR needs the tesseract binary. The default (slim) images ship without
|
||||
# it; the pre-built "-docling" image variant bakes tesseract, the docling
|
||||
# layout engine and its models in, so with Docker Hub images OCR means
|
||||
# switching the variant. Locally built images get it via build args.
|
||||
read -p "$(echo -e "${DEFAULT_FG}Enable OCR for scanned PDFs and images? (y/N): ${NC}")" ocr_enabled
|
||||
if [[ ! "$ocr_enabled" =~ ^[yY]$ ]]; then
|
||||
return
|
||||
fi
|
||||
|
||||
read -p "$(echo -e "${DEFAULT_FG}Enable OCR for scanned PDFs and images? (y/N): ${NC}")" ocr_enabled
|
||||
if [[ "$ocr_enabled" =~ ^[yY]$ ]]; then
|
||||
echo "OCR_ENABLED=true" >> "$ENV_FILE"
|
||||
# Bakes tesseract into the locally built images (docker compose
|
||||
# --env-file .env build).
|
||||
echo "INSTALL_TESSERACT=true" >> "$ENV_FILE"
|
||||
echo -e "${GREEN}OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env.${NC}"
|
||||
read -p "$(echo -e "${DEFAULT_FG}Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N): ${NC}")" docling_ocr
|
||||
if [[ "$docling_ocr" =~ ^[yY]$ ]]; then
|
||||
# Locally built images include docling via this build arg; it becomes
|
||||
# the OCR backend automatically (OCR_BACKEND=auto).
|
||||
echo "INSTALL_DOCLING=true" >> "$ENV_FILE"
|
||||
echo -e "${GREEN}Docling will be built into locally built images (docker compose --env-file .env build). Pre-built Docker Hub images do not include it.${NC}"
|
||||
fi
|
||||
echo "OCR_ENABLED=true" >> "$ENV_FILE"
|
||||
if [[ "$COMPOSE_FILE" != "$COMPOSE_FILE_LOCAL" ]]; then
|
||||
# Pre-built images: pull arc53/docsgpt:<tag>-docling instead of the
|
||||
# slim default (about 1.5 GB more to download).
|
||||
echo "DOCSGPT_IMAGE_VARIANT=-docling" >> "$ENV_FILE"
|
||||
echo -e "${GREEN}OCR enabled. The -docling image variant will be pulled (tesseract, docling layout engine and its models included). For a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env.${NC}"
|
||||
return
|
||||
fi
|
||||
# Bakes tesseract into the locally built images (docker compose
|
||||
# --env-file .env build).
|
||||
echo "INSTALL_TESSERACT=true" >> "$ENV_FILE"
|
||||
echo -e "${GREEN}OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env.${NC}"
|
||||
read -p "$(echo -e "${DEFAULT_FG}Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N): ${NC}")" docling_ocr
|
||||
if [[ "$docling_ocr" =~ ^[yY]$ ]]; then
|
||||
# Locally built images include docling via this build arg; it becomes
|
||||
# the OCR backend automatically (OCR_BACKEND=auto).
|
||||
echo "INSTALL_DOCLING=true" >> "$ENV_FILE"
|
||||
echo -e "${GREEN}Docling will be built into locally built images (docker compose --env-file .env build).${NC}"
|
||||
fi
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,60 @@
|
||||
"""Optional-extra bookkeeping: hints name the extra, require() explains absence."""
|
||||
|
||||
import sys
|
||||
import types
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from application.core import optional_deps
|
||||
|
||||
|
||||
class TestExtras:
|
||||
def test_every_extra_module_maps_back(self):
|
||||
for extra, modules in optional_deps.EXTRAS.items():
|
||||
for module in modules:
|
||||
assert optional_deps.extra_for(module) == extra
|
||||
|
||||
def test_submodule_resolves_to_its_extra(self):
|
||||
assert optional_deps.extra_for("docling.document_converter") == "docling"
|
||||
|
||||
def test_unknown_module_has_no_extra(self):
|
||||
assert optional_deps.extra_for("flask") is None
|
||||
|
||||
def test_install_hint_names_every_install_route(self):
|
||||
hint = optional_deps.install_hint("milvus")
|
||||
assert "requirements-milvus.txt" in hint
|
||||
assert "--extra milvus" in hint
|
||||
assert "EXTRAS=milvus" in hint
|
||||
|
||||
|
||||
class TestMissingMessage:
|
||||
def test_extra_module_points_at_the_extra(self):
|
||||
message = optional_deps.missing_message("pymilvus", "VECTOR_STORE=milvus")
|
||||
assert "pymilvus is not installed (VECTOR_STORE=milvus)" in message
|
||||
assert "'milvus' extra" in message
|
||||
assert optional_deps.install_hint("milvus") in message
|
||||
|
||||
def test_plain_module_gets_a_pip_line(self):
|
||||
assert optional_deps.missing_message("boto3") == "boto3 is not installed. Install it with: pip install boto3"
|
||||
|
||||
|
||||
class TestRequire:
|
||||
def test_returns_the_module_when_present(self):
|
||||
assert optional_deps.require("json").dumps({}) == "{}"
|
||||
|
||||
def test_absent_module_raises_with_hint(self):
|
||||
with patch.dict(sys.modules, {"pymilvus": None}):
|
||||
with pytest.raises(ImportError) as excinfo:
|
||||
optional_deps.require("pymilvus", "VECTOR_STORE=milvus")
|
||||
assert "'milvus' extra" in str(excinfo.value)
|
||||
|
||||
|
||||
class TestIsAvailable:
|
||||
def test_stubbed_module_counts_as_present(self):
|
||||
with patch.dict(sys.modules, {"docling": types.ModuleType("docling")}):
|
||||
assert optional_deps.is_available("docling.document_converter")
|
||||
|
||||
def test_blocked_module_counts_as_absent(self):
|
||||
with patch.dict(sys.modules, {"docling": None}):
|
||||
assert not optional_deps.is_available("docling")
|
||||
@@ -68,6 +68,19 @@ class TestDoclingParserInitParser:
|
||||
with pytest.raises(ImportError, match="docling is required"):
|
||||
parser._init_parser()
|
||||
|
||||
def test_init_parser_names_the_extra_when_docling_is_absent(self, monkeypatch):
|
||||
"""A missing parent package makes find_spec raise; the hint must still show."""
|
||||
import sys
|
||||
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
for name in [m for m in sys.modules if m == "docling" or m.startswith("docling.")]:
|
||||
monkeypatch.delitem(sys.modules, name)
|
||||
monkeypatch.setitem(sys.modules, "docling", None)
|
||||
|
||||
with pytest.raises(ImportError, match="requirements-docling.txt"):
|
||||
DoclingParser()._init_parser()
|
||||
|
||||
def test_init_parser_success(self):
|
||||
from application.parser.file.docling_parser import DoclingParser
|
||||
|
||||
|
||||
@@ -48,7 +48,7 @@ def _patch_validate_url(monkeypatch):
|
||||
@pytest.fixture(autouse=True)
|
||||
def _patch_tldextract(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
"application.parser.remote.crawler_markdown.tldextract.extract",
|
||||
"application.parser.remote.crawler_markdown._extract",
|
||||
_fake_extract,
|
||||
)
|
||||
|
||||
|
||||
@@ -1,6 +1,9 @@
|
||||
"""Chunk sizes must be counted in the embedding model's units, and splitting
|
||||
must never rewrite the text it splits."""
|
||||
|
||||
import sys
|
||||
import types
|
||||
|
||||
import pytest
|
||||
|
||||
from application.parser import tokenization
|
||||
@@ -298,3 +301,35 @@ class TestUnknownTokenCollapse:
|
||||
assert "".join(pieces) == text, "split must not lose or alter text"
|
||||
assert len(pieces) > 1, "a collapsed run must still be cut into pieces"
|
||||
assert all(counter.count(p) <= 20 for p in pieces)
|
||||
|
||||
|
||||
class TestTokenizerFile:
|
||||
"""The chunker's tokenizer must come from the hub cache without a network round trip."""
|
||||
|
||||
def test_cache_hit_makes_no_online_call(self, monkeypatch):
|
||||
calls = []
|
||||
|
||||
def fake_download(repo, filename, local_files_only=False):
|
||||
calls.append(local_files_only)
|
||||
return "/cache/tokenizer.json"
|
||||
|
||||
fake_hub = types.ModuleType("huggingface_hub")
|
||||
fake_hub.hf_hub_download = fake_download
|
||||
monkeypatch.setitem(sys.modules, "huggingface_hub", fake_hub)
|
||||
assert tokenization._tokenizer_file("org/model") == "/cache/tokenizer.json"
|
||||
assert calls == [True]
|
||||
|
||||
def test_cache_miss_falls_back_to_online(self, monkeypatch):
|
||||
calls = []
|
||||
|
||||
def fake_download(repo, filename, local_files_only=False):
|
||||
calls.append(local_files_only)
|
||||
if local_files_only:
|
||||
raise FileNotFoundError("not cached")
|
||||
return "/downloaded/tokenizer.json"
|
||||
|
||||
fake_hub = types.ModuleType("huggingface_hub")
|
||||
fake_hub.hf_hub_download = fake_download
|
||||
monkeypatch.setitem(sys.modules, "huggingface_hub", fake_hub)
|
||||
assert tokenization._tokenizer_file("org/model") == "/downloaded/tokenizer.json"
|
||||
assert calls == [True, False]
|
||||
@@ -94,3 +94,18 @@ class TestMain:
|
||||
with patch.object(prefetch_models, "prefetch", return_value=[]) as spy:
|
||||
prefetch_models.main(["granite-97m"])
|
||||
assert spy.call_args.args[1] == "/app/models"
|
||||
|
||||
|
||||
class TestPrefetchTiktoken:
|
||||
def test_warms_every_listed_encoding(self):
|
||||
"""The image sets TIKTOKEN_CACHE_DIR; warming fills it at build time."""
|
||||
fake = MagicMock()
|
||||
module = types.ModuleType("tiktoken")
|
||||
module.get_encoding = fake
|
||||
with patch.dict(sys.modules, {"tiktoken": module}):
|
||||
fetched = prefetch_models.prefetch_tiktoken()
|
||||
assert fetched == list(prefetch_models.TIKTOKEN_ENCODINGS)
|
||||
assert [c.args[0] for c in fake.call_args_list] == list(prefetch_models.TIKTOKEN_ENCODINGS)
|
||||
|
||||
def test_cl100k_is_the_encoding_token_counting_uses(self):
|
||||
assert "cl100k_base" in prefetch_models.TIKTOKEN_ENCODINGS
|
||||
@@ -0,0 +1,51 @@
|
||||
"""Offline verification: every check must pass, docling only when installed."""
|
||||
|
||||
import sys
|
||||
import types
|
||||
from unittest.mock import patch
|
||||
|
||||
from application.scripts import verify_offline
|
||||
|
||||
|
||||
def _fake_tiktoken(monkeypatch):
|
||||
module = types.ModuleType("tiktoken")
|
||||
module.get_encoding = lambda name: types.SimpleNamespace(encode=lambda text: [1, 2])
|
||||
monkeypatch.setitem(sys.modules, "tiktoken", module)
|
||||
|
||||
|
||||
class TestVerify:
|
||||
def test_passes_when_every_check_passes(self, monkeypatch, capsys):
|
||||
_fake_tiktoken(monkeypatch)
|
||||
counter = types.SimpleNamespace(name="org/model", count=lambda text: 4)
|
||||
with patch("application.parser.tokenization.get_token_counter", return_value=counter), \
|
||||
patch("application.vectorstore.embeddings_local.EmbeddingsWrapper") as wrapper, \
|
||||
patch.object(verify_offline, "is_available", return_value=False):
|
||||
wrapper.return_value.embed_query.return_value = [0.0] * 768
|
||||
assert verify_offline.verify(["ibm-granite/granite-embedding-311m-multilingual-r2"]) is True
|
||||
out = capsys.readouterr().out
|
||||
assert "ok tiktoken cl100k_base" in out
|
||||
assert "skip docling" in out
|
||||
|
||||
def test_fails_when_the_tokenizer_fell_back_to_cl100k(self, monkeypatch, capsys):
|
||||
"""A cache miss makes chunking silently use cl100k; that is a failed check."""
|
||||
_fake_tiktoken(monkeypatch)
|
||||
counter = types.SimpleNamespace(name="cl100k_base", count=lambda text: 4)
|
||||
with patch("application.parser.tokenization.get_token_counter", return_value=counter), \
|
||||
patch("application.vectorstore.embeddings_local.EmbeddingsWrapper") as wrapper, \
|
||||
patch.object(verify_offline, "is_available", return_value=False):
|
||||
wrapper.return_value.embed_query.return_value = [0.0] * 768
|
||||
assert verify_offline.verify(["ibm-granite/granite-embedding-311m-multilingual-r2"]) is False
|
||||
assert "FAIL tokenizer" in capsys.readouterr().out
|
||||
|
||||
def test_runs_the_docling_check_when_installed(self, monkeypatch):
|
||||
_fake_tiktoken(monkeypatch)
|
||||
with patch.object(verify_offline, "is_available", return_value=True), \
|
||||
patch.object(verify_offline, "_docling_check", return_value="models from /app/models/docling") as check:
|
||||
assert verify_offline.verify([]) is True
|
||||
check.assert_called_once()
|
||||
|
||||
def test_remote_models_are_skipped(self, monkeypatch, capsys):
|
||||
_fake_tiktoken(monkeypatch)
|
||||
with patch.object(verify_offline, "is_available", return_value=False):
|
||||
assert verify_offline.verify(["openai_text-embedding-ada-002"]) is True
|
||||
assert "skip openai_text-embedding-ada-002" in capsys.readouterr().out
|
||||
Reference in new issue
Block a user