Merge pull request #2729 from arc53/feat/slim-image-optional-deps

Slim default install and images: extras for docling/milvus, self-contained images, -docling variant, static frontend
This commit is contained in:
Alex authored and GitHub committed 2026-09-06 23:35:09 +01:00
commit 3e7f1f91b4
55 files changed
+11769 -411

No files matched your search

+80 -42
View File
@@ -7,106 +7,144 @@ on:
jobs:
build:
if: github.repository == 'arc53/DocsGPT'
# Publishing jobs run in a GitHub Actions environment so the registry
# credentials can be scoped to it and protection rules (required reviewers,
# branch restrictions) applied in the repository settings.
environment: docker-hub
env:
# Public namespace the compose files pull from; the login secret only
# authenticates the push.
DOCKERHUB_NAMESPACE: arc53
strategy:
matrix:
include:
- platform: linux/amd64
runner: ubuntu-latest
suffix: amd64
- platform: linux/arm64
runner: ubuntu-24.04-arm
suffix: arm64
runs-on: ${{ matrix.runner }}
platform: [linux/amd64, linux/arm64]
# "" is the slim default image; "-docling" bakes the docling parser
# engine, its models and tesseract in (OCR-ready).
variant: ["", "-docling"]
runs-on: ${{ matrix.platform == 'linux/arm64' && 'ubuntu-24.04-arm' || 'ubuntu-latest' }}
permissions:
contents: read
packages: write
steps:
- uses: actions/checkout@v4
- name: Set up QEMU # Only needed for emulation, not for native arm64 builds
if: matrix.platform == 'linux/arm64'
uses: docker/setup-qemu-action@v3
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
persist-credentials: false
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
with:
driver: docker-container
install: true
- name: Login to DockerHub
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to ghcr.io
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
registry: ghcr.io
username: ${{ github.repository_owner }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Image metadata (OCI labels)
id: meta
uses: docker/metadata-action@c299e40c65443455700f0fdfc63efafe5b349051 # v5.10.0
with:
images: |
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt
ghcr.io/${{ github.repository_owner }}/docsgpt
labels: |
org.opencontainers.image.title=DocsGPT${{ matrix.variant }}
org.opencontainers.image.version=${{ github.event.release.tag_name }}
- name: Build and push platform-specific images
uses: docker/build-push-action@v6
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
with:
file: './application/Dockerfile'
platforms: ${{ matrix.platform }}
context: ./application
push: true
build-args: |
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
tags: |
${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-${{ matrix.suffix }}
ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-${{ matrix.suffix }}
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:${{ github.event.release.tag_name }}${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
labels: ${{ steps.meta.outputs.labels }}
provenance: false
sbom: false
cache-from: type=registry,ref=${{ secrets.DOCKER_USERNAME }}/docsgpt:latest
cache-from: type=registry,ref=${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:latest${{ matrix.variant }}
cache-to: type=inline
manifest:
if: github.repository == 'arc53/DocsGPT'
# Publishing jobs run in a GitHub Actions environment so the registry
# credentials can be scoped to it and protection rules (required reviewers,
# branch restrictions) applied in the repository settings.
environment: docker-hub
env:
# Public namespace the compose files pull from; the login secret only
# authenticates the push.
DOCKERHUB_NAMESPACE: arc53
needs: build
strategy:
matrix:
variant: ["", "-docling"]
runs-on: ubuntu-latest
permissions:
packages: write
steps:
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
with:
driver: docker-container
install: true
- name: Login to DockerHub
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to ghcr.io
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
registry: ghcr.io
username: ${{ github.repository_owner }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Create and push manifest for DockerHub
- name: Create and push multi-arch manifests
env:
TAG: ${{ github.event.release.tag_name }}${{ matrix.variant }}
LATEST: latest${{ matrix.variant }}
run: |
set -e
docker manifest create ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }} \
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-arm64
docker manifest push ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}
docker manifest create ${{ secrets.DOCKER_USERNAME }}/docsgpt:latest \
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:${{ github.event.release.tag_name }}-arm64
docker manifest push ${{ secrets.DOCKER_USERNAME }}/docsgpt:latest
for repo in "$DOCKERHUB_NAMESPACE/docsgpt" "ghcr.io/${{ github.repository_owner }}/docsgpt"; do
for name in "$TAG" "$LATEST"; do
docker manifest create "$repo:$name" \
--amend "$repo:$TAG-amd64" \
--amend "$repo:$TAG-arm64"
docker manifest push "$repo:$name"
done
done
- name: Create and push manifest for ghcr.io
release-assets:
if: github.repository == 'arc53/DocsGPT'
needs: manifest
runs-on: ubuntu-latest
permissions:
contents: write
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
persist-credentials: false
- name: Attach the standalone compose file to the release
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
TAG: ${{ github.event.release.tag_name }}
run: |
set -e
docker manifest create ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }} \
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-arm64
docker manifest push ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}
docker manifest create ghcr.io/${{ github.repository_owner }}/docsgpt:latest \
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-amd64 \
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:${{ github.event.release.tag_name }}-arm64
docker manifest push ghcr.io/${{ github.repository_owner }}/docsgpt:latest
gh release upload "$TAG" deployment/docker-compose-standalone.yaml --clobber
+64 -33
View File
@@ -9,92 +9,123 @@ on:
jobs:
build:
if: github.repository == 'arc53/DocsGPT'
# Publishing jobs run in a GitHub Actions environment so the registry
# credentials can be scoped to it and protection rules (required reviewers,
# branch restrictions) applied in the repository settings.
environment: docker-hub
env:
# Public namespace the compose files pull from; the login secret only
# authenticates the push.
DOCKERHUB_NAMESPACE: arc53
strategy:
matrix:
include:
- platform: linux/amd64
runner: ubuntu-latest
suffix: amd64
- platform: linux/arm64
runner: ubuntu-24.04-arm
suffix: arm64
runs-on: ${{ matrix.runner }}
platform: [linux/amd64, linux/arm64]
# "" is the slim default image; "-docling" bakes the docling parser
# engine, its models and tesseract in (OCR-ready).
variant: ["", "-docling"]
runs-on: ${{ matrix.platform == 'linux/arm64' && 'ubuntu-24.04-arm' || 'ubuntu-latest' }}
permissions:
contents: read
packages: write
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
persist-credentials: false
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
with:
driver: docker-container
install: true
- name: Login to DockerHub
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to ghcr.io
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
registry: ghcr.io
username: ${{ github.repository_owner }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Image metadata (OCI labels)
id: meta
uses: docker/metadata-action@c299e40c65443455700f0fdfc63efafe5b349051 # v5.10.0
with:
images: |
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt
ghcr.io/${{ github.repository_owner }}/docsgpt
labels: |
org.opencontainers.image.title=DocsGPT${{ matrix.variant }}
org.opencontainers.image.version=develop
- name: Build and push platform-specific images
uses: docker/build-push-action@v6
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
with:
file: './application/Dockerfile'
platforms: ${{ matrix.platform }}
context: ./application
push: true
build-args: |
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
tags: |
${{ secrets.DOCKER_USERNAME }}/docsgpt:develop-${{ matrix.suffix }}
ghcr.io/${{ github.repository_owner }}/docsgpt:develop-${{ matrix.suffix }}
${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:develop${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
ghcr.io/${{ github.repository_owner }}/docsgpt:develop${{ matrix.variant }}-${{ matrix.platform == 'linux/arm64' && 'arm64' || 'amd64' }}
labels: ${{ steps.meta.outputs.labels }}
provenance: false
sbom: false
cache-from: type=registry,ref=${{ secrets.DOCKER_USERNAME }}/docsgpt:develop
cache-from: type=registry,ref=${{ env.DOCKERHUB_NAMESPACE }}/docsgpt:develop${{ matrix.variant }}
cache-to: type=inline
manifest:
if: github.repository == 'arc53/DocsGPT'
# Publishing jobs run in a GitHub Actions environment so the registry
# credentials can be scoped to it and protection rules (required reviewers,
# branch restrictions) applied in the repository settings.
environment: docker-hub
env:
# Public namespace the compose files pull from; the login secret only
# authenticates the push.
DOCKERHUB_NAMESPACE: arc53
needs: build
strategy:
matrix:
variant: ["", "-docling"]
runs-on: ubuntu-latest
permissions:
packages: write
steps:
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v3
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
with:
driver: docker-container
install: true
- name: Login to DockerHub
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to ghcr.io
uses: docker/login-action@v3
uses: docker/login-action@c94ce9fb468520275223c153574b00df6fe4bcc9 # v3.7.0
with:
registry: ghcr.io
username: ${{ github.repository_owner }}
password: ${{ secrets.GITHUB_TOKEN }}
- name: Create and push manifest for DockerHub
run: |
docker manifest create ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop \
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop-amd64 \
--amend ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop-arm64
docker manifest push ${{ secrets.DOCKER_USERNAME }}/docsgpt:develop
- name: Create and push manifest for ghcr.io
- name: Create and push multi-arch manifests
env:
TAG: develop${{ matrix.variant }}
run: |
docker manifest create ghcr.io/${{ github.repository_owner }}/docsgpt:develop \
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:develop-amd64 \
--amend ghcr.io/${{ github.repository_owner }}/docsgpt:develop-arm64
docker manifest push ghcr.io/${{ github.repository_owner }}/docsgpt:develop
set -e
for repo in "$DOCKERHUB_NAMESPACE/docsgpt" "ghcr.io/${{ github.repository_owner }}/docsgpt"; do
docker manifest create "$repo:$TAG" \
--amend "$repo:$TAG-amd64" \
--amend "$repo:$TAG-arm64"
docker manifest push "$repo:$TAG"
done
+66
View File
@@ -0,0 +1,66 @@
name: Verify the Docker image works offline
# Builds the backend image and runs its offline check with networking off, so
# a change that reintroduces a first-request download (a tokenizer, tiktoken's
# encoding, an embedding model) fails here instead of in an air-gapped install.
on:
workflow_dispatch:
pull_request:
paths:
- 'application/Dockerfile'
- 'application/.dockerignore'
- 'application/requirements*.txt'
- 'application/scripts/prefetch_models.py'
- 'application/scripts/verify_offline.py'
- 'application/vectorstore/model_registry.py'
- 'application/parser/tokenization.py'
- 'application/vectorstore/embeddings_local.py'
- '.github/workflows/docker-image-verify.yml'
permissions:
contents: read
jobs:
verify:
strategy:
matrix:
# "" is the slim default; "-docling" bakes docling, its models and
# tesseract in, so the conversion check in verify_offline runs too.
variant: ["", "-docling"]
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
persist-credentials: false
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@8d2750c68a42422c14e847fe6c8ac0403b4cbd6f # v3.12.0
- name: Build the image
uses: docker/build-push-action@10e90e3645eae34f1e60eeb005ba3a3d33f178e8 # v6.19.2
with:
file: ./application/Dockerfile
context: ./application
platforms: linux/amd64
load: true
tags: docsgpt:verify${{ matrix.variant }}
build-args: |
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
cache-from: type=gha,scope=verify${{ matrix.variant }}
cache-to: type=gha,mode=max,scope=verify${{ matrix.variant }}
- name: Image size
env:
IMAGE: docsgpt:verify${{ matrix.variant }}
run: |
docker image inspect "$IMAGE" --format '{{.Size}}' | awk '{printf "uncompressed: %.2f GB\n", $1/1e9}'
docker history "$IMAGE" --format '{{.Size}}\t{{.CreatedBy}}' | head -20
- name: Offline verification (no network)
env:
IMAGE: docsgpt:verify${{ matrix.variant }}
run: |
docker run --rm --network none "$IMAGE" \
python -m application.scripts.verify_offline
+17
View File
@@ -20,3 +20,20 @@ jobs:
uses: chartboost/ruff-action@v1
with:
version: 0.14.10
requirements-in-sync:
# application/requirements*.txt are exported from uv.lock; fail when a
# change to pyproject.toml or uv.lock was not re-exported.
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
persist-credentials: false
- uses: astral-sh/setup-uv@d0cc045d04ccac9d8b7881df0226f9e82c39688e # v6.8.0
- name: Re-export and diff
run: |
uv lock --check
bash scripts/export_requirements.sh
git diff --exit-code -- application/requirements.txt application/requirements-docling.txt application/requirements-milvus.txt
+12 -2
View File
@@ -29,10 +29,20 @@ Use these commands once the dev prerequisites above are satisfied.
```bash
source .venv/bin/activate # macOS/Linux
uv pip install -r application/requirements.txt # or: pip install -r application/requirements.txt
# Optional docling parser engine (OCR / read_document structured output):
# uv pip install -r application/requirements-docling.txt
# Optional extras (not installed by default; each file = core + the extra):
# uv pip install -r application/requirements-docling.txt # docling parser engine (OCR backend, structured output)
# uv pip install -r application/requirements-milvus.txt # VECTOR_STORE=milvus
# With uv alone: `uv sync --extra docling` (pyproject.toml + uv.lock are the source of truth).
# `uv pip install -r application/requirements-docling.txt` needs UV_INDEX_STRATEGY=unsafe-best-match
# (the file adds the PyTorch CPU index; prefer `uv sync --extra docling`).
```
Dependencies are declared in `pyproject.toml` and locked in `uv.lock`; the
`application/requirements*.txt` files are exported from the lock. To add or
bump a package: edit `pyproject.toml`, run `uv lock`, then
`bash scripts/export_requirements.sh` (CI fails if the exports are stale).
Never edit the requirements files by hand.
Run the API. For local dev, prefer the ASGI entrypoint under uvicorn — it
serves the **whole** app, matches production, and hot-reloads:
+23
View File
@@ -0,0 +1,23 @@
# Build context is application/. Keep local state and caches out of the image.
__pycache__/
*.py[cod]
.pytest_cache/
.ruff_cache/
.coverage
htmlcov/
*.log
# Runtime data: bind-mounted or created at run time, never baked in.
indexes/
inputs/
vectors/
*.faiss
*.pkl
# Secrets and local config.
.env
.env.*
# Not needed inside the image.
Dockerfile
.dockerignore
+126 -103
View File
@@ -1,74 +1,81 @@
# Builder Stage
FROM ubuntu:24.04 as builder
# DocsGPT backend image.
#
# Build args:
# EXTRAS comma-separated optional extras to bake in, matching the
# pyproject extras / requirements-<extra>.txt files:
# docling (layout-model parser + OCR backend), milvus.
# INSTALL_DOCLING legacy alias for EXTRAS=docling (setup.sh writes it).
# INSTALL_TESSERACT bake the tesseract binary for OCR_ENGINE=tesseract.
# EMBEDDINGS_PREFETCH registry names of the embedding models to bake; empty
# bakes both defaults (mpnet for upgrades, granite for
# new installs).
#
# Everything the default configuration needs is inside the image: embedding
# models, their tokenizers, tiktoken's encoding and, with the docling extra,
# docling's layout/table/OCR models. `python -m application.scripts.verify_offline`
# under `docker run --network none` proves it.
FROM ubuntu:24.04 AS builder
ENV DEBIAN_FRONTEND=noninteractive
# Ubuntu 24.04 ships Python 3.12 in its main archive: no PPA needed. Every pin
# resolves to a wheel, so no compiler toolchain either.
RUN apt-get update && \
apt-get install -y --no-install-recommends python3.12 python3.12-venv ca-certificates && \
rm -rf /var/lib/apt/lists/*
COPY requirements.txt requirements-docling.txt requirements-milvus.txt ./
RUN python3.12 -m venv /venv
ENV PATH="/venv/bin:$PATH"
RUN pip install --no-cache-dir --upgrade pip && \
pip install --no-cache-dir --only-binary=:all: -r requirements.txt
# Optional extras. Each requirements-<extra>.txt is exported from the same
# lock as requirements.txt, so installing it on top only adds the extra's
# packages. The docling file takes torch from the CPU-only PyTorch index.
# Not wheels-only: docling's antlr4 runtime ships as a pure-Python sdist.
ARG EXTRAS=""
ARG INSTALL_DOCLING=false
RUN set -e; \
extras="$EXTRAS"; \
if [ "$INSTALL_DOCLING" = "true" ]; then extras="$extras,docling"; fi; \
for extra in $(echo "$extras" | tr ',' ' '); do \
echo "Installing extra: $extra"; \
pip install --no-cache-dir -r "requirements-$extra.txt"; \
done
# google-api-python-client bundles discovery documents for ~600 Google APIs
# (99 MB). The application builds one client, Drive v3; keep only its document.
# Building another API's client needs its file back, or static_discovery=False.
RUN find /venv/lib/python3.12/site-packages/googleapiclient/discovery_cache/documents \
-type f ! -name 'drive.v3.json' -delete
FROM ubuntu:24.04 AS final
ENV DEBIAN_FRONTEND=noninteractive
RUN apt-get update && \
apt-get install -y software-properties-common && \
add-apt-repository ppa:deadsnakes/ppa && \
apt-get update && \
apt-get install -y --no-install-recommends gcc g++ wget unzip libc6-dev python3.12 python3.12-venv python3.12-dev && \
rm -rf /var/lib/apt/lists/*
# Verify Python installation and setup symlink
RUN if [ -f /usr/bin/python3.12 ]; then \
ln -s /usr/bin/python3.12 /usr/bin/python; \
else \
echo "Python 3.12 not found"; exit 1; \
fi
# Install Rust
RUN wget -q -O - https://sh.rustup.rs | sh -s -- -y
# Clean up to reduce container size
RUN apt-get remove --purge -y wget unzip && apt-get autoremove -y && rm -rf /var/lib/apt/lists/*
# Copy requirements manifests
COPY requirements.txt requirements-docling.txt ./
# Setup Python virtual environment
RUN python3.12 -m venv /venv
# Activate virtual environment and install Python packages
ENV PATH="/venv/bin:$PATH"
# Install Python packages
RUN pip install --no-cache-dir --upgrade pip && \
pip install --no-cache-dir tiktoken && \
pip install --no-cache-dir -r requirements.txt
# Optional docling parser engine (DOC_PARSER_ENGINE=docling, the docling OCR
# backend, read_document's structured output) — OFF by default: it pulls the
# layout/OCR model stack and adds gigabytes to the image. anydoc (in
# requirements.txt) is the default parser and needs none of it, and OCR runs
# natively on tesseract (below, also opt-in) without it.
ARG INSTALL_DOCLING=false
RUN if [ "$INSTALL_DOCLING" = "true" ]; then \
pip install --no-cache-dir -r requirements-docling.txt; \
fi
# Final Stage
FROM ubuntu:24.04 as final
RUN apt-get update && \
apt-get install -y software-properties-common && \
add-apt-repository ppa:deadsnakes/ppa && \
apt-get update && apt-get install -y --no-install-recommends \
python3.12 \
libgl1 \
libglib2.0-0 \
poppler-utils \
&& \
apt-get install -y --no-install-recommends python3.12 poppler-utils ca-certificates && \
ln -s /usr/bin/python3.12 /usr/bin/python && \
rm -rf /var/lib/apt/lists/*
# opencv (rapidocr, part of the docling extra) needs libGL at import time.
ARG EXTRAS=""
ARG INSTALL_DOCLING=false
RUN if [ "$INSTALL_DOCLING" = "true" ] || echo ",$EXTRAS," | grep -q ",docling,"; then \
apt-get update && \
apt-get install -y --no-install-recommends libgl1 libglib2.0-0 && \
rm -rf /var/lib/apt/lists/*; \
fi
# Optional tesseract OCR engine (OCR_ENABLED=true with OCR_ENGINE=tesseract,
# the default engine) — OFF by default like every other OCR dependency; OCR
# itself is off unless configured. Opt in with --build-arg
# INSTALL_TESSERACT=true (setup.sh writes it to .env when OCR is enabled);
# ~35 MB of system packages. Extra language packs are a deployment concern
# (apt: tesseract-ocr-<lang>, then list them in OCR_LANGS). A DeepSeek-OCR
# endpoint (OCR_ENGINE=deepseek) needs none of this.
# the default engine); ~35 MB of system packages. Extra language packs are a
# deployment concern (apt: tesseract-ocr-<lang>, then list them in OCR_LANGS).
# A DeepSeek-OCR endpoint (OCR_ENGINE=deepseek) needs none of this.
ARG INSTALL_TESSERACT=false
RUN if [ "$INSTALL_TESSERACT" = "true" ]; then \
apt-get update && \
@@ -76,61 +83,77 @@ RUN if [ "$INSTALL_TESSERACT" = "true" ]; then \
rm -rf /var/lib/apt/lists/*; \
fi
# Set working directory
LABEL org.opencontainers.image.source="https://github.com/arc53/DocsGPT" \
org.opencontainers.image.title="DocsGPT" \
org.opencontainers.image.description="DocsGPT backend: API and Celery worker" \
org.opencontainers.image.licenses="MIT"
WORKDIR /app
# Create a non-root user: `appuser` (Feel free to choose a name)
# The process user owns /app so the model prefetch below can run as it: an
# unprivileged prefetch writes the model files with the right owner up front,
# instead of a trailing chown -R that rewrites every model file into a second
# layer.
RUN groupadd -r appuser && \
useradd -r -g appuser -d /app -s /sbin/nologin -c "Docker image user" appuser
useradd -r -g appuser -d /app -s /sbin/nologin -c "Docker image user" appuser && \
chown appuser:appuser /app && \
install -d -o appuser -g appuser /app/models /app/application
# Copy the virtual environment and model from the builder stage
COPY --from=builder /venv /venv
# Pre-fetch the embedding models into FastEmbed's cache so a fresh container
# does not download on first ingest and an air-gapped install works at all.
# Both defaults are baked: an upgraded deployment keeps using mpnet until it
# runs the re-embed script, while a new one starts on granite.
# The prefetch writes hub-layout snapshots (including tokenizer.json) here, so
# HF_HUB_CACHE has to point at the same directory: chunking loads the tokenizer
# through ``tokenizers``, which reads the hub cache and would otherwise fetch
# over the network on first ingest -- and fall back to cl100k when offline.
# Every cache the application reads at run time lives under /app/models and is
# filled at build time:
# EMBEDDINGS_CACHE_DIR / HF_HUB_CACHE FastEmbed models and their tokenizers
# (chunking reads tokenizer.json from
# the same hub-layout snapshot)
# TIKTOKEN_CACHE_DIR cl100k_base for token accounting
# DOCLING_ARTIFACTS_PATH docling's models (docling extra only)
ENV EMBEDDINGS_CACHE_DIR=/app/models \
HF_HUB_CACHE=/app/models
# Only the modules the prefetch imports are copied first. It reaches nothing
# beyond model_registry, which is stdlib-only, so keeping the full source copy
# below this layer stops an unrelated edit from re-downloading ~780 MB of model
# artifacts on every build.
COPY __init__.py /app/application/__init__.py
COPY scripts/__init__.py scripts/prefetch_models.py /app/application/scripts/
COPY vectorstore/__init__.py vectorstore/model_registry.py /app/application/vectorstore/
ARG EMBEDDINGS_PREFETCH=""
RUN PYTHONPATH=/app /venv/bin/python -m application.scripts.prefetch_models ${EMBEDDINGS_PREFETCH}
# Copy your application code
COPY . /app/application
# Change the ownership of the /app directory to the appuser
RUN mkdir -p /app/application/inputs/local
RUN chown -R appuser:appuser /app
# Set environment variables
ENV FLASK_APP=app.py \
FLASK_DEBUG=true \
HF_HUB_CACHE=/app/models \
TIKTOKEN_CACHE_DIR=/app/models/tiktoken \
DOCLING_ARTIFACTS_PATH=/app/models/docling \
HF_HUB_DISABLE_TELEMETRY=1 \
PATH="/venv/bin:$PATH"
# Only the modules the prefetch imports are copied first, so an unrelated
# source edit does not invalidate the model layer.
COPY --chown=appuser:appuser __init__.py /app/application/__init__.py
COPY --chown=appuser:appuser scripts/__init__.py scripts/prefetch_models.py /app/application/scripts/
COPY --chown=appuser:appuser vectorstore/__init__.py vectorstore/model_registry.py /app/application/vectorstore/
USER appuser
ARG EMBEDDINGS_PREFETCH=""
RUN PYTHONPATH=/app python -m application.scripts.prefetch_models ${EMBEDDINGS_PREFETCH} && \
rm -rf /app/models/.locks /app/.cache
# docling downloads its layout, table-structure and OCR models on first parse;
# bake them so the docling variant is as self-contained as the default image.
RUN if python -c "import docling" 2>/dev/null; then \
docling-tools models download --output-dir /app/models/docling layout tableformer rapidocr && \
rm -rf /app/.cache; \
fi
COPY --chown=appuser:appuser . /app/application
# Runtime data directories, owned by the process user so a named volume
# mounted on them (docker-compose-standalone.yaml) inherits that ownership
# and uploads work without running the container as root.
RUN mkdir -p /app/application/inputs/local /app/inputs /app/indexes /app/vectors
ENV FLASK_APP=app.py
# Thread caps. onnxruntime (FastEmbed) ignores OMP_NUM_THREADS and sizes its
# pool to the host's core count, which a CPU-limited container still reports;
# EMBEDDINGS_THREADS pins it the way OMP_NUM_THREADS pinned torch before.
ENV MALLOC_ARENA_MAX=2 \
OMP_NUM_THREADS=4 \
MKL_NUM_THREADS=4 \
OPENBLAS_NUM_THREADS=4
OPENBLAS_NUM_THREADS=4 \
EMBEDDINGS_THREADS=4
# Expose the port the app runs on
EXPOSE 7091
# Switch to non-root user
USER appuser
# BoundedDrainUvicornWorker makes max_requests recycles safe with held-open SSE
# connections (see application/gunicorn_worker.py); with recycles now safe,
# --max-requests is raised (kept for memory hygiene) to cut churn.
+79
View File
@@ -0,0 +1,79 @@
"""Optional dependency extras and the one place their install hints come from.
Heavy or niche packages are not installed by default. Each extra maps to a
pyproject extra and to an exported ``application/requirements-<extra>.txt``,
so a missing module can always be explained with the exact command to run.
"""
from __future__ import annotations
import importlib
import importlib.util
import sys
from types import ModuleType
from typing import Dict, Tuple
#: Extra name -> top-level modules it provides. Keep in sync with
#: ``[project.optional-dependencies]`` in pyproject.toml.
EXTRAS: Dict[str, Tuple[str, ...]] = {
"docling": ("docling", "rapidocr"),
"milvus": ("pymilvus",),
}
_MODULE_TO_EXTRA: Dict[str, str] = {
module: extra for extra, modules in EXTRAS.items() for module in modules
}
def install_hint(extra: str) -> str:
"""Install command for ``extra``, for error messages and logs."""
return (
f"pip install -r application/requirements-{extra}.txt "
f"(or: uv sync --extra {extra}; Docker: --build-arg EXTRAS={extra})"
)
def extra_for(module: str) -> str | None:
"""Extra that provides top-level module ``module``, if any."""
return _MODULE_TO_EXTRA.get(module.split(".")[0])
def is_available(module: str) -> bool:
"""Whether ``module`` can be imported, without importing it."""
root = module.split(".")[0]
if root in sys.modules:
return sys.modules[root] is not None
try:
return importlib.util.find_spec(root) is not None
except (ImportError, ValueError):
return False
def missing_message(module: str, purpose: str | None = None) -> str:
"""Human-readable explanation that ``module`` is absent and how to add it."""
extra = extra_for(module)
what = f"{module} is not installed"
if purpose:
what += f" ({purpose})"
if extra:
return f"{what}. It is part of the optional '{extra}' extra: {install_hint(extra)}"
return f"{what}. Install it with: pip install {module}"
def require(module: str, purpose: str | None = None) -> ModuleType:
"""Import ``module`` or raise ``ImportError`` naming the extra to install.
Args:
module: Importable module path, e.g. ``"pymilvus"``.
purpose: Short note on what needed it, included in the error.
Returns:
The imported module.
Raises:
ImportError: With the install hint when the module is absent.
"""
try:
return importlib.import_module(module)
except ImportError as exc:
raise ImportError(missing_message(module, purpose)) from exc
+3 -1
View File
@@ -19,6 +19,7 @@ from application.parser.schema.base import Document
from application.stt.constants import SUPPORTED_AUDIO_EXTENSIONS
from application.utils import num_tokens_from_string
from application.core.settings import settings
from application.core.optional_deps import install_hint
def _build_audio_parser_mapping() -> Dict[str, BaseParser]:
@@ -199,8 +200,9 @@ def _docling_file_extractor(
logging.log(
missing_log_level,
"docling is not installed. Using standard parsers%s. For layout-model "
"parsing, install with: pip install -r application/requirements-docling.txt",
"parsing, install the docling extra: %s",
" with native OCR" if ocr_enabled else "",
install_hint("docling"),
)
return _legacy_file_extractor(pdf_text_fast_path, ocr_enabled=ocr_enabled)
+10 -5
View File
@@ -26,6 +26,8 @@ from application.parser.file.ocr_parser import VALID_OCR_ENGINES as _VALID_OCR_E
from application.parser.file.ocr_parser import collapse_cjk_spaces
from application.utils import truncate_to_line_boundary
from application.core.optional_deps import install_hint
logger = logging.getLogger(__name__)
@@ -590,11 +592,14 @@ class DoclingParser(BaseParser):
logger.info(f" force_full_page_ocr={self.force_full_page_ocr}")
logger.info(f" ocr_engine={self.ocr_engine or settings.OCR_ENGINE}")
if importlib.util.find_spec("docling.document_converter") is None:
raise ImportError(
"docling is required for DoclingParser. "
"Install it with: pip install -r application/requirements-docling.txt"
)
# find_spec raises when the parent package is absent, so the hint has
# to cover both a missing docling and a docling without the submodule.
try:
converter_spec = importlib.util.find_spec("docling.document_converter")
except ModuleNotFoundError:
converter_spec = None
if converter_spec is None:
raise ImportError(f"docling is required for DoclingParser. {install_hint('docling')}")
# Create converter with hybrid OCR (smart: text direct, bitmaps OCR'd)
self._converter = self._create_converter()
+4 -2
View File
@@ -39,6 +39,8 @@ from application.parser.file.base_parser import (
module_available,
)
from application.core.optional_deps import install_hint
logger = logging.getLogger(__name__)
# Every engine ``OCR_ENGINE`` accepts. ``auto``, ``ocrmac`` and ``rapidocr``
@@ -129,8 +131,8 @@ def resolve_ocr_backend(requested: Optional[str] = None) -> str:
docling_installed = module_available("docling")
if backend == "docling" and not docling_installed:
logger.warning(
"OCR_BACKEND=docling but docling is not installed (pip install -r "
"application/requirements-docling.txt); using the native OCR backend"
"OCR_BACKEND=docling but docling is not installed (%s); using the native OCR backend",
install_hint("docling"),
)
return "native"
if backend == "auto":
@@ -9,6 +9,11 @@ from application.parser.schema.base import Document
import tldextract
import os
# The bundled public-suffix snapshot is enough for domain matching; the
# default extractor would fetch the live list on first use and cache it on
# disk, which is a network round trip the ingest worker should not depend on.
_extract = tldextract.TLDExtract(suffix_list_urls=(), cache_dir=None)
class CrawlerLoader(BaseRemote):
def __init__(self, limit=10, allow_subdomains=False):
"""
@@ -124,7 +129,7 @@ class CrawlerLoader(BaseRemote):
return links
def _get_base_domain(self, url):
extracted = tldextract.extract(url)
extracted = _extract(url)
# Reconstruct the domain as domain.suffix
base_domain = f"{extracted.domain}.{extracted.suffix}"
return base_domain
@@ -141,7 +146,7 @@ class CrawlerLoader(BaseRemote):
if not parsed_link.netloc:
continue
extracted = tldextract.extract(parsed_link.netloc)
extracted = _extract(parsed_link.netloc)
link_base = f"{extracted.domain}.{extracted.suffix}"
if self.allow_subdomains:
+17 -1
View File
@@ -216,12 +216,28 @@ class HuggingFaceCounter(TokenCounter):
return _cap_piece_chars(pieces, first, rest)
def _tokenizer_file(repo: str) -> str:
"""Path to ``repo``'s ``tokenizer.json``, from the hub cache when present.
A warmed cache (the Docker image bakes the default models) answers without
touching the network. ``hf_hub_download`` would otherwise revalidate the
revision with a HEAD request on every process start, and stall for the
etag timeout on a host that cannot reach huggingface.co.
"""
from huggingface_hub import hf_hub_download
try:
return hf_hub_download(repo, "tokenizer.json", local_files_only=True)
except Exception: # noqa: BLE001 -- not cached: fetch it
return hf_hub_download(repo, "tokenizer.json")
def _load_hf_counter(repo: str) -> Optional[HuggingFaceCounter]:
"""Load ``repo``'s tokenizer, or ``None`` if it is not reachable."""
try:
from tokenizers import Tokenizer
tokenizer = Tokenizer.from_pretrained(repo)
tokenizer = Tokenizer.from_file(_tokenizer_file(repo))
# Repos ship padding and truncation defaults meant for inference
# batches. Left on, every count returns the padded width (128 for
# mpnet) and the offsets carry (0, 0) entries for the padding, so both
File diff suppressed because it is too large. Load diff
+977
View File
@@ -0,0 +1,977 @@
# GENERATED by scripts/export_requirements.sh from uv.lock -- do not edit.
#
# Core runtime plus the milvus extra (VECTOR_STORE=milvus): pymilvus and the
# embedded milvus-lite server, which pulls pyarrow.
# Docker: --build-arg EXTRAS=milvus
a2wsgi==1.10.10
# via docsgpt
aiofile==3.12.3
# via py-key-value-aio
aiofiles==25.1.0
# via daytona
aiohappyeyeballs==2.7.1
# via aiohttp
aiohttp==3.14.3
# via
# aiohttp-retry
# daytona
# daytona-analytics-api-client-async
# daytona-api-client-async
# daytona-toolbox-api-client-async
# python-socketio
aiohttp-retry==2.9.1
# via
# daytona-analytics-api-client-async
# daytona-api-client-async
# daytona-toolbox-api-client-async
aiosignal==1.4.0
# via aiohttp
alembic==1.19.2
# via docsgpt
amqp==5.3.1
# via kombu
aniso8601==10.0.1
# via flask-restx
annotated-doc==0.0.5
# via typer
annotated-types==0.8.0
# via pydantic
anthropic==0.121.0
# via docsgpt
anyio==4.15.1
# via
# anthropic
# google-genai
# httpx
# httpx-ws
# mcp
# openai
# py-key-value-aio
# sse-starlette
# starlette
# watchfiles
asgiref==3.12.1
# via opentelemetry-instrumentation-asgi
attrs==26.1.0
# via
# aiohttp
# cyclopts
# jsonschema
# jsonschema-path
# referencing
authlib==1.8.0
# via fastmcp-slim
beartype==0.22.9
# via py-key-value-aio
beautifulsoup4==4.15.0
# via
# docsgpt
# markdownify
bidict==0.24.1
# via python-socketio
billiard==4.2.4
# via celery
blinker==1.9.0
# via flask
boto3==1.43.67
# via docsgpt
botocore==1.43.89
# via
# boto3
# s3transfer
cachetools==7.1.8
# via
# py-key-value-aio
# pymilvus
caio==0.12.2
# via aiofile
cel-python==0.5.0
# via docsgpt
celery==5.6.3
# via
# celery-redbeat
# docsgpt
celery-redbeat==2.4.2
# via docsgpt
certifi==2026.7.22
# via
# httpcore
# httpx
# requests
cffi==2.1.1 ; platform_python_implementation != 'PyPy'
# via cryptography
chardet==7.6.0
# via prance
charset-normalizer==3.5.1
# via requests
click==8.1.8
# via
# celery
# click-didyoumean
# click-plugins
# click-repl
# ddgs
# flask
# gtts
# uvicorn
click-didyoumean==0.3.1
# via celery
click-plugins==1.1.1.2
# via celery
click-repl==0.3.0
# via celery
colorama==0.4.6 ; sys_platform == 'win32'
# via
# click
# loguru
# tqdm
# typer
croniter==6.2.4
# via docsgpt
cryptography==50.0.0
# via
# authlib
# docsgpt
# google-auth
# joserfc
# msal
# pyjwt
# secretstorage
cyclopts==4.24.0
# via fastmcp-slim
dataclasses-json==0.6.7
# via docsgpt
daytona==0.205.1
# via docsgpt
daytona-analytics-api-client==0.205.1
# via daytona
daytona-analytics-api-client-async==0.205.1
# via daytona
daytona-api-client==0.205.1
# via daytona
daytona-api-client-async==0.205.1
# via daytona
daytona-toolbox-api-client==0.205.1
# via daytona
daytona-toolbox-api-client-async==0.205.1
# via daytona
ddgs==9.16.0
# via docsgpt
decorator==5.3.1
# via retry
defusedxml==0.7.1
# via
# docsgpt
# praw
deprecated==1.3.1
# via daytona
distro==1.9.0
# via
# anthropic
# google-genai
# openai
dnspython==2.8.0
# via email-validator
docstring-parser==0.18.0
# via
# anthropic
# cyclopts
docx2txt==0.9
# via docsgpt
ecdsa==0.19.2
# via python-jose
elevenlabs==2.62.0
# via docsgpt
email-validator==2.3.0
# via pydantic
et-xmlfile==2.0.0
# via openpyxl
exceptiongroup==1.3.1
# via fastmcp-slim
faiss-cpu==1.15.0
# via
# docsgpt
# milvus-lite
fast-ebook==0.2.0
# via docsgpt
fastembed==0.8.0
# via docsgpt
fastmcp==3.4.6
# via docsgpt
fastmcp-slim==3.4.6
# via fastmcp
filelock==3.32.5
# via
# huggingface-hub
# tldextract
firecrawl-anydoc==0.2.3
# via docsgpt
flask==3.1.3
# via
# docsgpt
# flask-restx
flask-restx==1.3.2
# via docsgpt
flatbuffers==25.12.19
# via onnxruntime
frozenlist==1.8.0
# via
# aiohttp
# aiosignal
fsspec==2026.7.0
# via huggingface-hub
google-api-core==2.36.0
# via google-api-python-client
google-api-python-client==2.198.0
# via docsgpt
google-auth==2.57.1
# via
# google-api-core
# google-api-python-client
# google-auth-httplib2
# google-auth-oauthlib
# google-genai
google-auth-httplib2==0.4.2
# via google-api-python-client
google-auth-oauthlib==1.4.0
# via docsgpt
google-genai==2.17.0
# via docsgpt
google-re2==1.1.20251105
# via cel-python
googleapis-common-protos==1.75.3
# via
# google-api-core
# opentelemetry-exporter-otlp-proto-grpc
# opentelemetry-exporter-otlp-proto-http
greenlet==3.5.5 ; platform_machine == 'AMD64' or platform_machine == 'WIN32' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'ppc64le' or platform_machine == 'win32' or platform_machine == 'x86_64'
# via sqlalchemy
griffelib==2.3.0
# via fastmcp-slim
grpcio==1.83.1
# via
# milvus-lite
# opentelemetry-exporter-otlp-proto-grpc
# pymilvus
# qdrant-client
gtts==2.5.4
# via docsgpt
gunicorn==26.0.0
# via
# docsgpt
# uvicorn-worker
h11==0.16.0
# via
# httpcore
# uvicorn
# wsproto
h2==4.4.1
# via httpx
hf-xet==1.6.0 ; platform_machine == 'AMD64' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'arm64' or platform_machine == 'x86_64'
# via huggingface-hub
hpack==4.2.0
# via h2
httpcore==1.0.9
# via
# httpx
# httpx-ws
httplib2==0.32.0
# via
# google-api-python-client
# google-auth-httplib2
httptools==0.8.0
# via uvicorn
httpx==0.28.1
# via
# anthropic
# daytona
# elevenlabs
# fastmcp-slim
# google-genai
# httpx-ws
# huggingface-hub
# mcp
# openai
# qdrant-client
httpx-sse==0.4.3
# via mcp
httpx-ws==0.9.0
# via daytona
huggingface-hub==1.16.1
# via
# fastembed
# tokenizers
hyperframe==6.1.0
# via h2
idna==3.19
# via
# anyio
# email-validator
# httpx
# requests
# tldextract
# yarl
importlib-resources==7.1.0
# via flask-restx
itsdangerous==2.2.0
# via flask
jaraco-classes==3.4.0
# via keyring
jaraco-context==6.1.2
# via keyring
jaraco-functools==4.6.0
# via keyring
jeepney==0.9.0 ; sys_platform == 'linux'
# via
# keyring
# secretstorage
jinja2==3.1.6
# via
# docsgpt
# flask
jiter==0.16.0
# via
# anthropic
# openai
jmespath==1.1.0
# via
# boto3
# botocore
# cel-python
joserfc==1.7.5
# via
# authlib
# fastmcp-slim
jsonref==1.1.0
# via fastmcp-slim
jsonschema==4.26.0
# via
# flask-restx
# mcp
# openapi-schema-validator
# openapi-spec-validator
jsonschema-path==0.5.0
# via
# fastmcp-slim
# openapi-spec-validator
jsonschema-specifications==2025.9.1
# via
# jsonschema
# openapi-schema-validator
keyring==25.7.0
# via py-key-value-aio
kombu==5.6.2
# via
# celery
# docsgpt
lark==1.3.1
# via cel-python
lazy-object-proxy==1.12.0
# via openapi-spec-validator
loguru==0.7.3
# via fastembed
lxml==6.1.3
# via
# ddgs
# python-pptx
mako==1.4.1
# via alembic
markdown-it-py==4.2.0
# via rich
markdownify==1.2.3
# via docsgpt
markupsafe==3.0.3
# via
# flask
# jinja2
# mako
# werkzeug
marshmallow==3.26.2
# via dataclasses-json
mcp==1.29.1
# via fastmcp-slim
mdurl==0.1.2
# via markdown-it-py
milvus-lite==3.2.0 ; sys_platform != 'win32'
# via docsgpt
mmh3==5.3.0
# via fastembed
more-itertools==11.1.0
# via
# jaraco-classes
# jaraco-functools
msal==1.37.0
# via docsgpt
multidict==6.7.1
# via
# aiohttp
# yarl
mypy-extensions==1.1.0
# via typing-inspect
networkx==3.6.1
# via docsgpt
numpy==2.5.1
# via
# docsgpt
# faiss-cpu
# fastembed
# milvus-lite
# onnxruntime
# pandas
# qdrant-client
oauthlib==3.3.1
# via requests-oauthlib
obstore==0.11.1
# via daytona
onnxruntime==1.28.0
# via
# docsgpt
# fastembed
openai==2.53.0
# via docsgpt
openapi-pydantic==0.5.1
# via fastmcp-slim
openapi-schema-validator==0.9.0
# via openapi-spec-validator
openapi-spec-validator==0.9.0
# via openapi3-parser
openapi3-parser==1.1.22
# via docsgpt
openpyxl==3.1.5
# via docsgpt
opentelemetry-api==1.44.0
# via
# daytona
# fastmcp-slim
# google-api-core
# opentelemetry-distro
# opentelemetry-exporter-otlp-proto-grpc
# opentelemetry-exporter-otlp-proto-http
# opentelemetry-instrumentation
# opentelemetry-instrumentation-aiohttp-client
# opentelemetry-instrumentation-asgi
# opentelemetry-instrumentation-celery
# opentelemetry-instrumentation-dbapi
# opentelemetry-instrumentation-flask
# opentelemetry-instrumentation-logging
# opentelemetry-instrumentation-psycopg
# opentelemetry-instrumentation-redis
# opentelemetry-instrumentation-requests
# opentelemetry-instrumentation-sqlalchemy
# opentelemetry-instrumentation-starlette
# opentelemetry-instrumentation-wsgi
# opentelemetry-sdk
# opentelemetry-semantic-conventions
opentelemetry-distro==0.65b0
# via docsgpt
opentelemetry-exporter-otlp==1.44.0
# via docsgpt
opentelemetry-exporter-otlp-proto-common==1.44.0
# via
# opentelemetry-exporter-otlp-proto-grpc
# opentelemetry-exporter-otlp-proto-http
opentelemetry-exporter-otlp-proto-grpc==1.44.0
# via opentelemetry-exporter-otlp
opentelemetry-exporter-otlp-proto-http==1.44.0
# via
# daytona
# opentelemetry-exporter-otlp
opentelemetry-instrumentation==0.65b0
# via
# opentelemetry-distro
# opentelemetry-instrumentation-aiohttp-client
# opentelemetry-instrumentation-asgi
# opentelemetry-instrumentation-celery
# opentelemetry-instrumentation-dbapi
# opentelemetry-instrumentation-flask
# opentelemetry-instrumentation-logging
# opentelemetry-instrumentation-psycopg
# opentelemetry-instrumentation-redis
# opentelemetry-instrumentation-requests
# opentelemetry-instrumentation-sqlalchemy
# opentelemetry-instrumentation-starlette
# opentelemetry-instrumentation-wsgi
opentelemetry-instrumentation-aiohttp-client==0.65b0
# via daytona
opentelemetry-instrumentation-asgi==0.65b0
# via opentelemetry-instrumentation-starlette
opentelemetry-instrumentation-celery==0.65b0
# via docsgpt
opentelemetry-instrumentation-dbapi==0.65b0
# via opentelemetry-instrumentation-psycopg
opentelemetry-instrumentation-flask==0.65b0
# via docsgpt
opentelemetry-instrumentation-logging==0.65b0
# via docsgpt
opentelemetry-instrumentation-psycopg==0.65b0
# via docsgpt
opentelemetry-instrumentation-redis==0.65b0
# via docsgpt
opentelemetry-instrumentation-requests==0.65b0
# via docsgpt
opentelemetry-instrumentation-sqlalchemy==0.65b0
# via docsgpt
opentelemetry-instrumentation-starlette==0.65b0
# via docsgpt
opentelemetry-instrumentation-wsgi==0.65b0
# via opentelemetry-instrumentation-flask
opentelemetry-proto==1.44.0
# via
# opentelemetry-exporter-otlp-proto-common
# opentelemetry-exporter-otlp-proto-grpc
# opentelemetry-exporter-otlp-proto-http
opentelemetry-sdk==1.44.0
# via
# daytona
# opentelemetry-distro
# opentelemetry-exporter-otlp-proto-grpc
# opentelemetry-exporter-otlp-proto-http
opentelemetry-semantic-conventions==0.65b0
# via
# opentelemetry-instrumentation
# opentelemetry-instrumentation-aiohttp-client
# opentelemetry-instrumentation-asgi
# opentelemetry-instrumentation-celery
# opentelemetry-instrumentation-dbapi
# opentelemetry-instrumentation-flask
# opentelemetry-instrumentation-logging
# opentelemetry-instrumentation-redis
# opentelemetry-instrumentation-requests
# opentelemetry-instrumentation-sqlalchemy
# opentelemetry-instrumentation-starlette
# opentelemetry-instrumentation-wsgi
# opentelemetry-sdk
opentelemetry-util-http==0.65b0
# via
# opentelemetry-instrumentation-aiohttp-client
# opentelemetry-instrumentation-asgi
# opentelemetry-instrumentation-flask
# opentelemetry-instrumentation-requests
# opentelemetry-instrumentation-starlette
# opentelemetry-instrumentation-wsgi
orjson==3.12.0
# via pymilvus
packaging==26.3
# via
# faiss-cpu
# fastmcp-slim
# gunicorn
# huggingface-hub
# kombu
# marshmallow
# onnxruntime
# opentelemetry-instrumentation
# opentelemetry-instrumentation-flask
# opentelemetry-instrumentation-sqlalchemy
# prance
pandas==3.0.5
# via
# docsgpt
# pymilvus
pathable==0.6.0
# via jsonschema-path
pdf2image==1.17.0
# via docsgpt
pendulum==3.2.0
# via cel-python
pgvector==0.5.0
# via docsgpt
pillow==12.3.0
# via
# docsgpt
# fastembed
# pdf2image
# python-pptx
platformdirs==4.11.7
# via fastmcp-slim
portalocker==3.2.0
# via qdrant-client
prance==26.7.19.0
# via openapi3-parser
praw==8.0.2
# via docsgpt
prawcore==4.0.0
# via praw
primp==2.0.0
# via ddgs
prompt-toolkit==3.0.53
# via click-repl
propcache==0.5.2
# via
# aiohttp
# yarl
proto-plus==1.28.4
# via google-api-core
protobuf==7.36.1
# via
# google-api-core
# googleapis-common-protos
# onnxruntime
# opentelemetry-proto
# proto-plus
# pymilvus
# qdrant-client
psycopg==3.3.5
# via docsgpt
psycopg-binary==3.3.5 ; implementation_name != 'pypy'
# via psycopg
psycopg-pool==3.3.1
# via psycopg
py==1.11.0
# via retry
py-key-value-aio==0.4.5
# via fastmcp-slim
py-rust-stemmers==0.1.8
# via fastembed
pyarrow==25.0.1 ; sys_platform != 'win32'
# via milvus-lite
pyasn1==0.6.4
# via
# pyasn1-modules
# python-jose
# rsa
pyasn1-modules==0.4.2
# via google-auth
pycparser==3.0 ; implementation_name != 'PyPy' and platform_python_implementation != 'PyPy'
# via cffi
pydantic==2.13.5
# via
# anthropic
# daytona
# daytona-analytics-api-client
# daytona-analytics-api-client-async
# daytona-api-client
# daytona-api-client-async
# daytona-toolbox-api-client
# daytona-toolbox-api-client-async
# docsgpt
# elevenlabs
# fastmcp-slim
# google-genai
# mcp
# openai
# openapi-pydantic
# openapi-schema-validator
# openapi-spec-validator
# pydantic-settings
# qdrant-client
pydantic-core==2.46.5
# via
# elevenlabs
# pydantic
pydantic-settings==2.15.0
# via
# docsgpt
# fastmcp-slim
# mcp
# openapi-schema-validator
# openapi-spec-validator
pygments==2.21.0
# via
# rich
# rich-rst
pyjwt==2.13.0
# via
# mcp
# msal
pymilvus==3.0.1
# via docsgpt
pyparsing==3.3.2
# via httplib2
pypdf==6.15.0
# via docsgpt
pypdfium2==5.12.1
# via docsgpt
pyperclip==1.11.0
# via fastmcp-slim
python-dateutil==2.9.0.post0
# via
# botocore
# celery
# celery-redbeat
# croniter
# daytona-analytics-api-client
# daytona-analytics-api-client-async
# daytona-api-client
# daytona-api-client-async
# daytona-toolbox-api-client
# daytona-toolbox-api-client-async
# docsgpt
# pandas
# pendulum
python-dotenv==1.2.3
# via
# daytona
# docsgpt
# fastmcp-slim
# pydantic-settings
# pymilvus
# uvicorn
python-engineio==4.14.0
# via python-socketio
python-jose==3.5.0
# via docsgpt
python-multipart==0.0.32
# via
# daytona
# fastmcp-slim
# mcp
python-pptx==1.0.2
# via docsgpt
python-socketio==5.16.4
# via daytona
pywin32==312 ; sys_platform == 'win32'
# via
# mcp
# portalocker
pywin32-ctypes==0.2.3 ; sys_platform == 'win32'
# via keyring
pyyaml==6.0.3
# via
# cel-python
# docsgpt
# fastmcp-slim
# huggingface-hub
# jsonschema-path
# uvicorn
qdrant-client==1.19.0
# via docsgpt
redis==7.4.0
# via
# celery-redbeat
# docsgpt
referencing==0.37.0
# via
# flask-restx
# jsonschema
# jsonschema-path
# jsonschema-specifications
# openapi-schema-validator
regex==2026.9.3
# via tiktoken
requests==2.34.2
# via
# docsgpt
# elevenlabs
# fastembed
# google-api-core
# google-auth
# google-genai
# gtts
# msal
# opentelemetry-exporter-otlp-proto-http
# prance
# prawcore
# pymilvus
# python-socketio
# requests-file
# requests-oauthlib
# tiktoken
# tldextract
requests-file==3.0.1
# via tldextract
requests-oauthlib==2.0.0
# via google-auth-oauthlib
retry==0.9.2
# via docsgpt
rfc3339-validator==0.1.4
# via openapi-schema-validator
rich==15.0.0
# via
# cyclopts
# fastmcp-slim
# rich-rst
# typer
rich-rst==2.1.0
# via cyclopts
rpds-py==2026.6.3
# via
# jsonschema
# referencing
rsa==4.9.1
# via python-jose
ruamel-yaml==0.19.1
# via prance
s3transfer==0.19.2
# via boto3
secretstorage==3.5.0 ; sys_platform == 'linux'
# via keyring
shellingham==1.5.4
# via typer
simple-websocket==1.1.0
# via python-engineio
six==1.17.0
# via
# ecdsa
# markdownify
# python-dateutil
# rfc3339-validator
sniffio==1.3.1
# via
# anthropic
# google-genai
# openai
soupsieve==2.9.2
# via beautifulsoup4
sqlalchemy==2.0.52
# via
# alembic
# docsgpt
sse-starlette==3.4.11
# via mcp
starlette==1.6.0
# via
# docsgpt
# fastmcp-slim
# mcp
# sse-starlette
tenacity==9.1.4
# via
# celery-redbeat
# google-genai
tiktoken==0.13.0
# via docsgpt
tldextract==5.3.2
# via docsgpt
tokenizers==0.22.2
# via
# docsgpt
# fastembed
toml==0.10.2
# via daytona
tqdm==4.67.3
# via
# docsgpt
# fastembed
# huggingface-hub
# openai
typer==0.26.8
# via huggingface-hub
typing-extensions==4.16.0
# via
# aiohttp
# aiosignal
# alembic
# anthropic
# anyio
# beautifulsoup4
# daytona
# daytona-analytics-api-client
# daytona-analytics-api-client-async
# daytona-api-client
# daytona-api-client-async
# daytona-toolbox-api-client
# daytona-toolbox-api-client-async
# elevenlabs
# exceptiongroup
# fastmcp-slim
# google-genai
# grpcio
# huggingface-hub
# mcp
# obstore
# openai
# opentelemetry-api
# opentelemetry-exporter-otlp-proto-grpc
# opentelemetry-exporter-otlp-proto-http
# opentelemetry-sdk
# opentelemetry-semantic-conventions
# psycopg
# psycopg-pool
# py-key-value-aio
# pydantic
# pydantic-core
# python-pptx
# referencing
# sqlalchemy
# starlette
# typing-inspect
# typing-inspection
typing-inspect==0.9.0
# via dataclasses-json
typing-inspection==0.4.4
# via
# mcp
# pydantic
# pydantic-settings
tzdata==2026.3
# via
# kombu
# pandas
# pendulum
# psycopg
# tzlocal
tzlocal==5.4.4
# via celery
uncalled-for==0.4.0
# via fastmcp-slim
update-checker==1.0.0
# via praw
uritemplate==4.2.0
# via google-api-python-client
urllib3==2.7.0
# via
# botocore
# daytona
# daytona-analytics-api-client
# daytona-api-client
# daytona-toolbox-api-client
# qdrant-client
# requests
uvicorn==0.52.4
# via
# docsgpt
# fastmcp-slim
# mcp
# uvicorn-worker
uvicorn-worker==0.4.0
# via docsgpt
uvloop==0.22.1 ; platform_python_implementation != 'PyPy' and sys_platform != 'cygwin' and sys_platform != 'win32'
# via uvicorn
vine==5.1.0
# via
# amqp
# celery
# kombu
watchfiles==1.2.0
# via
# fastmcp-slim
# uvicorn
wcwidth==0.8.3
# via prompt-toolkit
websocket-client==1.9.0
# via
# docsgpt
# praw
# python-socketio
websockets==16.1.1
# via
# elevenlabs
# fastmcp-slim
# google-genai
# uvicorn
werkzeug==3.1.8
# via
# docsgpt
# flask
# flask-restx
win32-setctime==1.2.0 ; sys_platform == 'win32'
# via loguru
wrapt==2.4.0
# via
# deprecated
# opentelemetry-instrumentation
# opentelemetry-instrumentation-aiohttp-client
# opentelemetry-instrumentation-dbapi
# opentelemetry-instrumentation-redis
# opentelemetry-instrumentation-sqlalchemy
wsproto==1.3.2
# via
# daytona
# httpx-ws
# simple-websocket
xlsxwriter==3.2.9
# via python-pptx
yarl==1.24.5
# via aiohttp
File diff suppressed because it is too large. Load diff
+29 -5
View File
@@ -1,9 +1,14 @@
"""Download embedding model artifacts into FastEmbed's cache.
"""Download the model artifacts a fresh container would otherwise fetch.
Run at image build time so a fresh container does not download a model on its
first ingest, and an air-gapped install works at all. Both the legacy and the
current default are baked: an upgraded deployment keeps using mpnet until it
runs ``reembed``, while a new one starts on granite.
Run at image build time so a fresh container does not download on its first
request, and an air-gapped install works at all. Two things are warmed:
* Embedding models, into FastEmbed's cache. Both the legacy and the current
default are baked: an upgraded deployment keeps using mpnet until it runs
``reembed``, while a new one starts on granite.
* tiktoken's ``cl100k_base`` encoding, which token accounting uses on every
chat. tiktoken caches it under ``TIKTOKEN_CACHE_DIR`` (a temp dir when
unset), so the image sets that variable and this warms it.
Usage::
@@ -29,6 +34,23 @@ logger = logging.getLogger("prefetch_models")
#: Fetched when no names are given.
DEFAULT_MODELS = (DEFAULT_LEGACY, DEFAULT_NEW_INSTALL)
#: tiktoken encodings the application loads (``application.utils.get_encoding``).
TIKTOKEN_ENCODINGS = ("cl100k_base",)
def prefetch_tiktoken(names: Sequence[str] = TIKTOKEN_ENCODINGS) -> List[str]:
"""Warm tiktoken's cache for each encoding in ``names``.
Returns:
The encodings fetched.
"""
import tiktoken
for name in names:
logger.info("Fetching tiktoken encoding %s", name)
tiktoken.get_encoding(name)
return list(names)
def prefetch(names: Sequence[str], cache_dir: Optional[str] = None) -> List[str]:
"""Fetch each named model's artifacts.
@@ -82,6 +104,8 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
names = list(argv) if argv else list(DEFAULT_MODELS)
fetched = prefetch(names, os.environ.get("EMBEDDINGS_CACHE_DIR"))
logger.info("Cached %d model(s): %s", len(fetched), ", ".join(fetched))
encodings = prefetch_tiktoken()
logger.info("Cached tiktoken encoding(s): %s", ", ".join(encodings))
return 0
+153
View File
@@ -0,0 +1,153 @@
"""Check that an image can serve its defaults without any network access.
Exercises the code paths a fresh container hits first, the way the
application does: tiktoken token counting, the chunker's tokenizer for each
baked embedding model, and a FastEmbed embed with each. Run it inside the
image with networking disabled; every check must pass with zero requests::
docker run --rm --network none arc53/docsgpt:latest \\
python -m application.scripts.verify_offline
Exit status is non-zero on the first failure. Models to check default to
the prefetch defaults; pass registry names to check a different set.
"""
from __future__ import annotations
import logging
import socket
import sys
import time
from typing import Callable, List, Optional, Sequence
from application.core.optional_deps import is_available
from application.scripts.prefetch_models import DEFAULT_MODELS, TIKTOKEN_ENCODINGS
from application.vectorstore.model_registry import resolve
logger = logging.getLogger("verify_offline")
def _network_reachable(host: str = "huggingface.co", port: int = 443) -> bool:
try:
socket.create_connection((host, port), timeout=2).close()
return True
except OSError:
return False
def _check(name: str, fn: Callable[[], object]) -> bool:
started = time.time()
try:
detail = fn()
except Exception as exc: # noqa: BLE001 -- report every failure the same way
print(f"FAIL {name}: {type(exc).__name__}: {exc}")
return False
print(f"ok {name}: {detail} ({time.time() - started:.2f}s)")
return True
def verify(models: Sequence[str]) -> bool:
"""Run every check; return whether all passed."""
ok = True
def tiktoken_check(encoding: str) -> Callable[[], object]:
def run() -> object:
import tiktoken
return f"{len(tiktoken.get_encoding(encoding).encode('hello world'))} tokens"
return run
for encoding in TIKTOKEN_ENCODINGS:
ok &= _check(f"tiktoken {encoding}", tiktoken_check(encoding))
for name in models:
spec = resolve(name)
if spec is None or spec.provider != "fastembed":
print(f"skip {name}: not a local model")
continue
def tokenizer_check(model_name: str = name) -> object:
from application.parser.tokenization import get_token_counter
counter = get_token_counter(model_name)
if counter.name == "cl100k_base":
raise RuntimeError("tokenizer missing from the cache; chunking fell back to cl100k")
return f"{counter.name}, {counter.count('The quick brown fox')} tokens"
def embed_check(model_name: str = name) -> object:
from application.vectorstore.embeddings_local import EmbeddingsWrapper
vector = EmbeddingsWrapper(model_name).embed_query("hello")
return f"dimension {len(vector)}"
ok &= _check(f"tokenizer {name}", tokenizer_check)
ok &= _check(f"embeddings {name}", embed_check)
if is_available("docling"):
ok &= _check("docling PDF conversion", _docling_check)
else:
print("skip docling: not installed (slim image)")
return ok
# A one-page PDF with a single text run; enough for the layout model to have
# something to look at.
_TINY_PDF = (
b"%PDF-1.4\n"
b"1 0 obj<</Type/Catalog/Pages 2 0 R>>endobj\n"
b"2 0 obj<</Type/Pages/Kids[3 0 R]/Count 1>>endobj\n"
b"3 0 obj<</Type/Page/Parent 2 0 R/MediaBox[0 0 300 144]/Contents 4 0 R"
b"/Resources<</Font<</F1 5 0 R>>>>>>endobj\n"
b"4 0 obj<</Length 58>>stream\n"
b"BT /F1 18 Tf 20 100 Td (Offline verification page) Tj ET\n"
b"endstream\nendobj\n"
b"5 0 obj<</Type/Font/Subtype/Type1/BaseFont/Helvetica>>endobj\n"
b"trailer<</Root 1 0 R>>\n%%EOF\n"
)
def _docling_check() -> object:
"""Convert a tiny PDF through docling; its models must come from DOCLING_ARTIFACTS_PATH."""
import os
import tempfile
from docling.datamodel.base_models import InputFormat
from docling.datamodel.pipeline_options import PdfPipelineOptions
from docling.document_converter import DocumentConverter, PdfFormatOption
from application.parser.file.docling_parser import _apply_inference_settings
# Same global docling settings the parser applies: torch.compile stays off
# unless DOCLING_COMPILE_TORCH_MODELS asks for it (it needs a C++ toolchain).
_apply_inference_settings()
artifacts = os.environ.get("DOCLING_ARTIFACTS_PATH")
options = PdfPipelineOptions(artifacts_path=artifacts, do_ocr=False, do_table_structure=True)
converter = DocumentConverter(format_options={InputFormat.PDF: PdfFormatOption(pipeline_options=options)})
with tempfile.NamedTemporaryFile(suffix=".pdf", delete=False) as handle:
handle.write(_TINY_PDF)
path = handle.name
try:
text = converter.convert(path).document.export_to_markdown()
finally:
os.unlink(path)
if "Offline verification" not in text:
raise RuntimeError(f"unexpected conversion output: {text[:80]!r}")
return f"models from {artifacts or 'default cache'}, {len(text)} chars"
def main(argv: Optional[Sequence[str]] = None) -> int:
logging.basicConfig(level=logging.WARNING, format="%(levelname)s %(message)s")
models: List[str] = list(argv) if argv else list(DEFAULT_MODELS)
if _network_reachable():
print("note network is reachable; run with --network none to prove the offline path")
else:
print("note network unreachable, as intended")
passed = verify(models)
print("VERIFY OFFLINE: " + ("PASS" if passed else "FAIL"))
return 0 if passed else 1
if __name__ == "__main__":
sys.exit(main(sys.argv[1:]))
+5 -1
View File
@@ -104,7 +104,11 @@ def _read_repo_json(repo: str, filename: str) -> Optional[dict]:
try:
from huggingface_hub import hf_hub_download
with open(hf_hub_download(repo_id=repo, filename=filename), encoding="utf-8") as handle:
try:
path = hf_hub_download(repo_id=repo, filename=filename, local_files_only=True)
except Exception: # noqa: BLE001 -- not cached: fetch it
path = hf_hub_download(repo_id=repo, filename=filename)
with open(path, encoding="utf-8") as handle:
return json.load(handle)
except Exception as exc:
logger.debug("No %s for %s (%s)", filename, repo, exc)
+3 -1
View File
@@ -4,6 +4,7 @@ import uuid
from contextlib import contextmanager
from typing import Any, Dict, List, Optional, Tuple
from application.core.optional_deps import require
from application.core.settings import settings
from application.vectorstore.base import BaseVectorStore
from application.vectorstore.document_class import Document
@@ -41,7 +42,8 @@ class MilvusStore(BaseVectorStore):
def __init__(self, source_id: str = "", embeddings_key: str = "embeddings"):
super().__init__()
with _without_milvus_uri_env():
from pymilvus import DataType, MilvusClient
pymilvus = require("pymilvus", "VECTOR_STORE=milvus")
DataType, MilvusClient = pymilvus.DataType, pymilvus.MilvusClient
self._DataType = DataType
self._source_id = str(source_id).replace("application/indexes/", "").rstrip("/")
+19 -2
View File
@@ -1,9 +1,24 @@
services:
frontend:
build: ../frontend
build:
context: ../frontend
target: dev
environment:
# Every VITE_* the app reads. A bare name is passed through only when it is
# set in the shell or the --env-file, so an unset one does not reach the
# container as an empty string and override the image's own default.
- VITE_API_HOST=http://localhost:7091
- VITE_API_STREAMING=$VITE_API_STREAMING
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
- VITE_BASE_URL
- VITE_GOOGLE_CLIENT_ID
- VITE_GOOGLE_PICKER_API_KEY
- VITE_SHARE_POINT_CLIENT_ID
- VITE_CONFLUENCE_CLIENT_ID
- VITE_NOTIFICATION_TEXT
- VITE_NOTIFICATION_LINK
- VITE_ENABLE_VOICE_INPUT
- VITE_DISABLE_SOURCE_FE
- VITE_USE_V
ports:
- "5173:5173"
depends_on:
@@ -13,6 +28,7 @@ services:
build:
context: ../application
args:
EXTRAS: ${EXTRAS:-}
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
# Off by default; deployments running OCR_ENABLED=true with tesseract
# must set INSTALL_TESSERACT=true before rebuilding (see docker-compose.yaml).
@@ -40,6 +56,7 @@ services:
build:
context: ../application
args:
EXTRAS: ${EXTRAS:-}
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
# Off by default; deployments running OCR_ENABLED=true with tesseract
# must set INSTALL_TESSERACT=true before rebuilding (see docker-compose.yaml).
+22 -4
View File
@@ -1,12 +1,30 @@
# Pre-built images from Docker Hub (mirrored at ghcr.io/arc53).
# DOCSGPT_IMAGE_TAG develop (default, follows main) or a release, e.g. 0.20.0
# DOCSGPT_IMAGE_VARIANT empty (default, slim) or -docling: docling parser engine,
# its models, and tesseract baked in (OCR-ready)
# Set them in ../.env or the shell. deployment/docker-compose-standalone.yaml is
# the same stack without a git checkout.
name: docsgpt-oss
services:
frontend:
image: arc53/docsgpt-fe:develop
image: arc53/docsgpt-fe:${DOCSGPT_IMAGE_TAG:-develop}
environment:
# Every VITE_* the app reads. A bare name is passed through only when it is
# set in the shell or the --env-file, so an unset one does not reach the
# container as an empty string and override the image's own default.
- VITE_API_HOST=http://localhost:7091
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
- VITE_GOOGLE_CLIENT_ID=${VITE_GOOGLE_CLIENT_ID:-}
- VITE_BASE_URL
- VITE_GOOGLE_CLIENT_ID
- VITE_GOOGLE_PICKER_API_KEY
- VITE_SHARE_POINT_CLIENT_ID
- VITE_CONFLUENCE_CLIENT_ID
- VITE_NOTIFICATION_TEXT
- VITE_NOTIFICATION_LINK
- VITE_ENABLE_VOICE_INPUT
- VITE_DISABLE_SOURCE_FE
- VITE_USE_V
ports:
- "5173:5173"
depends_on:
@@ -15,7 +33,7 @@ services:
backend:
user: root
image: arc53/docsgpt:develop
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-develop}${DOCSGPT_IMAGE_VARIANT:-}
env_file:
- ../.env
environment:
@@ -38,7 +56,7 @@ services:
worker:
user: root
image: arc53/docsgpt:develop
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-develop}${DOCSGPT_IMAGE_VARIANT:-}
# `parsing` queue carries read_document/parse_document; required for its await to resolve.
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
env_file:
+130
View File
@@ -0,0 +1,130 @@
# DocsGPT from pre-built images, with no git checkout.
#
# curl -fsSLO https://raw.githubusercontent.com/arc53/DocsGPT/main/deployment/docker-compose-standalone.yaml
# printf 'LLM_PROVIDER=docsgpt\nVITE_API_STREAMING=true\nINTERNAL_KEY=%s\n' "$(openssl rand -hex 16)" > .env
# docker compose -f docker-compose-standalone.yaml up -d
#
# INTERNAL_KEY is the shared secret the worker uses to hand finished indexes to
# the API; without it every ingest fails with a 401 (setup.sh generates one).
# open http://localhost:5173
#
# Every release also attaches this file as an asset. Settings come from .env
# next to this file (any DocsGPT setting; the compose-internal service URLs
# below take precedence). Data lives in named volumes, so `docker compose
# down` keeps it and `docker compose down -v` removes it.
#
# DOCSGPT_IMAGE_TAG release to run, e.g. 0.20.0 (default: latest release);
# develop follows the main branch
# DOCSGPT_IMAGE_VARIANT empty (slim, default) or -docling: docling parser
# engine, its models, and tesseract baked in (OCR-ready)
# EMBEDDINGS_NAME defaults to granite here (this stack always starts on
# fresh volumes, so there is no older index to keep
# compatible); the code default stays mpnet for upgrades.
name: docsgpt
services:
frontend:
image: arc53/docsgpt-fe:${DOCSGPT_IMAGE_TAG:-latest}
env_file:
- path: .env
required: false
environment:
# Every VITE_* the app reads. A bare name is passed through only when it is
# set in the shell or the --env-file, so an unset one does not reach the
# container as an empty string and override the image's own default.
- VITE_API_HOST=${VITE_API_HOST:-http://localhost:7091}
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
- VITE_BASE_URL
- VITE_GOOGLE_CLIENT_ID
- VITE_GOOGLE_PICKER_API_KEY
- VITE_SHARE_POINT_CLIENT_ID
- VITE_CONFLUENCE_CLIENT_ID
- VITE_NOTIFICATION_TEXT
- VITE_NOTIFICATION_LINK
- VITE_ENABLE_VOICE_INPUT
- VITE_DISABLE_SOURCE_FE
- VITE_USE_V
ports:
- "5173:5173"
depends_on:
- backend
backend:
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-latest}${DOCSGPT_IMAGE_VARIANT:-}
# Same as docker-compose-hub.yaml: the data volumes are written by root so
# any image tag works, including releases that predate the appuser-owned
# /app/inputs, /app/indexes and /app/vectors directories.
user: root
env_file:
- path: .env
required: false
environment:
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
- EMBEDDINGS_NAME=${EMBEDDINGS_NAME:-ibm-granite/granite-embedding-311m-multilingual-r2}
ports:
- "7091:7091"
volumes:
- indexes:/app/indexes
- inputs:/app/inputs
- vectors:/app/vectors
depends_on:
redis:
condition: service_started
postgres:
condition: service_healthy
restart: unless-stopped
worker:
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-latest}${DOCSGPT_IMAGE_VARIANT:-}
user: root
# Consumes the default queue plus `parsing` (read_document) and `embeddings`
# (query embedding); without the latter every search times out.
command: celery -A application.app.celery worker -l INFO -B -Q docsgpt,parsing,embeddings
env_file:
- path: .env
required: false
environment:
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
- API_URL=http://backend:7091
- EMBEDDINGS_NAME=${EMBEDDINGS_NAME:-ibm-granite/granite-embedding-311m-multilingual-r2}
volumes:
- indexes:/app/indexes
- inputs:/app/inputs
- vectors:/app/vectors
depends_on:
redis:
condition: service_started
postgres:
condition: service_healthy
restart: unless-stopped
redis:
image: redis:6-alpine
restart: unless-stopped
postgres:
image: postgres:16-alpine
environment:
- POSTGRES_USER=docsgpt
- POSTGRES_PASSWORD=docsgpt
- POSTGRES_DB=docsgpt
volumes:
- postgres_data:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U docsgpt -d docsgpt"]
interval: 5s
timeout: 5s
retries: 10
restart: unless-stopped
volumes:
indexes:
inputs:
vectors:
postgres_data:
+24 -5
View File
@@ -1,13 +1,29 @@
name: docsgpt-oss
services:
frontend:
build: ../frontend
build:
context: ../frontend
# Vite dev server with hot reload over the bind mount below. The default
# target (what Docker Hub publishes) is a static build behind nginx.
target: dev
volumes:
- ../frontend/src:/app/src
environment:
# Every VITE_* the app reads. A bare name is passed through only when it is
# set in the shell or the --env-file, so an unset one does not reach the
# container as an empty string and override the image's own default.
- VITE_API_HOST=http://localhost:7091
- VITE_API_STREAMING=$VITE_API_STREAMING
- VITE_GOOGLE_CLIENT_ID=$VITE_GOOGLE_CLIENT_ID
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
- VITE_BASE_URL
- VITE_GOOGLE_CLIENT_ID
- VITE_GOOGLE_PICKER_API_KEY
- VITE_SHARE_POINT_CLIENT_ID
- VITE_CONFLUENCE_CLIENT_ID
- VITE_NOTIFICATION_TEXT
- VITE_NOTIFICATION_LINK
- VITE_ENABLE_VOICE_INPUT
- VITE_DISABLE_SOURCE_FE
- VITE_USE_V
ports:
- "5173:5173"
depends_on:
@@ -18,8 +34,10 @@ services:
build:
context: ../application
args:
# Bake the optional docling engine (layout-model OCR backend, structured
# output) into the image: set INSTALL_DOCLING=true in ../.env or the shell.
# Optional extras to bake in (comma-separated): docling, milvus. The
# docling extra brings the layout-model parser/OCR backend and its
# models. INSTALL_DOCLING=true is the older spelling of EXTRAS=docling.
EXTRAS: ${EXTRAS:-}
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
# Bake the tesseract binary behind OCR_ENABLED=true (~35 MB): set
# INSTALL_TESSERACT=true in ../.env or the shell (setup.sh does this
@@ -53,6 +71,7 @@ services:
build:
context: ../application
args:
EXTRAS: ${EXTRAS:-}
INSTALL_DOCLING: ${INSTALL_DOCLING:-false}
INSTALL_TESSERACT: ${INSTALL_TESSERACT:-false}
# Consumes the default queue AND the dedicated `parsing` (read_document /
@@ -94,13 +94,29 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
pip install -r application/requirements.txt
```
Optionally add the docling parser engine (OCR, `structured` output for
`read_document`; the default anydoc engine does not need it):
Dependencies are declared in `pyproject.toml` and locked in `uv.lock`; the
`requirements*.txt` files are exported from that lock, so with
[uv](https://docs.astral.sh/uv/) installed `uv sync` sets up the same
environment (plus the test tools) in one step.
Optional extras are not installed by default. Add them when you need the
feature (each file is the core set plus the extra):
```bash
pip install -r application/requirements-docling.txt
pip install -r application/requirements-docling.txt # docling parser engine: OCR backend, read_document structured output
pip install -r application/requirements-milvus.txt # VECTOR_STORE=milvus
# or with uv: uv sync --extra docling --extra milvus
```
The docling file adds the PyTorch CPU index; pip handles that as-is, while
`uv pip install -r` needs `UV_INDEX_STRATEGY=unsafe-best-match` (or use
`uv sync --extra docling`, which reads the lock).
A feature whose extra is missing fails with the exact command to run.
After changing `pyproject.toml`, run `uv lock` and
`bash scripts/export_requirements.sh` so the exported files stay in sync
(CI checks this).
5. **Run the Backend:**
For local development, run the ASGI composition under uvicorn. It serves the **whole** application, hot-reloads on source changes, and matches the production runtime:
+71 -11
View File
@@ -17,9 +17,62 @@ Docker is the recommended method for deploying DocsGPT, providing a consistent a
**Important Note for Windows Users:** Docker Desktop on Windows generally requires the WSL 2 backend to function correctly, especially when using features like host networking which are utilized in DocsGPT's Docker Compose setup. Ensure WSL 2 is enabled and configured in Docker Desktop settings.
## Quickest Setup: Using DocsGPT Public API
## Quickest Setup: Pre-built Images, No Checkout
The fastest way to try out DocsGPT is by using the public API endpoint. This requires minimal configuration and no local LLM setup.
Every release publishes ready-to-run images to Docker Hub (`arc53/docsgpt`,
`arc53/docsgpt-fe`) and GitHub Container Registry (`ghcr.io/arc53/docsgpt`,
`ghcr.io/arc53/docsgpt-fe`) for `linux/amd64` and `linux/arm64`. The images
contain everything the default configuration needs (embedding models,
tokenizers, tiktoken's encoding), so a fresh container makes no downloads on
first use. You do not need the source tree to run them:
1. **Download the standalone Compose file** (also attached to every
[release](https://github.com/arc53/DocsGPT/releases)):
```bash
mkdir docsgpt && cd docsgpt
curl -fsSLO https://raw.githubusercontent.com/arc53/DocsGPT/main/deployment/docker-compose-standalone.yaml
```
2. **Create a `.env` next to it** with your settings, for example the public API:
```bash
printf 'LLM_PROVIDER=docsgpt\nVITE_API_STREAMING=true\nINTERNAL_KEY=%s\n' "$(openssl rand -hex 16)" > .env
```
`INTERNAL_KEY` is the secret the worker uses to hand finished indexes to
the API; without it every upload fails with a 401. `setup.sh` generates
one for you, a hand-written `.env` has to include it. This stack runs the
granite embedding model unless `.env` sets `EMBEDDINGS_NAME`; both granite
and mpnet are baked into the image.
3. **Start it:**
```bash
docker compose -f docker-compose-standalone.yaml up -d
```
Then open [http://localhost:5173/](http://localhost:5173/). Data lives in
named Docker volumes; `docker compose -f docker-compose-standalone.yaml down`
keeps it and `down -v` removes it.
**Tags and variants.** `DOCSGPT_IMAGE_TAG` picks the version: a release such
as `0.20.0`, `latest` (the newest release, the default) or `develop` (follows
the `main` branch). `DOCSGPT_IMAGE_VARIANT` picks the flavour: empty for the
slim default image, or `-docling` for the image with the docling parser
engine, its models and tesseract baked in (needed for OCR of scanned
documents, see the [OCR guide](/Guides/ocr)). Both are read from `.env` or
the shell, e.g. `DOCSGPT_IMAGE_TAG=0.20.0 DOCSGPT_IMAGE_VARIANT=-docling`.
The same two variables drive `deployment/docker-compose-hub.yaml` in a
checkout.
## Using the Source Checkout
With a clone of the repository, `deployment/docker-compose-hub.yaml` runs the
same pre-built images while keeping your data in `application/indexes`,
`application/inputs` and `application/vectors`, and `deployment/docker-compose.yaml`
builds the images from your working tree (for local changes, or a build with
extra packages: `EXTRAS=docling` in `.env`).
1. **Clone the DocsGPT Repository (if you haven't already):**
@@ -39,19 +92,26 @@ The fastest way to try out DocsGPT is by using the public API endpoint. This req
```
LLM_PROVIDER=docsgpt
VITE_API_STREAMING=true
INTERNAL_KEY=<any random string, e.g. openssl rand -hex 16>
EMBEDDINGS_NAME=ibm-granite/granite-embedding-311m-multilingual-r2
```
This minimal configuration tells DocsGPT to use the public API. For more advanced settings and other LLM options, refer to the [DocsGPT Settings Guide](/Deploying/DocsGPT-Settings).
This minimal configuration tells DocsGPT to use the public API. The
`EMBEDDINGS_NAME` line is what `setup.sh` writes for a new install; without
it the code falls back to mpnet, the model earlier releases indexed with,
so that an upgraded deployment keeps its existing index working. For more advanced settings and other LLM options, refer to the [DocsGPT Settings Guide](/Deploying/DocsGPT-Settings).
4. **Launch DocsGPT with Docker Compose:**
Navigate to the root directory of the DocsGPT repository in your terminal and run:
```bash
docker compose --env-file .env -f deployment/docker-compose.yaml up -d
docker compose --env-file .env -f deployment/docker-compose-hub.yaml up -d
```
The `-d` flag runs Docker Compose in detached mode (in the background).
To build the images from your working tree instead of pulling them, use
`deployment/docker-compose.yaml` with `up --build -d`.
5. **Access DocsGPT in your browser:**
@@ -62,7 +122,7 @@ The fastest way to try out DocsGPT is by using the public API endpoint. This req
To stop the application, navigate to the same directory in your terminal and run:
```bash
docker compose -f deployment/docker-compose.yaml down
docker compose -f deployment/docker-compose-hub.yaml down
```
## Optional Ollama Setup (Local Models)
@@ -84,11 +144,11 @@ There are two Ollama optional files:
**CPU:**
```bash
docker compose --env-file .env -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml up -d
docker compose --env-file .env -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml up -d
```
**GPU:**
```bash
docker compose --env-file .env -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml up -d
docker compose --env-file .env -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml up -d
```
3. **Pull the Ollama Model:**
@@ -96,11 +156,11 @@ There are two Ollama optional files:
**Crucially, after launching with Ollama, you need to pull the desired model into the Ollama container.** Find the `LLM_NAME` you configured in your `.env` file (e.g., `llama3.2:1b`). Then execute the following command to pull the model *inside* the running Ollama container:
```bash
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml exec -it ollama ollama pull <LLM_NAME>
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml exec -it ollama ollama pull <LLM_NAME>
```
or (for GPU):
```bash
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml exec -it ollama ollama pull <LLM_NAME>
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml exec -it ollama ollama pull <LLM_NAME>
```
Replace `<LLM_NAME>` with the actual model name from your `.env` file.
@@ -113,12 +173,12 @@ There are two Ollama optional files:
To stop a DocsGPT setup launched with Ollama optional files, use `docker compose down` and include all the compose files used during the `up` command:
```bash
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml down
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-cpu.yaml down
```
or
```bash
docker compose -f deployment/docker-compose.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml down
docker compose -f deployment/docker-compose-hub.yaml -f deployment/optional/docker-compose.optional.ollama-gpu.yaml down
```
**Important for GPU Usage:**
@@ -471,6 +471,7 @@ See [Embeddings](/Models/embeddings) for full guidance.
| --- | --- | --- |
| `EMBEDDINGS_NAME` | `huggingface_sentence-transformers/all-mpnet-base-v2` | The embedding model. New installs use `ibm-granite/granite-embedding-311m-multilingual-r2`. Changing it on a populated index requires `application.scripts.reembed`. |
| `EMBEDDINGS_BASE_URL` | unset | Base URL of a remote OpenAI-compatible embeddings server. Setting it routes all embedding calls there. |
| `EMBEDDINGS_THREADS` | unset (Docker image: `4`) | Threads one local FastEmbed/onnxruntime session may use. onnxruntime otherwise sizes its pool to the host's core count, which a CPU-limited container still reports, so the image pins it like `OMP_NUM_THREADS`. Raise it on a large dedicated worker. |
| `EMBEDDINGS_KEY` | unset | Optional bearer token for the remote embeddings server. |
| `EMBEDDINGS_MAX_INPUT_TOKENS` | unset | Truncate each remote embedding input to N tokens (guards servers that reject oversized inputs). |
| `EMBEDDINGS_DELEGATE_TO_WORKER` | `true` | Embed queries on the Celery worker instead of loading a model in the API. Requires a worker consuming `EMBEDDINGS_QUEUE`; set `false` to run the API standalone. Ignored when `EMBEDDINGS_BASE_URL` is set. |
+27 -13
View File
@@ -64,11 +64,17 @@ docker build --build-arg INSTALL_TESSERACT=true ./application
`INSTALL_TESSERACT=true` in `.env` (or the shell) bakes tesseract plus the
English pack into locally built backend and worker images; `setup.sh` writes
it when you answer yes to the OCR question after choosing to build images
locally. Pre-built Docker Hub images (`docker-compose-hub.yaml`) do not
include it, so `setup.sh` leaves OCR at its default (off) for them; to OCR
there, point `OCR_ENGINE=deepseek` at a DeepSeek-OCR endpoint or install
`tesseract-ocr` in a derived image. With `OCR_ENABLED=true` and no binary on
`PATH`, scanned pages fail with an install hint (text-layer documents are
locally.
With pre-built images the switch is the image variant: every tag is
published twice, slim (`arc53/docsgpt:<tag>`) and `-docling`
(`arc53/docsgpt:<tag>-docling`), and the latter bakes tesseract, the docling
engine and its models in. Set `DOCSGPT_IMAGE_VARIANT=-docling` in `.env` for
`docker-compose-hub.yaml` or `docker-compose-standalone.yaml`; `setup.sh`
writes it when you answer yes to the OCR question with Docker Hub images.
Alternatively point `OCR_ENGINE=deepseek` at a DeepSeek-OCR endpoint, which
needs no system package. With `OCR_ENABLED=true` and no binary on `PATH`,
scanned pages fail with an install hint (text-layer documents are
unaffected).
<Callout type="warning" emoji="⚠️">
@@ -88,23 +94,31 @@ docling is not part of the base install, and OCR does not need it (see
output:
```bash
pip install -r application/requirements-docling.txt
pip install -r application/requirements-docling.txt # or: uv sync --extra docling
```
Docker images build without it by default; opt in with the build argument:
That file is the core set plus the `docling` extra, exported from the same
lock. On Linux it takes torch from the CPU-only PyTorch index, so the extra
costs about 1.5 GB rather than the 2.7 GB the CUDA build of torch would; a
GPU deployment can reinstall torch from PyPI on top.
Pre-built images: use the `-docling` variant (`arc53/docsgpt:<tag>-docling`,
`DOCSGPT_IMAGE_VARIANT=-docling` in `.env`), which also bakes docling's
layout, table-structure and RapidOCR models in so the first parse does not
download them. Local builds opt in with the build argument:
```bash
docker build --build-arg INSTALL_DOCLING=true ./application
docker build --build-arg EXTRAS=docling ./application
```
`deployment/docker-compose.yaml` forwards the same switch, so setting
`INSTALL_DOCLING=true` in `.env` (or the shell) bakes docling into locally
built backend and worker images; `setup.sh` offers it as a follow-up to the
OCR question. Compose reads build arguments from the shell or from the
`EXTRAS=docling` (or the older `INSTALL_DOCLING=true`) in `.env` (or the
shell) bakes docling into locally built backend and worker images; `setup.sh`
offers it as a follow-up to the OCR question. Compose reads build arguments from the shell or from the
`.env` you pass with `--env-file .env` (not from the containers' `env_file`),
so build with `docker compose --env-file .env -f deployment/docker-compose.yaml build`
as `setup.sh` does. Pre-built Docker Hub images (`docker-compose-hub.yaml`)
never include docling; install it in a derived image instead. Either way no
as `setup.sh` does. Of the pre-built Docker Hub images only the slim default
excludes docling; the `-docling` variant ships it with its models. Either way no
code changes are needed — docling is picked up as
the fallback engine (and, under `OCR_BACKEND=auto`, as the OCR backend) as
soon as it is importable, and `DOC_PARSER_ENGINE=docling` makes it the
+8
View File
@@ -0,0 +1,8 @@
node_modules/
dist/
# Local overrides never belong in the image; .env.development and .env.production do.
.env.local
.env.*.local
Dockerfile
.dockerignore
*.log
+48 -4
View File
@@ -1,11 +1,55 @@
FROM node:22-bullseye-slim
# DocsGPT frontend image: a static production build served by nginx.
#
# Vite inlines VITE_* settings at build time, so the old image ran the Vite
# dev server just to read VITE_API_HOST from the container environment. This
# image builds once and injects the container's VITE_* variables at start-up
# instead (docker/40-runtime-env.sh writes them to /config.js, which the app
# reads before its own bundle; see src/env.ts). Same env vars, same port.
#
# Targets:
# (default) nginx serving the built bundle -- what Docker Hub publishes
# dev the Vite dev server with hot reload, for docker-compose.yaml's
# bind-mounted frontend (build: target: dev)
FROM node:22-alpine AS deps
WORKDIR /app
COPY package*.json ./
RUN npm install
RUN npm ci --no-audit --no-fund
FROM deps AS dev
COPY . .
# vite.config.ts polls the filesystem when DOCKER is set: native fs events do
# not cross a Windows host into a Linux container.
ENV DOCKER=1
EXPOSE 5173
CMD ["npm", "run", "dev", "--", "--host"]
CMD [ "npm", "run", "dev", "--" , "--host"]
FROM deps AS build
COPY . .
# The image used to run the dev server, so .env.development was its set of
# defaults (notification banner, Google client id, local API host). Keep them
# as the production build's baseline; the container's VITE_* values override
# any of them at start-up.
RUN cp .env.development .env.production.local && \
npm run build && \
sed -i 's|<head>|<head><script src="/config.js"></script>|' dist/index.html
FROM nginx:1.27-alpine
LABEL org.opencontainers.image.source="https://github.com/arc53/DocsGPT" \
org.opencontainers.image.title="DocsGPT frontend" \
org.opencontainers.image.licenses="MIT"
COPY docker/nginx.conf /etc/nginx/conf.d/default.conf
COPY docker/40-runtime-env.sh /docker-entrypoint.d/40-runtime-env.sh
RUN chmod +x /docker-entrypoint.d/40-runtime-env.sh
COPY --from=build /app/dist /usr/share/nginx/html
# Same port the dev server used, so compose files and docs keep working.
EXPOSE 5173
+25
View File
@@ -0,0 +1,25 @@
#!/bin/sh
# Expose the container's VITE_* environment to the static bundle.
#
# nginx's entrypoint runs everything in /docker-entrypoint.d before serving.
# The generated /config.js is loaded by index.html ahead of the app bundle and
# read by src/env.ts, so VITE_API_HOST and friends can differ per deployment
# without rebuilding the image.
set -eu
out=/usr/share/nginx/html/config.js
{
printf 'window.__DOCSGPT_ENV__ = {'
first=1
env | grep -E '^VITE_[A-Za-z0-9_]+=.' | while IFS='=' read -r key value; do
# Empty values are skipped above (=.) so a compose passthrough like
# ${VITE_X:-} leaves the build-time default in place.
# JSON-escape backslashes and double quotes; values are plain URLs/ids.
escaped=$(printf '%s' "$value" | sed 's/\\/\\\\/g; s/"/\\"/g')
if [ "$first" -eq 1 ]; then first=0; else printf ','; fi
printf '"%s":"%s"' "$key" "$escaped"
done
printf '};\n'
} > "$out"
echo "runtime-env: wrote $(grep -o 'VITE_[A-Za-z0-9_]*' "$out" | wc -l | tr -d ' ') VITE_* values to /config.js"
+25
View File
@@ -0,0 +1,25 @@
server {
listen 5173;
server_name _;
root /usr/share/nginx/html;
index index.html;
gzip on;
gzip_types text/plain text/css application/javascript application/json image/svg+xml;
# Hashed bundle assets are immutable; index.html and config.js are not.
location /assets/ {
add_header Cache-Control "public, max-age=31536000, immutable";
try_files $uri =404;
}
location = /config.js {
add_header Cache-Control "no-store";
}
# Single-page app: every unknown path renders index.html.
location / {
add_header Cache-Control "no-cache";
try_files $uri $uri/ /index.html;
}
}
+3 -2
View File
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import './locale/i18n';
import { useState } from 'react';
@@ -108,8 +109,8 @@ export default function App() {
const saved = localStorage.getItem('showNotification');
return saved ? JSON.parse(saved) : true;
});
const notificationText = import.meta.env.VITE_NOTIFICATION_TEXT;
const notificationLink = import.meta.env.VITE_NOTIFICATION_LINK;
const notificationText = envVar('VITE_NOTIFICATION_TEXT');
const notificationLink = envVar('VITE_NOTIFICATION_LINK');
// Hide the changelog banner on public share routes — those pages are
// embedded / shared externally and shouldn't carry product chrome.
const isPublicShareRoute =
+2 -1
View File
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import { createAsyncThunk, createSlice, PayloadAction } from '@reduxjs/toolkit';
import {
@@ -26,7 +27,7 @@ const initialState: ConversationState = {
conversationId: null,
};
const API_STREAMING = import.meta.env.VITE_API_STREAMING === 'true';
const API_STREAMING = envVar('VITE_API_STREAMING') === 'true';
let abortController: AbortController | null = null;
export function handlePreviewAbort() {
+2 -1
View File
@@ -1,7 +1,8 @@
import { envVar } from '@/env';
import { withThrottle, type FetchLike } from './throttle';
export const baseURL =
import.meta.env.VITE_API_HOST || 'https://docsapi.arc53.com';
envVar('VITE_API_HOST') || 'https://docsapi.arc53.com';
const getHeaders = (
token: string | null,
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import React, { useState, useEffect } from 'react';
import { useTranslation } from 'react-i18next';
import drivePickerImport from 'react-google-drive-picker';
@@ -121,9 +122,9 @@ const GoogleDrivePicker: React.FC<GoogleDrivePickerProps> = ({
}
try {
const clientId: string = import.meta.env.VITE_GOOGLE_CLIENT_ID;
const clientId: string = envVar('VITE_GOOGLE_CLIENT_ID');
const developerKey: string =
import.meta.env.VITE_GOOGLE_PICKER_API_KEY ?? '';
envVar('VITE_GOOGLE_PICKER_API_KEY') ?? '';
// Derive appId from clientId (extract numeric part before first dash)
const appId = clientId ? clientId.split('-')[0] : null;
+3 -2
View File
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import {
useCallback,
useEffect,
@@ -63,7 +64,7 @@ const LIVE_TRANSCRIPTION_TIMESLICE_MS = 1000;
const LIVE_CAPTURE_SAMPLE_RATE = 16000;
const LIVE_CAPTURE_MAX_BUFFER_SECONDS = 20;
const LIVE_SILENCE_RMS_THRESHOLD = 0.015;
const ENABLE_VOICE_INPUT = import.meta.env.VITE_ENABLE_VOICE_INPUT === 'true';
const ENABLE_VOICE_INPUT = envVar('VITE_ENABLE_VOICE_INPUT') === 'true';
type AudioContextWindow = Window &
typeof globalThis & {
@@ -526,7 +527,7 @@ export default function MessageInput({
if (supported.length === 0) return;
const files = supported;
const apiHost = import.meta.env.VITE_API_HOST;
const apiHost = envVar('VITE_API_HOST');
if (files.length > 1) {
const formData = new FormData();
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import 'katex/dist/katex.min.css';
import { Pencil } from 'lucide-react';
@@ -37,7 +38,7 @@ import ResearchProgress from './ResearchProgress';
import { ToolCallsType } from './types';
import { wikiWriteActionKey, wikiWritePath } from './wikiToolCall';
const DisableSourceFE = import.meta.env.VITE_DISABLE_SOURCE_FE || false;
const DisableSourceFE = envVar('VITE_DISABLE_SOURCE_FE') === 'true';
const ConversationBubble = forwardRef<
HTMLDivElement,
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import {
createAsyncThunk,
createListenerMiddleware,
@@ -90,8 +91,8 @@ const initialState: ConversationState = {
conversationId: null,
};
const API_STREAMING = import.meta.env.VITE_API_STREAMING === 'true';
const USE_V1_API = import.meta.env.VITE_USE_V1_API === 'true';
const API_STREAMING = envVar('VITE_API_STREAMING') === 'true';
const USE_V1_API = envVar('VITE_USE_V1_API') === 'true';
let abortController: AbortController | null = null;
export function handleAbort() {
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import { createSlice } from '@reduxjs/toolkit';
import type { PayloadAction } from '@reduxjs/toolkit';
import store from '../store';
@@ -12,7 +13,7 @@ import {
clearAttachments,
} from '../upload/uploadSlice';
const API_STREAMING = import.meta.env.VITE_API_STREAMING === 'true';
const API_STREAMING = envVar('VITE_API_STREAMING') === 'true';
interface SharedConversationsType {
queries: Query[];
apiKey?: string;
+24
View File
@@ -0,0 +1,24 @@
/**
* Runtime-overridable build settings.
*
* Vite inlines `import.meta.env.VITE_*` at build time, which forced the
* Docker image to run the dev server so `VITE_API_HOST` could change per
* deployment. The production image serves a static build instead and writes
* the container's `VITE_*` environment into `window.__DOCSGPT_ENV__` (see
* frontend/docker/40-runtime-env.sh). That object wins over the build-time
* value; outside Docker nothing sets it and the build-time value applies.
*/
declare global {
interface Window {
__DOCSGPT_ENV__?: Record<string, string | undefined>;
}
}
export function envVar(name: string): string {
const runtime =
typeof window !== 'undefined' ? window.__DOCSGPT_ENV__?.[name] : undefined;
if (runtime !== undefined && runtime !== '') return runtime;
const buildTime = (import.meta.env as Record<string, unknown>)[name];
return typeof buildTime === 'string' ? buildTime : '';
}
+2 -1
View File
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import { useEffect, useState } from 'react';
import { useTranslation } from 'react-i18next';
import { useSelector } from 'react-redux';
@@ -13,7 +14,7 @@ import { ActiveState } from '../models/misc';
import { selectToken } from '../preferences/preferenceSlice';
import ConfirmationModal from './ConfirmationModal';
const baseURL = import.meta.env.VITE_BASE_URL;
const baseURL = envVar('VITE_BASE_URL');
type AgentDetailsModalProps = {
agent: Agent;
+3 -2
View File
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import { useCallback, useEffect, useState } from 'react';
import { nanoid } from '@reduxjs/toolkit';
import { useDropzone } from 'react-dropzone';
@@ -575,7 +576,7 @@ function Upload({
JSON.stringify(optionsToConfig(retrievalOptions)),
);
const apiHost = import.meta.env.VITE_API_HOST;
const apiHost = envVar('VITE_API_HOST');
const xhr = new XMLHttpRequest();
dispatch(
@@ -706,7 +707,7 @@ function Upload({
formData.append('data', JSON.stringify(configData));
const apiHost: string = import.meta.env.VITE_API_HOST;
const apiHost: string = envVar('VITE_API_HOST');
const endpoint =
ingestor.type === 'local_file'
? `${apiHost}/api/upload`
+4 -3
View File
@@ -1,3 +1,4 @@
import { envVar } from '@/env';
import CrawlerIcon from '../../assets/crawler.svg';
import FileUploadIcon from '../../assets/file_upload.svg';
import UrlIcon from '../../assets/url.svg';
@@ -146,7 +147,7 @@ export const IngestorFormSchemas: IngestorSchema[] = [
icon: DriveIcon,
heading: 'Upload from Google Drive',
validate: () => {
const googleClientId = import.meta.env.VITE_GOOGLE_CLIENT_ID;
const googleClientId = envVar('VITE_GOOGLE_CLIENT_ID');
return !!googleClientId;
},
fields: [
@@ -208,7 +209,7 @@ export const IngestorFormSchemas: IngestorSchema[] = [
icon: SharePoint,
heading: 'Upload from Share Point',
validate: () => {
const sharePointClientId = import.meta.env.VITE_SHARE_POINT_CLIENT_ID;
const sharePointClientId = envVar('VITE_SHARE_POINT_CLIENT_ID');
return !!sharePointClientId;
},
fields: [
@@ -226,7 +227,7 @@ export const IngestorFormSchemas: IngestorSchema[] = [
icon: ConfluenceIcon,
heading: 'Upload from Confluence',
validate: () => {
const confluenceClientId = import.meta.env.VITE_CONFLUENCE_CLIENT_ID;
const confluenceClientId = envVar('VITE_CONFLUENCE_CLIENT_ID');
return !!confluenceClientId;
},
fields: [
+149
View File
@@ -0,0 +1,149 @@
[project]
name = "docsgpt"
# Bump together with application/version.py and frontend/package.json.
version = "0.19.0"
description = "DocsGPT backend: chat with your documents, agents, and tools."
readme = "README.md"
requires-python = ">=3.12"
license = { file = "LICENSE" }
# Direct dependencies only. Transitive pins live in uv.lock; the pip-facing
# files under application/ (requirements*.txt) are exported from that lock by
# scripts/export_requirements.sh and must not be edited by hand.
dependencies = [
"a2wsgi==1.10.10",
"alembic>=1.13,<2",
"anthropic==0.121.0",
"beautifulsoup4==4.15.0",
"boto3==1.43.67",
"cel-python==0.5.0",
"celery==5.6.3",
"celery-redbeat==2.4.2",
"croniter==6.2.4",
"cryptography==50.0.0",
"dataclasses-json==0.6.7",
"daytona==0.205.1",
"ddgs>=8.0.0",
"defusedxml==0.7.1",
"docx2txt==0.9",
"elevenlabs==2.62.0",
"faiss-cpu==1.15.0",
"fast-ebook",
# Default document converter (DOC_PARSER_ENGINE=anydoc): a Rust extension
# with no model downloads. The docling engine is the `docling` extra.
"firecrawl-anydoc==0.2.3",
"fastembed==0.8.0",
"fastmcp==3.4.6",
"Flask==3.1.3",
"flask-restx==1.3.2",
"google-api-python-client==2.198.0",
"google-auth-oauthlib==1.4.0",
"google-genai==2.17.0",
"gTTS==2.5.4",
"gunicorn==26.0.0",
"jinja2==3.1.6",
"kombu==5.6.2",
"markdownify==1.2.3",
"msal==1.37.0",
"networkx==3.6.1",
"numpy==2.5.1",
# fastembed's runtime: local embeddings execute on it.
"onnxruntime==1.28.0",
"openai==2.53.0",
"openapi3-parser==1.1.22",
# pandas reads .xlsx through openpyxl but does not depend on it.
"openpyxl==3.1.5",
"opentelemetry-distro>=0.50b0,<1",
"opentelemetry-exporter-otlp>=1.29.0,<2",
"opentelemetry-instrumentation-celery>=0.50b0,<1",
"opentelemetry-instrumentation-flask>=0.50b0,<1",
"opentelemetry-instrumentation-logging>=0.50b0,<1",
"opentelemetry-instrumentation-psycopg>=0.50b0,<1",
"opentelemetry-instrumentation-redis>=0.50b0,<1",
"opentelemetry-instrumentation-requests>=0.50b0,<1",
"opentelemetry-instrumentation-sqlalchemy>=0.50b0,<1",
"opentelemetry-instrumentation-starlette>=0.50b0,<1",
"pandas==3.0.5",
"pdf2image>=1.17.0",
"pgvector>=0.5,<1",
"pillow==12.3.0",
"praw==8.0.2",
"psycopg[binary,pool]>=3.1,<4",
"pydantic",
"pydantic-settings",
"pypdf==6.15.0",
"pypdfium2==5.12.1",
"python-dateutil==2.9.0.post0",
"python-dotenv",
"python-jose==3.5.0",
"python-pptx==1.0.2",
"PyYAML",
"qdrant-client==1.19.0",
"redis==7.4.0",
"requests==2.34.2",
"retry==0.9.2",
"sqlalchemy>=2.0,<3",
"starlette>=1.0,<2",
"tiktoken==0.13.0",
"tldextract==5.3.2",
"tokenizers==0.22.2",
"tqdm==4.67.3",
"uvicorn[standard]>=0.30,<1",
"uvicorn-worker>=0.4,<1",
"websocket-client==1.9.0",
"werkzeug>=3.1.0",
]
[project.optional-dependencies]
# Docling parser engine: DOC_PARSER_ENGINE=docling, the docling OCR backend
# (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
# attachment parsing, and read_document's `structured` output. Pulls torch and
# transformers; on Linux torch resolves from the CPU-only PyTorch index (see
# [tool.uv.sources]) so the extra does not drag the CUDA stack in.
docling = [
"docling==2.119.0",
"rapidocr==3.9.2",
# docling's model stack. Declared here (not left transitive) so the pins
# hold and the CPU index source below applies. transformers is capped by
# docling-core at <5.9: 5.9+ breaks the PDF layout model on Apple Silicon.
"torch==2.11.0",
"torchvision==0.26.0",
"transformers==5.8.1",
]
# VECTOR_STORE=milvus. milvus-lite (the embedded server) pulls pyarrow.
milvus = [
"pymilvus==3.0.1",
"milvus-lite==3.2.0; sys_platform != 'win32'",
]
[dependency-groups]
# Mirrors tests/requirements.txt for `uv sync`; pip users install that file.
dev = [
"pytest>=8.0.0",
"pytest-asyncio>=0.23",
"pytest-cov>=4.1.0",
"pytest-xdist>=3.5",
"coverage>=7.4.0",
"pytest-postgresql>=6.0.0",
"jupyter-client>=8.0",
"python-docx>=1.1",
"reportlab>=4.0,<5",
"ruff",
]
[tool.uv]
# The repo is run in place (`application.*` imported from the checkout), not
# installed as a distribution.
package = false
[[tool.uv.index]]
name = "pytorch-cpu"
url = "https://download.pytorch.org/whl/cpu"
explicit = true
[tool.uv.sources]
# PyPI's Linux torch wheels depend on the full CUDA 13 stack (~2.7 GB of
# wheels). The docling extra runs its models on CPU, so take torch from the
# CPU index there. macOS and Windows PyPI wheels are CPU-only already.
torch = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
torchvision = [{ index = "pytorch-cpu", marker = "sys_platform == 'linux'" }]
+68
View File
@@ -0,0 +1,68 @@
#!/usr/bin/env bash
# Regenerate the pip-facing requirements files from uv.lock.
#
# pyproject.toml declares the direct dependencies and the optional extras;
# uv.lock pins everything. pip users, the Dockerfile and CI install from the
# exported files, so run this after any change to pyproject.toml or uv.lock:
#
# uv lock # or: uv lock --upgrade-package <name>
# bash scripts/export_requirements.sh
#
# Each exported file is a complete environment (core plus the named extra),
# so `pip install -r application/requirements-docling.txt` on its own works,
# and installing it on top of requirements.txt only adds the extra's packages.
set -euo pipefail
cd "$(dirname "$0")/.."
UV=(uv)
if ! uv --version 2>/dev/null | grep -qE '^uv 0\.([89]|[1-9][0-9])\.'; then
# `uv export` needs a current uv; run one through uvx without touching the
# machine's install.
UV=(uv tool run --from 'uv>=0.8' uv)
fi
export_file() {
local out="$1"; shift
local header="$1"; shift
local index_url="${INDEX_URL:-}"
{
echo "# GENERATED by scripts/export_requirements.sh from uv.lock -- do not edit."
echo "#"
# shellcheck disable=SC2001
echo "$header" | sed 's/^/# /'
echo
# uv export records the wheel's origin in uv.lock but writes no index
# directive; pip needs one to find the +cpu torch build.
if [ -n "$index_url" ]; then echo "--extra-index-url $index_url"; echo; fi
"${UV[@]}" export --frozen --no-hashes --no-dev --no-emit-project --no-header --quiet "$@"
} > "$out"
echo "wrote $out ($(grep -cE '^[A-Za-z0-9]' "$out") packages)"
}
export_file application/requirements.txt \
"Core runtime. Optional extras live in requirements-<extra>.txt:
docling DOC_PARSER_ENGINE=docling, docling OCR backend, read_document structured output
milvus VECTOR_STORE=milvus"
INDEX_URL=https://download.pytorch.org/whl/cpu export_file application/requirements-docling.txt \
"Core runtime plus the docling extra: DOC_PARSER_ENGINE=docling, the docling
OCR backend (layout-model hybrid OCR, ocrmac/rapidocr engines), .adoc/.vtt/.xml
attachment parsing, and read_document's 'structured' output. The default
anydoc engine needs none of this, and OCR itself does not either:
OCR_ENABLED=true with the tesseract binary (or a DeepSeek-OCR endpoint) runs
through application/parser/file/ocr_parser.py.
On Linux torch comes from the CPU-only PyTorch index (no CUDA stack); a GPU
deployment can reinstall torch from PyPI on top.
pip resolves the extra index as expected. uv only takes a package from the
first index that lists it, and the PyTorch index carries stale copies of
common packages, so with uv either run 'uv sync --extra docling' (the lock
pins the index per package) or set UV_INDEX_STRATEGY=unsafe-best-match.
Docker: --build-arg EXTRAS=docling" \
--extra docling
export_file application/requirements-milvus.txt \
"Core runtime plus the milvus extra (VECTOR_STORE=milvus): pymilvus and the
embedded milvus-lite server, which pulls pyarrow.
Docker: --build-arg EXTRAS=milvus" \
--extra milvus
+24 -21
View File
@@ -513,29 +513,32 @@ function Configure-DocProcessing {
Write-ColorText "PDF-as-image parsing enabled." -ForegroundColor "Green"
}
# OCR needs the tesseract binary, an optional system package that only
# locally built images can include (INSTALL_TESSERACT build arg). The
# pre-built Docker Hub images ship without it, so there OCR stays off
# (its default) rather than being switched on to fail on every scan.
if ($COMPOSE_FILE -ne $COMPOSE_FILE_LOCAL) {
Write-ColorText "OCR for scanned PDFs and images stays off: the pre-built Docker Hub images do not include tesseract. To use OCR, rerun setup and choose option 5 (build images locally), or use a DeepSeek-OCR endpoint by adding OCR_ENABLED=true, OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> to .env." -ForegroundColor "Yellow"
# OCR needs the tesseract binary. The default (slim) images ship without
# it; the pre-built "-docling" image variant bakes tesseract, the docling
# layout engine and its models in, so with Docker Hub images OCR means
# switching the variant. Locally built images get it via build args.
$ocr_enabled = Read-Host "Enable OCR for scanned PDFs and images? (y/N)"
if (-not ($ocr_enabled -eq "y" -or $ocr_enabled -eq "Y")) {
return
}
$ocr_enabled = Read-Host "Enable OCR for scanned PDFs and images? (y/N)"
if ($ocr_enabled -eq "y" -or $ocr_enabled -eq "Y") {
"OCR_ENABLED=true" | Add-Content -Path $ENV_FILE -Encoding utf8
# Bakes tesseract into the locally built images (docker compose
# --env-file .env build).
"INSTALL_TESSERACT=true" | Add-Content -Path $ENV_FILE -Encoding utf8
Write-ColorText "OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env." -ForegroundColor "Green"
$docling_ocr = Read-Host "Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N)"
if ($docling_ocr -eq "y" -or $docling_ocr -eq "Y") {
# Locally built images include docling via this build arg; it becomes
# the OCR backend automatically (OCR_BACKEND=auto).
"INSTALL_DOCLING=true" | Add-Content -Path $ENV_FILE -Encoding utf8
Write-ColorText "Docling will be built into locally built images (docker compose --env-file .env build). Pre-built Docker Hub images do not include it." -ForegroundColor "Green"
}
"OCR_ENABLED=true" | Add-Content -Path $ENV_FILE -Encoding utf8
if ($COMPOSE_FILE -ne $COMPOSE_FILE_LOCAL) {
# Pre-built images: pull arc53/docsgpt:<tag>-docling instead of the
# slim default (about 1.5 GB more to download).
"DOCSGPT_IMAGE_VARIANT=-docling" | Add-Content -Path $ENV_FILE -Encoding utf8
Write-ColorText "OCR enabled. The -docling image variant will be pulled (tesseract, docling layout engine and its models included). For a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env." -ForegroundColor "Green"
return
}
# Bakes tesseract into the locally built images (docker compose
# --env-file .env build).
"INSTALL_TESSERACT=true" | Add-Content -Path $ENV_FILE -Encoding utf8
Write-ColorText "OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env." -ForegroundColor "Green"
$docling_ocr = Read-Host "Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N)"
if ($docling_ocr -eq "y" -or $docling_ocr -eq "Y") {
# Locally built images include docling via this build arg; it becomes
# the OCR backend automatically (OCR_BACKEND=auto).
"INSTALL_DOCLING=true" | Add-Content -Path $ENV_FILE -Encoding utf8
Write-ColorText "Docling will be built into locally built images (docker compose --env-file .env build)." -ForegroundColor "Green"
}
}
+24 -21
View File
@@ -367,29 +367,32 @@ configure_doc_processing() {
echo -e "${GREEN}PDF-as-image parsing enabled.${NC}"
fi
# OCR needs the tesseract binary, an optional system package that only
# locally built images can include (INSTALL_TESSERACT build arg). The
# pre-built Docker Hub images ship without it, so there OCR stays off
# (its default) rather than being switched on to fail on every scan.
if [[ "$COMPOSE_FILE" != "$COMPOSE_FILE_LOCAL" ]]; then
echo -e "${YELLOW}OCR for scanned PDFs and images stays off: the pre-built Docker Hub images do not include tesseract. To use OCR, rerun setup and choose option 5 (build images locally), or use a DeepSeek-OCR endpoint by adding OCR_ENABLED=true, OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> to .env.${NC}"
# OCR needs the tesseract binary. The default (slim) images ship without
# it; the pre-built "-docling" image variant bakes tesseract, the docling
# layout engine and its models in, so with Docker Hub images OCR means
# switching the variant. Locally built images get it via build args.
read -p "$(echo -e "${DEFAULT_FG}Enable OCR for scanned PDFs and images? (y/N): ${NC}")" ocr_enabled
if [[ ! "$ocr_enabled" =~ ^[yY]$ ]]; then
return
fi
read -p "$(echo -e "${DEFAULT_FG}Enable OCR for scanned PDFs and images? (y/N): ${NC}")" ocr_enabled
if [[ "$ocr_enabled" =~ ^[yY]$ ]]; then
echo "OCR_ENABLED=true" >> "$ENV_FILE"
# Bakes tesseract into the locally built images (docker compose
# --env-file .env build).
echo "INSTALL_TESSERACT=true" >> "$ENV_FILE"
echo -e "${GREEN}OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env.${NC}"
read -p "$(echo -e "${DEFAULT_FG}Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N): ${NC}")" docling_ocr
if [[ "$docling_ocr" =~ ^[yY]$ ]]; then
# Locally built images include docling via this build arg; it becomes
# the OCR backend automatically (OCR_BACKEND=auto).
echo "INSTALL_DOCLING=true" >> "$ENV_FILE"
echo -e "${GREEN}Docling will be built into locally built images (docker compose --env-file .env build). Pre-built Docker Hub images do not include it.${NC}"
fi
echo "OCR_ENABLED=true" >> "$ENV_FILE"
if [[ "$COMPOSE_FILE" != "$COMPOSE_FILE_LOCAL" ]]; then
# Pre-built images: pull arc53/docsgpt:<tag>-docling instead of the
# slim default (about 1.5 GB more to download).
echo "DOCSGPT_IMAGE_VARIANT=-docling" >> "$ENV_FILE"
echo -e "${GREEN}OCR enabled. The -docling image variant will be pulled (tesseract, docling layout engine and its models included). For a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env.${NC}"
return
fi
# Bakes tesseract into the locally built images (docker compose
# --env-file .env build).
echo "INSTALL_TESSERACT=true" >> "$ENV_FILE"
echo -e "${GREEN}OCR enabled. tesseract will be built into the images (INSTALL_TESSERACT=true); for a DeepSeek-OCR endpoint instead, set OCR_ENGINE=deepseek and OCR_DEEPSEEK_URL=<endpoint> in .env.${NC}"
read -p "$(echo -e "${DEFAULT_FG}Also install the Docling layout engine for OCR (better tables/reading order, several GB heavier)? (y/N): ${NC}")" docling_ocr
if [[ "$docling_ocr" =~ ^[yY]$ ]]; then
# Locally built images include docling via this build arg; it becomes
# the OCR backend automatically (OCR_BACKEND=auto).
echo "INSTALL_DOCLING=true" >> "$ENV_FILE"
echo -e "${GREEN}Docling will be built into locally built images (docker compose --env-file .env build).${NC}"
fi
}
+60
View File
@@ -0,0 +1,60 @@
"""Optional-extra bookkeeping: hints name the extra, require() explains absence."""
import sys
import types
from unittest.mock import patch
import pytest
from application.core import optional_deps
class TestExtras:
def test_every_extra_module_maps_back(self):
for extra, modules in optional_deps.EXTRAS.items():
for module in modules:
assert optional_deps.extra_for(module) == extra
def test_submodule_resolves_to_its_extra(self):
assert optional_deps.extra_for("docling.document_converter") == "docling"
def test_unknown_module_has_no_extra(self):
assert optional_deps.extra_for("flask") is None
def test_install_hint_names_every_install_route(self):
hint = optional_deps.install_hint("milvus")
assert "requirements-milvus.txt" in hint
assert "--extra milvus" in hint
assert "EXTRAS=milvus" in hint
class TestMissingMessage:
def test_extra_module_points_at_the_extra(self):
message = optional_deps.missing_message("pymilvus", "VECTOR_STORE=milvus")
assert "pymilvus is not installed (VECTOR_STORE=milvus)" in message
assert "'milvus' extra" in message
assert optional_deps.install_hint("milvus") in message
def test_plain_module_gets_a_pip_line(self):
assert optional_deps.missing_message("boto3") == "boto3 is not installed. Install it with: pip install boto3"
class TestRequire:
def test_returns_the_module_when_present(self):
assert optional_deps.require("json").dumps({}) == "{}"
def test_absent_module_raises_with_hint(self):
with patch.dict(sys.modules, {"pymilvus": None}):
with pytest.raises(ImportError) as excinfo:
optional_deps.require("pymilvus", "VECTOR_STORE=milvus")
assert "'milvus' extra" in str(excinfo.value)
class TestIsAvailable:
def test_stubbed_module_counts_as_present(self):
with patch.dict(sys.modules, {"docling": types.ModuleType("docling")}):
assert optional_deps.is_available("docling.document_converter")
def test_blocked_module_counts_as_absent(self):
with patch.dict(sys.modules, {"docling": None}):
assert not optional_deps.is_available("docling")
+13
View File
@@ -68,6 +68,19 @@ class TestDoclingParserInitParser:
with pytest.raises(ImportError, match="docling is required"):
parser._init_parser()
def test_init_parser_names_the_extra_when_docling_is_absent(self, monkeypatch):
"""A missing parent package makes find_spec raise; the hint must still show."""
import sys
from application.parser.file.docling_parser import DoclingParser
for name in [m for m in sys.modules if m == "docling" or m.startswith("docling.")]:
monkeypatch.delitem(sys.modules, name)
monkeypatch.setitem(sys.modules, "docling", None)
with pytest.raises(ImportError, match="requirements-docling.txt"):
DoclingParser()._init_parser()
def test_init_parser_success(self):
from application.parser.file.docling_parser import DoclingParser
+1 -1
View File
@@ -48,7 +48,7 @@ def _patch_validate_url(monkeypatch):
@pytest.fixture(autouse=True)
def _patch_tldextract(monkeypatch):
monkeypatch.setattr(
"application.parser.remote.crawler_markdown.tldextract.extract",
"application.parser.remote.crawler_markdown._extract",
_fake_extract,
)
+35
View File
@@ -1,6 +1,9 @@
"""Chunk sizes must be counted in the embedding model's units, and splitting
must never rewrite the text it splits."""
import sys
import types
import pytest
from application.parser import tokenization
@@ -298,3 +301,35 @@ class TestUnknownTokenCollapse:
assert "".join(pieces) == text, "split must not lose or alter text"
assert len(pieces) > 1, "a collapsed run must still be cut into pieces"
assert all(counter.count(p) <= 20 for p in pieces)
class TestTokenizerFile:
"""The chunker's tokenizer must come from the hub cache without a network round trip."""
def test_cache_hit_makes_no_online_call(self, monkeypatch):
calls = []
def fake_download(repo, filename, local_files_only=False):
calls.append(local_files_only)
return "/cache/tokenizer.json"
fake_hub = types.ModuleType("huggingface_hub")
fake_hub.hf_hub_download = fake_download
monkeypatch.setitem(sys.modules, "huggingface_hub", fake_hub)
assert tokenization._tokenizer_file("org/model") == "/cache/tokenizer.json"
assert calls == [True]
def test_cache_miss_falls_back_to_online(self, monkeypatch):
calls = []
def fake_download(repo, filename, local_files_only=False):
calls.append(local_files_only)
if local_files_only:
raise FileNotFoundError("not cached")
return "/downloaded/tokenizer.json"
fake_hub = types.ModuleType("huggingface_hub")
fake_hub.hf_hub_download = fake_download
monkeypatch.setitem(sys.modules, "huggingface_hub", fake_hub)
assert tokenization._tokenizer_file("org/model") == "/downloaded/tokenizer.json"
assert calls == [True, False]
+15
View File
@@ -94,3 +94,18 @@ class TestMain:
with patch.object(prefetch_models, "prefetch", return_value=[]) as spy:
prefetch_models.main(["granite-97m"])
assert spy.call_args.args[1] == "/app/models"
class TestPrefetchTiktoken:
def test_warms_every_listed_encoding(self):
"""The image sets TIKTOKEN_CACHE_DIR; warming fills it at build time."""
fake = MagicMock()
module = types.ModuleType("tiktoken")
module.get_encoding = fake
with patch.dict(sys.modules, {"tiktoken": module}):
fetched = prefetch_models.prefetch_tiktoken()
assert fetched == list(prefetch_models.TIKTOKEN_ENCODINGS)
assert [c.args[0] for c in fake.call_args_list] == list(prefetch_models.TIKTOKEN_ENCODINGS)
def test_cl100k_is_the_encoding_token_counting_uses(self):
assert "cl100k_base" in prefetch_models.TIKTOKEN_ENCODINGS
+51
View File
@@ -0,0 +1,51 @@
"""Offline verification: every check must pass, docling only when installed."""
import sys
import types
from unittest.mock import patch
from application.scripts import verify_offline
def _fake_tiktoken(monkeypatch):
module = types.ModuleType("tiktoken")
module.get_encoding = lambda name: types.SimpleNamespace(encode=lambda text: [1, 2])
monkeypatch.setitem(sys.modules, "tiktoken", module)
class TestVerify:
def test_passes_when_every_check_passes(self, monkeypatch, capsys):
_fake_tiktoken(monkeypatch)
counter = types.SimpleNamespace(name="org/model", count=lambda text: 4)
with patch("application.parser.tokenization.get_token_counter", return_value=counter), \
patch("application.vectorstore.embeddings_local.EmbeddingsWrapper") as wrapper, \
patch.object(verify_offline, "is_available", return_value=False):
wrapper.return_value.embed_query.return_value = [0.0] * 768
assert verify_offline.verify(["ibm-granite/granite-embedding-311m-multilingual-r2"]) is True
out = capsys.readouterr().out
assert "ok tiktoken cl100k_base" in out
assert "skip docling" in out
def test_fails_when_the_tokenizer_fell_back_to_cl100k(self, monkeypatch, capsys):
"""A cache miss makes chunking silently use cl100k; that is a failed check."""
_fake_tiktoken(monkeypatch)
counter = types.SimpleNamespace(name="cl100k_base", count=lambda text: 4)
with patch("application.parser.tokenization.get_token_counter", return_value=counter), \
patch("application.vectorstore.embeddings_local.EmbeddingsWrapper") as wrapper, \
patch.object(verify_offline, "is_available", return_value=False):
wrapper.return_value.embed_query.return_value = [0.0] * 768
assert verify_offline.verify(["ibm-granite/granite-embedding-311m-multilingual-r2"]) is False
assert "FAIL tokenizer" in capsys.readouterr().out
def test_runs_the_docling_check_when_installed(self, monkeypatch):
_fake_tiktoken(monkeypatch)
with patch.object(verify_offline, "is_available", return_value=True), \
patch.object(verify_offline, "_docling_check", return_value="models from /app/models/docling") as check:
assert verify_offline.verify([]) is True
check.assert_called_once()
def test_remote_models_are_skipped(self, monkeypatch, capsys):
_fake_tiktoken(monkeypatch)
with patch.object(verify_offline, "is_available", return_value=False):
assert verify_offline.verify(["openai_text-embedding-ada-002"]) is True
assert "skip openai_text-embedding-ada-002" in capsys.readouterr().out
Generated
+7068
View File
File diff suppressed because it is too large. Load diff