Merge remote-tracking branch 'origin/main' into branding-upd

# Conflicts:
#	docsgpt/core/models/anthropic.yaml
#	docsgpt/core/models/openai.yaml
#	frontend/src/admin/AdminUI.tsx
This commit is contained in:
Alex committed 2026-09-21 17:52:53 +01:00
commit a25bec6821
300 files changed
+27029 -1745

No files matched your search

+17 -2
View File
@@ -1,11 +1,14 @@
# Build context for docsgpt/Dockerfile is the repository root, so the image can
# carry the `application` import alias next to the `docsgpt` package. Allow only
# what the image needs; everything else (frontend, docs, tests, venvs) stays out.
# what the image needs; everything else (docs, tests, venvs) stays out.
*
!docsgpt/
# Only the alias file: an upgraded checkout may still hold gitignored
# application/{inputs,indexes,vectors,.env} from the old layout.
!application/__init__.py
# The web UI's source and the script that builds it (the `ui` stage).
!frontend/
!scripts/build_frontend.sh
# Inside the package: caches, local runtime data and secrets never ship.
**/__pycache__/
@@ -23,5 +26,17 @@ docsgpt/*.pkl
docsgpt/.env
docsgpt/.env.*
docsgpt/Dockerfile
# The backend image serves no UI (the frontend image does); a local UI build stays out.
# A local UI build stays out: the `ui` stage builds the one the image serves.
docsgpt/static/
# Inside the frontend: installed packages and local builds are redone in the
# `ui` stage, and local VITE_* overrides must not be baked into a published image.
frontend/node_modules/
frontend/dist/
frontend/.env.local
frontend/.env.*.local
frontend/*.log
# The frontend image's own build files; changing them must not rebuild this UI.
frontend/Dockerfile
frontend/.dockerignore
frontend/docker/
+33
View File
@@ -0,0 +1,33 @@
# https://editorconfig.org — shared whitespace rules for every editor.
root = true
[*]
charset = utf-8
end_of_line = lf
insert_final_newline = true
trim_trailing_whitespace = true
indent_style = space
indent_size = 2
[*.{py,pyi}]
indent_size = 4
max_line_length = 120
[*.{sh,ps1,ini,toml}]
indent_size = 4
[{Dockerfile,Dockerfile.*,*.dockerfile}]
indent_size = 4
# Two trailing spaces are a hard line break in Markdown.
[*.{md,mdx}]
trim_trailing_whitespace = false
[Makefile]
indent_style = tab
# Generated or vendored; leave as produced.
[{uv.lock,package-lock.json,docsgpt/requirements*.txt}]
indent_size = unset
insert_final_newline = unset
trim_trailing_whitespace = unset
+14
View File
@@ -101,3 +101,17 @@ MICROSOFT_AUTHORITY=https://{tenantId}.ciamlogin.com/{tenantId}
# pair with OIDC_USER_ID_CLAIM=email so SCIM userName matches the OIDC user id)
# SCIM_ENABLED=false
# SCIM_TOKEN=<long random bearer token presented by the IdP's SCIM client>
# Personal access tokens (scoped API tokens for CLI and CI/CD; Settings → Access Tokens).
# Available with AUTH_TYPE=oidc or unset.
# PAT_ENABLED=true
# PAT_DEFAULT_LIFETIME_DAYS=90
# PAT_MAX_LIFETIME_DAYS=365
# PAT_ALLOW_NON_EXPIRING=false
# PAT_MAX_PER_USER=25
# Usage quotas (set limits in Admin → Quotas). Usage is counted per calendar
# day, week or month in UTC. Models without a declared price are recorded at $0
# unless a fallback [input, output] USD rate per 1M tokens is given.
# QUOTA_PERIOD=month
# QUOTA_UNPRICED_RATE_PER_MILLION=[0.5, 1.5]
@@ -3,9 +3,9 @@ Anthropic's
api
APIs
Atlassian
automations
autoescaping
Autoescaping
automations
backfill
backfills
bool
@@ -21,9 +21,9 @@ diarization
Docling
docsgpt
docstrings
enqueues
Entra
env
enqueues
EOL
ESLint
feedbacks
@@ -35,6 +35,7 @@ hardcoding
Idempotency
JSONPath
kubectl
launchd
Lightsail
llama_cpp
llm
@@ -70,6 +71,7 @@ SGLang
Shareability
Signup
Supabase
systemd
UIs
uncomment
URl
+9 -2
View File
@@ -206,9 +206,16 @@ jobs:
ref: ${{ inputs.version && format('refs/tags/{0}', inputs.version) || github.ref }}
persist-credentials: false
- name: Attach the standalone compose file to the release
# The installers are served from these assets: docs.ac/install redirects to
# releases/latest/download/install.sh (and install.ps1), so the script a
# user runs always comes from the newest release.
- name: Attach the standalone compose file and the installers to the release
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
TAG: ${{ env.RELEASE_TAG }}
run: |
gh release upload "$TAG" deployment/docker-compose-standalone.yaml --clobber
gh release upload "$TAG" \
deployment/docker-compose-standalone.yaml \
deployment/install.sh \
deployment/install.ps1 \
--clobber
+85 -3
View File
@@ -3,6 +3,8 @@ name: Verify the Docker image works offline
# Builds the backend image and runs its offline check with networking off, so
# a change that reintroduces a first-request download (a tokenizer, tiktoken's
# encoding, an embedding model) fails here instead of in an air-gapped install.
# Then starts the standalone Compose stack on the image and checks that the one
# published port serves both the API and the web UI.
on:
workflow_dispatch:
@@ -17,6 +19,15 @@ on:
- 'docsgpt/vectorstore/model_registry.py'
- 'docsgpt/parser/tokenization.py'
- 'docsgpt/vectorstore/embeddings_local.py'
- 'docsgpt/ui.py'
- 'frontend/**'
- 'scripts/build_frontend.sh'
- 'deployment/docker-compose-standalone.yaml'
- 'deployment/install.sh'
- 'docsgpt/deploy/**'
- 'docsgpt/cli.py'
- 'docsgpt/core/paths.py'
- 'pyproject.toml'
- '.github/workflows/docker-image-verify.yml'
permissions:
@@ -45,7 +56,8 @@ jobs:
context: .
platforms: linux/amd64
load: true
tags: docsgpt:verify${{ matrix.variant }}
# The name the standalone Compose file runs, under a tag no registry has.
tags: arc53/docsgpt:verify${{ matrix.variant }}
build-args: |
EXTRAS=${{ matrix.variant == '-docling' && 'docling' || '' }}
INSTALL_TESSERACT=${{ matrix.variant == '-docling' && 'true' || 'false' }}
@@ -54,14 +66,84 @@ jobs:
- name: Image size
env:
IMAGE: docsgpt:verify${{ matrix.variant }}
IMAGE: arc53/docsgpt:verify${{ matrix.variant }}
run: |
docker image inspect "$IMAGE" --format '{{.Size}}' | awk '{printf "uncompressed: %.2f GB\n", $1/1e9}'
docker history "$IMAGE" --format '{{.Size}}\t{{.CreatedBy}}' | head -20
- name: Offline verification (no network)
env:
IMAGE: docsgpt:verify${{ matrix.variant }}
IMAGE: arc53/docsgpt:verify${{ matrix.variant }}
run: |
docker run --rm --network none "$IMAGE" \
python -m docsgpt.scripts.verify_offline
- name: The standalone stack serves the API and the UI on one port
env:
DOCSGPT_IMAGE_TAG: verify
DOCSGPT_IMAGE_VARIANT: ${{ matrix.variant }}
run: |
set -euo pipefail
# --pull missing keeps the image built above; postgres and redis are pulled.
docker compose -f deployment/docker-compose-standalone.yaml up -d --pull missing backend
base=http://127.0.0.1:7091
for _ in $(seq 1 90); do
if curl -fsS "$base/api/health" >/dev/null 2>&1; then break; fi
sleep 2
done
curl -fsS "$base/api/health"
echo
curl -fsS "$base/" | grep -q 'src="/config.js"'
curl -fsS "$base/config.js" | grep -q 'window.__DOCSGPT_ENV__'
# A client-side route falls back to the UI's index.html.
curl -fsS "$base/settings" | grep -q 'src="/config.js"'
echo "API and UI served on $base"
- name: The installer runs docsgpt up on the same image
if: matrix.variant == ''
env:
DOCSGPT_NO_MODIFY_PATH: "1"
run: |
set -euo pipefail
# Same Compose project name as the step above; stop that stack first.
docker compose -f deployment/docker-compose-standalone.yaml down -v
pipx run build --wheel --outdir "$RUNNER_TEMP/dist"
# Assigned before export, so a missing wheel fails here instead of installing from PyPI.
DOCSGPT_PACKAGE="$(ls "$RUNNER_TEMP"/dist/docsgpt-*.whl)"
export DOCSGPT_PACKAGE
# No uv is set up beforehand, so the installer's pinned uv download runs too.
# Without a terminal the installer passes --yes to docsgpt up.
bash deployment/install.sh --image-tag verify </dev/null
docsgpt="$HOME/.local/bin/docsgpt"
stack="$HOME/.docsgpt/server"
"$docsgpt" status
curl -fsS http://127.0.0.1:7091/ | grep -q 'src="/config.js"'
# Each secret must appear exactly once with a value: a missing or empty one
# falls back to a default silently.
check_secrets() {
for key in POSTGRES_PASSWORD JWT_SECRET_KEY; do
[ "$(grep -Ec "^$key=.+$" "$stack/.env")" -eq 1 ]
done
}
check_secrets
secrets=$(grep -E '^(POSTGRES_PASSWORD|JWT_SECRET_KEY)=.+$' "$stack/.env" | sort)
# Running the installer again upgrades in place and keeps both secrets.
bash deployment/install.sh --image-tag verify </dev/null
check_secrets
[ "$(grep -E '^(POSTGRES_PASSWORD|JWT_SECRET_KEY)=.+$' "$stack/.env" | sort)" = "$secrets" ]
"$docsgpt" uninstall --yes --purge
test ! -e "$stack"
- name: Stack logs
if: failure()
env:
DOCSGPT_IMAGE_TAG: verify
DOCSGPT_IMAGE_VARIANT: ${{ matrix.variant }}
run: docker compose -f deployment/docker-compose-standalone.yaml logs --no-color
- name: Stop the stack
if: always()
env:
DOCSGPT_IMAGE_TAG: verify
DOCSGPT_IMAGE_VARIANT: ${{ matrix.variant }}
run: docker compose -f deployment/docker-compose-standalone.yaml down -v
+47
View File
@@ -0,0 +1,47 @@
name: Lint the installers
# deployment/install.sh runs as `curl | bash` on macOS (bash 3.2) and Linux, and
# deployment/install.ps1 as `irm | iex` on Windows; a syntax error in either
# breaks every install at once. The end-to-end run through install.sh is in
# docker-image-verify.yml.
on:
pull_request:
paths:
- 'deployment/install.sh'
- 'deployment/install.ps1'
- '.github/workflows/installer-lint.yml'
push:
branches: [main]
paths:
- 'deployment/install.sh'
- 'deployment/install.ps1'
- '.github/workflows/installer-lint.yml'
permissions:
contents: read
jobs:
lint:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4.4.0
with:
persist-credentials: false
- name: shellcheck install.sh
run: |
bash -n deployment/install.sh
shellcheck --shell=bash deployment/install.sh
- name: Parse install.ps1
shell: pwsh
run: |
$tokens = $null
$errors = $null
[System.Management.Automation.Language.Parser]::ParseFile("$PWD/deployment/install.ps1", [ref]$tokens, [ref]$errors) | Out-Null
if ($errors.Count) {
$errors | ForEach-Object { Write-Host "install.ps1:$($_.Extent.StartLineNumber): $($_.Message)" }
exit 1
}
Write-Host 'install.ps1 parses'
+2
View File
@@ -65,6 +65,8 @@ jobs:
names = set(zipfile.ZipFile(glob.glob("dist/*.whl")[0]).namelist())
for required in (
"docsgpt/cli.py",
"docsgpt/deploy/commands.py",
"docsgpt/deploy/docker-compose.yaml",
"docsgpt/alembic.ini",
"docsgpt/alembic/env.py",
"docsgpt/alembic/script.py.mako",
+5
View File
@@ -195,6 +195,11 @@ docsgpt/static/
node_modules/
.vscode/settings.json
.vscode/sftp.json
# Zed: the shared project config is tracked, anything else under .zed/ is local
.zed/*
!.zed/settings.json
!.zed/tasks.json
!.zed/debug.json
/models/
model/
+24 -27
View File
@@ -2,39 +2,36 @@
"version": "0.2.0",
"configurations": [
{
"name": "Frontend Debug (npm)",
"name": "Frontend (npm)",
"type": "node-terminal",
"request": "launch",
"command": "npm run dev",
"cwd": "${workspaceFolder}/frontend"
},
{
"name": "Flask Debugger",
"type": "debugpy",
"request": "launch",
"module": "flask",
"env": {
"FLASK_APP": "docsgpt/app.py",
"PYTHONPATH": "${workspaceFolder}",
"FLASK_ENV": "development",
"FLASK_DEBUG": "1",
"FLASK_RUN_PORT": "7091",
"FLASK_RUN_HOST": "0.0.0.0"
},
"args": [
"run",
"--no-debugger"
],
"cwd": "${workspaceFolder}",
"name": "API (uvicorn)",
"type": "debugpy",
"request": "launch",
"module": "uvicorn",
"env": {
"PYTHONPATH": "${workspaceFolder}"
},
"args": [
"docsgpt.asgi:asgi_app",
"--host",
"127.0.0.1",
"--port",
"7091"
],
"cwd": "${workspaceFolder}"
},
{
"name": "Celery Debugger",
"name": "Celery worker",
"type": "debugpy",
"request": "launch",
"module": "celery",
"env": {
"PYTHONPATH": "${workspaceFolder}",
"PYTHONPATH": "${workspaceFolder}"
},
"args": [
"-A",
@@ -47,10 +44,10 @@
"cwd": "${workspaceFolder}"
},
{
"name": "Dev Containers (Mongo + Redis)",
"name": "Dev services (Postgres + Redis)",
"type": "node-terminal",
"request": "launch",
"command": "docker compose -f deployment/docker-compose-dev.yaml up --build",
"command": "docker compose -f deployment/docker-compose-dev.yaml up",
"cwd": "${workspaceFolder}"
}
],
@@ -58,9 +55,9 @@
{
"name": "DocsGPT: Full Stack",
"configurations": [
"Frontend Debug (npm)",
"Flask Debugger",
"Celery Debugger"
"Frontend (npm)",
"API (uvicorn)",
"Celery worker"
],
"presentation": {
"group": "DocsGPT",
@@ -68,4 +65,4 @@
}
}
]
}
}
+44
View File
@@ -0,0 +1,44 @@
// Debug configurations for the Zed editor (https://zed.dev/docs/debugger),
// the counterparts of .vscode/launch.json. Start one with `debugger: start`.
// Zed picks the interpreter from the checkout's `.venv`.
[
{
"label": "API (uvicorn)",
"adapter": "Debugpy",
"request": "launch",
"module": "uvicorn",
"args": ["docsgpt.asgi:asgi_app", "--host", "127.0.0.1", "--port", "7091"],
"cwd": "$ZED_WORKTREE_ROOT",
"env": { "PYTHONPATH": "$ZED_WORKTREE_ROOT" },
"justMyCode": true
},
{
// The solo pool keeps tasks in the debugged process, so breakpoints hit.
"label": "Celery worker (solo pool)",
"adapter": "Debugpy",
"request": "launch",
"module": "celery",
"args": ["-A", "docsgpt.app.celery", "worker", "-l", "INFO", "--pool=solo"],
"cwd": "$ZED_WORKTREE_ROOT",
"env": { "PYTHONPATH": "$ZED_WORKTREE_ROOT" },
"justMyCode": true
},
{
"label": "pytest: this file",
"adapter": "Debugpy",
"request": "launch",
"module": "pytest",
"args": ["--no-cov", "$ZED_RELATIVE_FILE"],
"cwd": "$ZED_WORKTREE_ROOT",
"env": { "PYTHONPATH": "$ZED_WORKTREE_ROOT" },
"justMyCode": false
},
{
"label": "Python: this file",
"adapter": "Debugpy",
"request": "launch",
"program": "$ZED_FILE",
"cwd": "$ZED_WORKTREE_ROOT",
"env": { "PYTHONPATH": "$ZED_WORKTREE_ROOT" }
}
]
+89
View File
@@ -0,0 +1,89 @@
// Project settings for the Zed editor (https://zed.dev/docs/configuring-zed).
// They mirror what CI and the pre-commit hook enforce; whitespace rules live in
// .editorconfig and the Python analysis config in pyproject.toml
// ([tool.pyright]) so other editors share them. Personal preferences belong in
// your user settings, not here.
{
// Zed replaces its defaults when this key is set, so they are repeated first.
// Build outputs, caches and local runtime data only add noise to the file
// finder and project search.
"file_scan_exclusions": [
"**/.git",
"**/.svn",
"**/.hg",
"**/.jj",
"**/.sl",
"**/.repo",
"**/CVS",
"**/.DS_Store",
"**/Thumbs.db",
"**/.classpath",
"**/.settings",
"**/__pycache__",
"**/.ruff_cache",
"**/.pytest_cache",
"**/.mypy_cache",
"**/htmlcov",
"**/.next",
"frontend/dist",
"docsgpt/static",
"**/indexes",
"**/inputs",
"**/vectors",
"models"
],
// Never shared with collaborators or sent to an AI assistant. The first six
// are Zed's defaults, which this key also replaces.
"private_files": [
"**/.env*",
"**/*.pem",
"**/*.key",
"**/*.cert",
"**/*.crt",
"**/secrets.yml",
"**/.jwt_secret_key"
],
"file_types": {
"Shell Script": [".env-template"],
"Dockerfile": ["Dockerfile*"]
},
"languages": {
"Python": {
"language_servers": ["basedpyright", "ruff", "..."],
"formatter": { "language_server": { "name": "ruff" } },
// CI runs `ruff check` only and most of the tree is not `ruff format`
// clean, so formatting on save would bury a change in unrelated diffs.
"format_on_save": "off",
"preferred_line_length": 120,
"wrap_guides": [120]
},
// Same order as the lint-staged hook: ESLint fixes, then Prettier.
"TypeScript": {
"formatter": "prettier",
"format_on_save": "on",
"code_actions_on_format": { "source.fixAll.eslint": true }
},
"TSX": {
"formatter": "prettier",
"format_on_save": "on",
"code_actions_on_format": { "source.fixAll.eslint": true }
},
"JavaScript": {
"formatter": "prettier",
"format_on_save": "on",
"code_actions_on_format": { "source.fixAll.eslint": true }
},
// Docs prose is reviewed by Vale, not reflowed by a formatter.
"Markdown": { "format_on_save": "off" },
"MDX": { "format_on_save": "off" }
},
"lsp": {
// The ESLint config is frontend/eslint.config.js, not at the root.
"eslint": {
"settings": { "workingDirectory": { "mode": "auto" } }
},
"tailwindcss-language-server": {
"settings": { "classFunctions": ["cn", "cva", "clsx", "twMerge"] }
}
}
}
+100
View File
@@ -0,0 +1,100 @@
// Project tasks for the Zed editor (https://zed.dev/docs/tasks). Run one with
// `task: spawn`. Python commands go through `uv run --no-sync`, which uses the
// checkout's `.venv` without changing what is installed in it; create it first
// with `uv sync`. See AGENTS.md for what each command does.
[
{
"label": "services: Postgres + Redis",
"command": "docker compose -f deployment/docker-compose-dev.yaml up",
"cwd": "$ZED_WORKTREE_ROOT",
"use_new_terminal": true
},
{
"label": "dev: API + worker + frontend",
"command": "uv run --no-sync docsgpt dev --ui",
"cwd": "$ZED_WORKTREE_ROOT",
"use_new_terminal": true
},
{
"label": "dev: API + worker + frontend (mock LLM, no API key)",
"command": "uv run --no-sync docsgpt dev --ui --mock-llm",
"cwd": "$ZED_WORKTREE_ROOT",
"use_new_terminal": true
},
{
"label": "backend: API",
"command": "uv run --no-sync docsgpt api --reload",
"cwd": "$ZED_WORKTREE_ROOT",
"use_new_terminal": true
},
{
"label": "backend: Celery worker",
"command": "uv run --no-sync docsgpt worker",
"cwd": "$ZED_WORKTREE_ROOT",
"use_new_terminal": true
},
{
"label": "backend: run migrations",
"command": "uv run --no-sync docsgpt migrate",
"cwd": "$ZED_WORKTREE_ROOT"
},
{
"label": "frontend: dev server",
"command": "npm run dev",
"cwd": "$ZED_WORKTREE_ROOT/frontend",
"use_new_terminal": true
},
{
"label": "frontend: build into docsgpt/static",
"command": "bash scripts/build_frontend.sh",
"cwd": "$ZED_WORKTREE_ROOT"
},
// Coverage is switched off for partial runs: pytest.ini turns it on, and a
// report for one file is slow and misleading.
{
"label": "pytest: all",
"command": "uv run --no-sync python -m pytest",
"cwd": "$ZED_WORKTREE_ROOT"
},
{
"label": "pytest: this file",
"command": "uv run --no-sync python -m pytest --no-cov \"$ZED_RELATIVE_FILE\"",
"cwd": "$ZED_WORKTREE_ROOT"
},
{
"label": "pytest: test under cursor ($ZED_SYMBOL)",
"command": "uv run --no-sync python -m pytest --no-cov \"$ZED_RELATIVE_FILE\" -k \"$ZED_SYMBOL\"",
"cwd": "$ZED_WORKTREE_ROOT"
},
{
"label": "pytest: last failed",
"command": "uv run --no-sync python -m pytest --no-cov --lf",
"cwd": "$ZED_WORKTREE_ROOT"
},
{
"label": "vitest: all",
"command": "npm run test",
"cwd": "$ZED_WORKTREE_ROOT/frontend"
},
{
"label": "vitest: this file",
"command": "npx vitest run \"$ZED_FILE\"",
"cwd": "$ZED_WORKTREE_ROOT/frontend"
},
{
"label": "lint: ruff check --fix",
"command": "uv run --no-sync ruff check --fix .",
"cwd": "$ZED_WORKTREE_ROOT"
},
{
"label": "lint: frontend (eslint --fix + prettier)",
"command": "npm run lint-fix && npm run format",
"cwd": "$ZED_WORKTREE_ROOT/frontend"
},
// CI fails when docsgpt/requirements*.txt are stale against uv.lock.
{
"label": "deps: uv lock + export requirements",
"command": "uv lock && bash scripts/export_requirements.sh",
"cwd": "$ZED_WORKTREE_ROOT"
}
]
+1 -1
View File
@@ -198,7 +198,7 @@ vale .
- Parsers live in `docsgpt/parser/` and handle different document formats in the ingestion stage.
- Agents and tools are in `docsgpt/agents/` and `docsgpt/agents/tools/`.
- Celery setup/config lives in `docsgpt/celery_init.py` and `docsgpt/celeryconfig.py`.
- Settings and env vars are managed via Pydantic in `docsgpt/core/settings.py`.
- Settings and env vars are managed via Pydantic in `docsgpt/core/settings/` (one module per domain, composed into `Settings`). Every field needs a `description`; regenerate the docs reference with `python -m docsgpt.core.settings.reference --write`.
### Frontend
+27 -1
View File
@@ -43,7 +43,7 @@ Tech Stack Overview:
### 🌐 Frontend Contributions (⚛️ React, Vite)
* The updated Figma design can be found [here](https://www.figma.com/file/OXLtrl1EAy885to6S69554/DocsGPT?node-id=0%3A1&t=hjWVuxRg9yi5YkJ9-1). Please try to follow the guidelines.
* **Coding Style:** We follow a strict coding style enforced by ESLint and Prettier. Please ensure your code adheres to the configuration provided in our repository's `fronetend/.eslintrc.js` file. We recommend configuring your editor with ESLint and Prettier to help with this.
* **Coding Style:** We follow a strict coding style enforced by ESLint and Prettier. Please ensure your code adheres to the configuration provided in our repository's `frontend/eslint.config.js` and `frontend/prettier.config.cjs` files. We recommend configuring your editor with ESLint and Prettier to help with this.
* **Component Structure:** Strive for small, reusable components. Favor functional components and hooks over class components where possible.
* **State Management** If you need to add stores, please use Redux.
@@ -75,6 +75,32 @@ Tech Stack Overview:
...
```
### Editor setup
Some configuration is shared by every editor, so you rarely need to set anything up by hand:
- [`.editorconfig`](https://editorconfig.org) holds the whitespace rules (4 spaces for Python, 2 for TypeScript/JSON/YAML, LF line endings, final newline). Most editors read it natively or through a plugin.
- `[tool.pyright]` in `pyproject.toml` points Pyright, basedpyright and Pylance at the `.venv` created by `uv sync` and at the repository root for imports.
- `.ruff.toml`, `frontend/eslint.config.js` and `frontend/prettier.config.cjs` are picked up by the matching editor integrations.
Editor-specific configuration that is tracked:
- **VS Code:** `.vscode/launch.json` has debug targets for the API, the Celery worker and the frontend.
- **Zed:** open the repository root (not `frontend/`). `.zed/settings.json` configures the language servers and formatters, `.zed/tasks.json` adds tasks (`task: spawn`) for the dev services, the API, the worker, the frontend, tests and linting, and `.zed/debug.json` adds debug targets (`debugger: start`). Python files are not formatted on save because most of the tree is not `ruff format` clean; frontend files are, with ESLint fixes followed by Prettier, as in the pre-commit hook. Project settings cannot install extensions, so if you want the matching syntax support add this to your own Zed settings:
```json
{
"auto_install_extensions": {
"dockerfile": true,
"docker-compose": true,
"toml": true,
"mdx": true
}
}
```
Personal preferences belong in your user settings; `.vscode/settings.json` and any other file under `.zed/` are ignored by git.
### Testing
To run unit tests from the root of the repository, execute:
+26 -2
View File
@@ -80,9 +80,33 @@ Calling all developers and GenAI innovators! The **DocsGPT Lighthouse Program**
## QuickStart
> [!Note]
> Make sure you have [Docker](https://docs.docker.com/engine/install/) installed
> DocsGPT runs on [Docker](https://docs.docker.com/engine/install/). The installer checks for it first.
A more detailed [Quickstart](https://docs.docsgpt.cloud/quickstart) is available in our documentation
**macOS and Linux:**
```bash
curl -fsSL https://docs.ac/install | bash
```
**Windows (PowerShell):**
```powershell
irm https://docs.ac/install.ps1 | iex
```
The installer gets [uv](https://docs.astral.sh/uv/), installs the `docsgpt` Python package with it, and runs `docsgpt up`. That asks who should reach DocsGPT (only this computer, your network, or a domain with HTTPS) and which model provider to use, then starts it, at http://localhost:7091 for a local install. Afterwards, `docsgpt status`, `docsgpt logs`, `docsgpt upgrade`, `docsgpt down` and `docsgpt uninstall` manage it.
To read the script before running it:
```bash
curl -fsSL https://docs.ac/install -o install.sh
less install.sh
bash install.sh
```
A more detailed [Quickstart](https://docs.docsgpt.cloud/quickstart) is available in our documentation.
### From a clone, with the setup script
1. **Clone the repository:**
+2 -2
View File
@@ -2,8 +2,8 @@
# DOCSGPT_IMAGE_TAG develop (default, follows main) or a release, e.g. 0.20.0
# DOCSGPT_IMAGE_VARIANT empty (default, slim) or -docling: docling parser engine,
# its models, and tesseract baked in (OCR-ready)
# Set them in ../.env or the shell. deployment/docker-compose-standalone.yaml is
# the same stack without a git checkout.
# Set them in ../.env or the shell. deployment/docker-compose-standalone.yaml runs
# the same images without a git checkout, with the backend serving the UI on one port.
name: docsgpt-oss
services:
+61 -35
View File
@@ -3,52 +3,44 @@
# curl -fsSLO https://raw.githubusercontent.com/arc53/DocsGPT/main/deployment/docker-compose-standalone.yaml
# printf 'LLM_PROVIDER=docsgpt\nVITE_API_STREAMING=true\nINTERNAL_KEY=%s\n' "$(openssl rand -hex 16)" > .env
# docker compose -f docker-compose-standalone.yaml up -d
# open http://localhost:7091
#
# INTERNAL_KEY is the shared secret the worker uses to hand finished indexes to
# the API; without it every ingest fails with a 401 (setup.sh generates one).
# open http://localhost:5173
#
# Every release also attaches this file as an asset. Settings come from .env
# next to this file (any DocsGPT setting; the compose-internal service URLs
# below take precedence). Data lives in named volumes, so `docker compose
# down` keeps it and `docker compose down -v` removes it.
# The backend image serves the web UI and the API on one port. Every release
# also attaches this file as an asset, and `docsgpt up` runs it from the Python
# package. Settings come from .env next to this file (any DocsGPT setting,
# VITE_* included; the compose-internal service URLs below take precedence).
# Data lives in named volumes, so `docker compose down` keeps it and
# `docker compose down -v` removes it.
#
# DOCSGPT_IMAGE_TAG release to run, e.g. 0.20.0 (default: latest release);
# develop follows the main branch
# DOCSGPT_IMAGE_VARIANT empty (slim, default) or -docling: docling parser
# engine, its models, and tesseract baked in (OCR-ready)
# DOCSGPT_BIND interface the port is published on: 127.0.0.1 (default,
# this machine only) or 0.0.0.0 (every interface; set
# AUTH_TYPE, see the DocsGPT settings guide)
# DOCSGPT_PORT host port for the UI and API (default: 7091)
# POSTGRES_PASSWORD database password (default: docsgpt). Read when the
# postgres volume is first created; changing it later
# does not change the existing database's password.
# Use URL-safe characters (e.g. openssl rand -hex 24).
# DOCSGPT_DOMAIN public domain for the `https` profile (below)
# EMBEDDINGS_NAME defaults to granite here (this stack always starts on
# fresh volumes, so there is no older index to keep
# compatible); the code default stays mpnet for upgrades.
#
# HTTPS for a public domain: point the domain's DNS at this machine, open ports
# 80 and 443, then
# DOCSGPT_DOMAIN=docs.example.com docker compose -f docker-compose-standalone.yaml --profile https up -d
# Caddy obtains and renews the certificate and proxies to the backend. Putting
# DOCSGPT_DOMAIN and COMPOSE_PROFILES=https in .env instead makes every later
# `up`, `down` and `logs` include Caddy without the flag.
name: docsgpt
services:
frontend:
image: arc53/docsgpt-fe:${DOCSGPT_IMAGE_TAG:-latest}
env_file:
- path: .env
required: false
environment:
# Every VITE_* the app reads. A bare name is passed through only when it is
# set in the shell or the --env-file, so an unset one does not reach the
# container as an empty string and override the image's own default.
- VITE_API_HOST=${VITE_API_HOST:-http://localhost:7091}
- VITE_API_STREAMING=${VITE_API_STREAMING:-true}
- VITE_BASE_URL
- VITE_GOOGLE_CLIENT_ID
- VITE_GOOGLE_PICKER_API_KEY
- VITE_SHARE_POINT_CLIENT_ID
- VITE_CONFLUENCE_CLIENT_ID
- VITE_NOTIFICATION_TEXT
- VITE_NOTIFICATION_LINK
- VITE_ENABLE_VOICE_INPUT
- VITE_DISABLE_SOURCE_FE
- VITE_USE_V
ports:
- "5173:5173"
depends_on:
- backend
backend:
image: arc53/docsgpt:${DOCSGPT_IMAGE_TAG:-latest}${DOCSGPT_IMAGE_VARIANT:-}
# Same as docker-compose-hub.yaml: the data volumes are written by root so
@@ -62,10 +54,14 @@ services:
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
- POSTGRES_URI=postgresql://docsgpt:${POSTGRES_PASSWORD:-docsgpt}@postgres:5432/docsgpt
- EMBEDDINGS_NAME=${EMBEDDINGS_NAME:-ibm-granite/granite-embedding-311m-multilingual-r2}
# A model server on the host (Ollama, vLLM, ...) is reachable as
# host.docker.internal on Linux too, as it is on Docker Desktop.
extra_hosts:
- "host.docker.internal:host-gateway"
ports:
- "7091:7091"
- "${DOCSGPT_BIND:-127.0.0.1}:${DOCSGPT_PORT:-7091}:7091"
volumes:
- indexes:/app/indexes
- inputs:/app/inputs
@@ -90,9 +86,11 @@ services:
- CELERY_BROKER_URL=redis://redis:6379/0
- CELERY_RESULT_BACKEND=redis://redis:6379/1
- CACHE_REDIS_URL=redis://redis:6379/2
- POSTGRES_URI=postgresql://docsgpt:docsgpt@postgres:5432/docsgpt
- POSTGRES_URI=postgresql://docsgpt:${POSTGRES_PASSWORD:-docsgpt}@postgres:5432/docsgpt
- API_URL=http://backend:7091
- EMBEDDINGS_NAME=${EMBEDDINGS_NAME:-ibm-granite/granite-embedding-311m-multilingual-r2}
extra_hosts:
- "host.docker.internal:host-gateway"
volumes:
- indexes:/app/indexes
- inputs:/app/inputs
@@ -112,7 +110,7 @@ services:
image: postgres:16-alpine
environment:
- POSTGRES_USER=docsgpt
- POSTGRES_PASSWORD=docsgpt
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-docsgpt}
- POSTGRES_DB=docsgpt
volumes:
- postgres_data:/var/lib/postgresql/data
@@ -123,8 +121,36 @@ services:
retries: 10
restart: unless-stopped
caddy:
image: caddy:2-alpine
profiles: [https]
environment:
- DOCSGPT_DOMAIN=${DOCSGPT_DOMAIN:-}
# The domain is checked here rather than with ${DOCSGPT_DOMAIN:?}: compose
# interpolates every service, so a required variable would break the stack
# for everyone who does not use this profile.
entrypoint: ["/bin/sh", "-c"]
command:
- >-
if [ -z "$$DOCSGPT_DOMAIN" ]; then
echo "caddy: set DOCSGPT_DOMAIN to the public domain" >&2; exit 1;
fi;
exec caddy reverse-proxy --from "$$DOCSGPT_DOMAIN" --to backend:7091
ports:
- "80:80"
- "443:443"
- "443:443/udp"
volumes:
- caddy_data:/data
- caddy_config:/config
depends_on:
- backend
restart: unless-stopped
volumes:
indexes:
inputs:
vectors:
postgres_data:
caddy_data:
caddy_config:
+113
View File
@@ -0,0 +1,113 @@
# DocsGPT installer for Windows.
#
# irm https://docs.ac/install.ps1 | iex
#
# Installs uv when it is missing or too old, installs the docsgpt Python
# package with it, then runs `docsgpt up`, which sets up DocsGPT on Docker
# Desktop and starts it. Running it again upgrades the package and keeps your
# settings. To pass options to `docsgpt up`:
#
# & ([scriptblock]::Create((irm https://docs.ac/install.ps1))) --domain docs.example.com --yes
#
# Environment:
# DOCSGPT_VERSION package version to install (default: the latest release)
# DOCSGPT_PACKAGE install this instead of docsgpt from PyPI (a wheel path or URL)
# DOCSGPT_NO_MODIFY_PATH set to 1 to leave PATH alone
#
# Everything runs inside a function, so a download cut short runs nothing, and
# nothing calls `exit`, which would close the window `iex` runs in.
function Install-DocsGPT {
param([string[]]$UpArguments)
$ErrorActionPreference = 'Stop'
$UvVersion = '0.12.15'
$UvMinVersion = [version]'0.8.0'
# sha256 of https://astral.sh/uv/$UvVersion/install.ps1, checked before it runs. Bump it with
# $UvVersion: (Invoke-WebRequest "https://astral.sh/uv/<version>/install.ps1").Content | …
$UvInstallerSha256 = '63f2d7e2ccc347cc018127b22f56f569080e5730a390dcb702b654daf6822193'
function Say([string]$Message) { Write-Host "==> $Message" }
if (-not (Get-Command docker -ErrorAction SilentlyContinue)) {
throw 'DocsGPT runs on Docker. Install Docker Desktop (https://docs.docker.com/desktop/setup/install/windows-install/), start it, and run this again.'
}
# uv installs and upgrades the package, and brings Python 3.12 when the system has none.
$uv = $null
$candidates = @(
(Get-Command uv -ErrorAction SilentlyContinue | Select-Object -ExpandProperty Source -First 1),
(Join-Path $HOME '.local\bin\uv.exe'),
(Join-Path $HOME '.cargo\bin\uv.exe')
) | Where-Object { $_ -and (Test-Path $_) }
foreach ($candidate in $candidates) {
$found = "$(& $candidate --version 2>$null)" -replace '^uv\s+([0-9.]+).*$', '$1'
if ($found -match '^\d+\.\d+(\.\d+)?$' -and [version]$found -ge $UvMinVersion) {
$uv = $candidate
break
}
}
if (-not $uv) {
$uvDir = Join-Path $HOME '.local\bin'
Say "Installing uv $UvVersion into $uvDir"
$env:UV_INSTALL_DIR = $uvDir
$env:UV_NO_MODIFY_PATH = '1'
$env:UV_PRINT_QUIET = '1'
# Downloaded and checked before it runs, and run in a child PowerShell so nothing the
# uv installer does can end this session.
$installer = Join-Path ([System.IO.Path]::GetTempPath()) "uv-installer-$UvVersion.ps1"
try {
Invoke-WebRequest -Uri "https://astral.sh/uv/$UvVersion/install.ps1" -OutFile $installer -UseBasicParsing
$actual = (Get-FileHash -Path $installer -Algorithm SHA256).Hash.ToLower()
if ($actual -ne $UvInstallerSha256) {
throw "The uv installer does not match its pinned sha256: got $actual, expected $UvInstallerSha256. Refusing to run it."
}
$shell = (Get-Process -Id $PID).Path
& $shell -NoProfile -ExecutionPolicy Bypass -File $installer
if ($LASTEXITCODE -ne 0) {
# Without this, a uv.exe left behind by an older install would be accepted below.
throw "The uv installer exited with code $LASTEXITCODE."
}
} finally {
Remove-Item $installer -Force -ErrorAction SilentlyContinue
}
$uv = Join-Path $uvDir 'uv.exe'
if (-not (Test-Path $uv)) { throw "uv did not install into $uvDir" }
}
if ($env:DOCSGPT_VERSION -and $env:DOCSGPT_PACKAGE) {
throw 'Set DOCSGPT_VERSION or DOCSGPT_PACKAGE, not both.'
}
if ($env:DOCSGPT_PACKAGE) {
Say 'Installing docsgpt from DOCSGPT_PACKAGE'
& $uv tool install --reinstall --python 3.12 $env:DOCSGPT_PACKAGE
} elseif ($env:DOCSGPT_VERSION) {
Say "Installing docsgpt $env:DOCSGPT_VERSION"
& $uv tool install --force --python 3.12 "docsgpt==$env:DOCSGPT_VERSION"
} else {
Say 'Installing the latest docsgpt'
& $uv tool install --upgrade --python 3.12 docsgpt
}
if ($LASTEXITCODE -ne 0) { throw 'Installing the docsgpt package failed.' }
$binDir = "$(& $uv tool dir --bin)".Trim()
$docsgpt = Join-Path $binDir 'docsgpt.exe'
if (-not (Test-Path $docsgpt)) { throw "The docsgpt command is missing from $binDir." }
if (($env:Path -split ';') -notcontains $binDir) {
if ($env:DOCSGPT_NO_MODIFY_PATH -eq '1') {
Say "Add $binDir to PATH to run docsgpt from a new terminal"
} else {
& $uv tool update-shell *> $null
Say "Added $binDir to PATH for new terminals"
}
$env:Path = "$binDir;$env:Path"
}
& $docsgpt up @UpArguments
if ($LASTEXITCODE -ne 0) {
# throw, not Write-Error: the caller (and any automation) must see this fail.
throw "docsgpt up exited with code $LASTEXITCODE. Fix the problem above and run it again: docsgpt up"
}
}
Install-DocsGPT -UpArguments $args
+217
View File
@@ -0,0 +1,217 @@
#!/usr/bin/env bash
# DocsGPT installer for macOS and Linux.
#
# curl -fsSL https://docs.ac/install | bash
#
# Installs uv when it is missing or too old, installs the `docsgpt` Python
# package with it, then runs `docsgpt up`, which sets up DocsGPT on Docker and
# starts it. Running it again upgrades the package and keeps your settings.
# Arguments go to `docsgpt up` (see `docsgpt up --help`):
#
# curl -fsSL https://docs.ac/install | bash -s -- --domain docs.example.com --yes
#
# Environment:
# DOCSGPT_VERSION package version to install (default: the latest release)
# DOCSGPT_PACKAGE install this instead of docsgpt from PyPI (a wheel path or URL)
# DOCSGPT_NO_MODIFY_PATH set to 1 to leave shell profiles alone
# DOCSGPT_INSTALL_DOCKER set to 1 to install Docker on Linux without asking
#
# Everything runs inside main(), so a download cut short runs nothing.
UV_VERSION="0.12.15"
UV_MIN_VERSION="0.8.0"
# sha256 of https://astral.sh/uv/$UV_VERSION/install.sh, checked before it runs. Bump it with
# UV_VERSION: curl -fsSL https://astral.sh/uv/<version>/install.sh | shasum -a 256
UV_INSTALLER_SHA256="716a1d6844740756c68770fcec2f79c2013fb9b03869a113f61e15f6f482a6a1"
main() {
set -euo pipefail
local bold="" red="" reset=""
if [ -t 2 ]; then
bold=$'\033[1m' red=$'\033[31m' reset=$'\033[0m'
fi
say() { printf '%s==>%s %s\n' "$bold" "$reset" "$*" >&2; }
die() { printf '%serror:%s %s\n' "$red" "$reset" "$*" >&2; exit 1; }
has() { command -v "$1" >/dev/null 2>&1; }
have_tty() { (exec </dev/tty) 2>/dev/null; }
ask_yes() {
local answer
printf '%s [y/N] ' "$1" >/dev/tty
read -r answer </dev/tty || return 1
case "$answer" in y | Y | yes | YES) return 0 ;; *) return 1 ;; esac
}
download() {
if has curl; then
curl -fsSL --retry 3 "$1"
elif has wget; then
wget -qO- "$1"
else
die "curl or wget is needed to download $1"
fi
}
# POSIX single-quote encoding: what needs quoting is decided here, not by the shell that runs it.
shell_quote() {
local arg out=""
for arg in "$@"; do
out="$out'$(printf '%s' "$arg" | sed "s/'/'\\\\''/g")' "
done
printf '%s' "$out"
}
sha256_of() {
if has shasum; then
shasum -a 256 "$1" | awk '{print $1}'
elif has sha256sum; then
sha256sum "$1" | awk '{print $1}'
else
die "neither shasum nor sha256sum is available to check $1"
fi
}
# run_downloaded URL SHA256 COMMAND...: save URL to a file, check it against SHA256 ("-" to
# skip), then run COMMAND with the file as its last argument. A transfer cut short, or content
# that does not match, fails before anything runs.
run_downloaded() {
local url="$1" expected="$2" script status=0 actual
shift 2
script="$(mktemp)"
if ! download "$url" >"$script"; then
rm -f "$script"
die "could not download $url"
fi
if [ "$expected" != "-" ]; then
actual="$(sha256_of "$script")"
if [ "$actual" != "$expected" ]; then
rm -f "$script"
die "$url does not match its pinned sha256: got $actual, expected $expected. Refusing to run it."
fi
fi
"$@" "$script" || status=$?
rm -f "$script"
return "$status"
}
# version_ge A B: A >= B for dotted version numbers.
version_ge() {
local -a left right
IFS=. read -r -a left <<<"$1"
IFS=. read -r -a right <<<"$2"
local i x y
for i in 0 1 2; do
x="${left[i]:-0}" y="${right[i]:-0}"
x="${x%%[!0-9]*}" y="${y%%[!0-9]*}"
if (( 10#${x:-0} > 10#${y:-0} )); then return 0; fi
if (( 10#${x:-0} < 10#${y:-0} )); then return 1; fi
done
return 0
}
local os
os="$(uname -s)"
case "$os" in
Linux | Darwin) ;;
*) die "this installer is for macOS and Linux. On Windows, in PowerShell: irm https://docs.ac/install.ps1 | iex" ;;
esac
# Docker first: without it nothing below is useful.
local docker_group_pending=0
if ! has docker; then
if [ "$os" = Darwin ]; then
die "DocsGPT runs on Docker. Install Docker Desktop (https://docs.docker.com/desktop/setup/install/mac-install/) or OrbStack (https://orbstack.dev), start it, and run this again."
fi
if [ "${DOCSGPT_INSTALL_DOCKER:-}" = 1 ] || { have_tty && ask_yes "Docker is not installed. Install it now with Docker's script from get.docker.com?"; }; then
local sudo=""
if [ "$(id -u)" -ne 0 ]; then
has sudo || die "installing Docker needs root. Install it (https://docs.docker.com/engine/install/) and run this again."
sudo="sudo"
fi
say "Installing Docker"
# Docker's script changes over time and publishes no checksum, so it is only downloaded
# in full before it runs.
run_downloaded https://get.docker.com - $sudo sh
$sudo systemctl enable --now docker >/dev/null 2>&1 || true
if [ -n "$sudo" ]; then
$sudo usermod -aG docker "$(id -un)"
docker_group_pending=1
fi
else
die "DocsGPT runs on Docker. Install it (https://docs.docker.com/engine/install/) and run this again."
fi
fi
# uv installs and upgrades the package, and brings Python 3.12 when the system has none.
local uv="" candidate found
for candidate in "$(command -v uv 2>/dev/null || true)" "$HOME/.local/bin/uv" "$HOME/.cargo/bin/uv"; do
if [ -z "$candidate" ] || [ ! -x "$candidate" ]; then
continue
fi
found="$("$candidate" --version 2>/dev/null | awk '{print $2}')" || continue
if [ -n "$found" ] && version_ge "$found" "$UV_MIN_VERSION"; then
uv="$candidate"
break
fi
done
if [ -z "$uv" ]; then
local uv_dir="${XDG_BIN_HOME:-$HOME/.local/bin}"
say "Installing uv $UV_VERSION into $uv_dir"
run_downloaded "https://astral.sh/uv/$UV_VERSION/install.sh" "$UV_INSTALLER_SHA256" \
env UV_INSTALL_DIR="$uv_dir" UV_NO_MODIFY_PATH=1 UV_PRINT_QUIET=1 sh
uv="$uv_dir/uv"
[ -x "$uv" ] || die "uv did not install into $uv_dir"
fi
if [ -n "${DOCSGPT_VERSION:-}" ] && [ -n "${DOCSGPT_PACKAGE:-}" ]; then
die "set DOCSGPT_VERSION or DOCSGPT_PACKAGE, not both"
fi
if [ -n "${DOCSGPT_PACKAGE:-}" ]; then
say "Installing docsgpt from DOCSGPT_PACKAGE"
"$uv" tool install --reinstall --python 3.12 "$DOCSGPT_PACKAGE"
elif [ -n "${DOCSGPT_VERSION:-}" ]; then
say "Installing docsgpt $DOCSGPT_VERSION"
"$uv" tool install --force --python 3.12 "docsgpt==$DOCSGPT_VERSION"
else
say "Installing the latest docsgpt"
"$uv" tool install --upgrade --python 3.12 docsgpt
fi
local bin_dir docsgpt
bin_dir="$("$uv" tool dir --bin)"
docsgpt="$bin_dir/docsgpt"
[ -x "$docsgpt" ] || die "the docsgpt command is missing from $bin_dir"
case ":$PATH:" in
*":$bin_dir:"*) ;;
*)
if [ "${DOCSGPT_NO_MODIFY_PATH:-}" = 1 ]; then
say "Add $bin_dir to PATH to run docsgpt from a new terminal"
else
"$uv" tool update-shell >/dev/null 2>&1 || true
say "Added $bin_dir to PATH for new terminals"
fi
;;
esac
if [ "$docker_group_pending" = 1 ]; then
if has sg; then
# The docker group applies to new logins; sg gives it to this command now. sg runs the
# command with /bin/sh, which need not be bash, so quote for POSIX sh rather than with %q.
local command
command="$(shell_quote "$docsgpt" up "$@")"
if have_tty; then
exec sg docker -c "$command </dev/tty"
fi
exec sg docker -c "$command --yes"
fi
say "Docker is installed and your user joined the docker group. Log out and back in, then run: docsgpt up"
exit 0
fi
# Hand the terminal to docsgpt up: when this script is piped into bash, its
# standard input is the script, not the keyboard.
if [ -t 0 ]; then
exec "$docsgpt" up "$@"
fi
if have_tty; then
exec "$docsgpt" up "$@" </dev/tty
fi
exec "$docsgpt" up --yes "$@"
}
main "$@"
+4
View File
@@ -3,6 +3,10 @@ export default {
"title": "🤖 Agent Basics",
"href": "/Agents/basics"
},
"guardrails": {
"title": "🛡️ Guardrails",
"href": "/Agents/guardrails"
},
"api": {
"title": "🔌 Agent API",
"href": "/Agents/api"
+384
View File
@@ -0,0 +1,384 @@
---
title: Agent Guardrails
description: Scan user input, retrieved sources, tool results and answers with built-in checks — PII, secrets, banned terms, link policy, prompt injection, groundedness and an LLM judge — and choose to flag, redact or block. Includes the audit journal, instance floor and API reference.
---
import { Callout, Tabs } from 'nextra/components';
# Agent Guardrails 🛡️
Guardrails are per-agent content controls that run **inside** an agent turn, at the points where text changes hands: when the user's question arrives, when retrieved documents are about to reach the model, when a tool returns, and when the answer is on its way out. Each control pairs a **check** (a detector) with a **stage** (where it runs) and an **action** (what happens on a match).
They apply everywhere the agent runs — the chat UI, the [Agent API](/Agents/api), the [OpenAI-compatible API](/Agents/openai-compatible), [webhooks](/Agents/webhooks), scheduled runs and the embeddable widget — and every decision is written to an audit journal you can review from the agent's logs page.
<Callout type="info" emoji="ℹ️">
Guardrails are a defence-in-depth layer, not a replacement for a well-written prompt or for tool approvals. The pattern and heuristic checks are deterministic and fast; the LLM judge is semantic but costs a model call. Start in **monitor mode**, read the journal, then promote to enforcement.
</Callout>
## How a turn is scanned
An agent turn passes through four intervention points. A control attached to a stage sees the text at that stage and can leave it alone, mask parts of it, or stop the turn.
| Stage | What is scanned | `redact` does | `block` does |
| --- | --- | --- | --- |
| `input` | The user's question, before anything is sent to the model | The model and the stored conversation both receive the masked question | The turn ends immediately with the block message; nothing reaches the model |
| `retrieval` | The retrieved document chunks, formatted as they will appear in the prompt | Masked in the prompt **and** in the sources shown to the user and stored with the conversation | The model is told the sources were withheld by policy; the user sees `[Withheld by a content policy.]` in place of each source |
| `tool_result` | Each tool's result string, before it fans out to the model, the UI and the journal | The masked result is what the model and the user see | The result is replaced with a note telling the model it could not be used and must not speculate about its contents |
| `output` | The answer as it streams from the model | Masked **before** the text leaves the server | The stream stops with the block message; any tokens already delivered are retracted from the client and the stored message |
A few details worth knowing:
- **Input redaction is what gets stored.** If a PII control redacts an email address in the question, the conversation history holds the redacted version. The raw text never lands in the database.
- **Retrieval scanning covers custom prompts too.** If your prompt template interpolates documents itself (see [Customising prompts](/Guides/Customising-prompts)), the rendered documents are still scanned and the verdict is patched back into the prompt.
- **Structured output is scanned whole.** When an agent has a JSON schema, redacting mid-token would produce invalid JSON, so the complete document is buffered and scanned once.
- **Output blocks after streaming has begun are retractions.** Tokens on the wire cannot be recalled, so the server tells the client to clear the partial answer, replaces the persisted message with the block message, and clears any reasoning trace. On a reload the user sees only the block message.
### Streaming without leaks
Output controls run *before* a token is released. Deterministic checks hold a small lookback window (sized from the longest match any active check can produce, up to 8 KB for a PEM private key) and re-scan `held + new` on every chunk, so a card number or API key split across two stream deltas is still caught. Redacted spans are never cut in half: the release point is pulled back so a match is either fully masked or fully held.
Remote checks (the LLM judge) cannot afford a call per token, so the guard accumulates text to a sentence boundary (around 400 characters) and evaluates whole segments. A stream that never produces a sentence boundary is force-released past a 16 KB ceiling so it cannot stall forever.
The **groundedness** check only makes sense over a finished answer, so it is deferred to the end of the stream. A `block` from it is therefore always a retraction.
## Actions
| Action | Effect | Available for |
| --- | --- | --- |
| `flag` | Record the decision in the journal and the turn's activity log. Nothing about the answer changes. | Every check |
| `redact` | Replace each matched span with a mask and continue. | Checks that report spans: `pii`, `secrets`, `denylist`, `url` |
| `block` | Stop the turn and return the agent's block message. | Every check |
Within one stage the **most restrictive outcome wins**: if two controls match and one says block, the stage blocks. If several redact, all of their spans are masked, and overlapping spans are unioned so a short match can never leave part of a longer one in the clear.
Redaction masks are check-specific: PII uses the entity label (`[EMAIL]`, `[CREDIT_CARD]`), secrets use `[REDACTED]`, banned terms use `***`, and disallowed links use `<url redacted>`.
## Enforcement modes
| Mode | Behaviour |
| --- | --- |
| `monitor_only` (default) | Every control runs, but every action is downgraded to `flag`. Nothing is changed or blocked; the journal shows what *would* have happened. Streamed answers pass through untouched and are scanned once at the end. |
| `scan_all` | Actions are enforced as configured. |
Monitor mode is the supported rollout path. Turn on the checks you want, run real traffic for a few days, look at the **Guardrail activity** panel for the control that is over-triggering, tune its settings, then switch to `scan_all`.
## Built-in checks
| Key | Label | Stages | Redacts | Remote | Typical latency |
| --- | --- | --- | --- | --- | --- |
| `pii` | Personal information | input, retrieval, tool_result, output | Yes | No | ~2 ms |
| `secrets` | Credentials and secrets | input, retrieval, tool_result, output | Yes | No | ~2 ms |
| `denylist` | Banned terms | input, retrieval, tool_result, output | Yes | No | ~1 ms |
| `url` | Link policy | input, retrieval, tool_result, output | Yes | No | ~2 ms |
| `injection` | Prompt injection (heuristic) | input, retrieval, tool_result | No | No | ~3 ms |
| `groundedness` | Grounding in sources | output | No | No | ~5 ms |
| `policy` | Custom policy (LLM judge) | input, retrieval, tool_result, output | No | Yes | ~1 s |
The live catalog for your instance, including which checks the operator has allowed, is served by `GET /api/guardrails/catalog`.
### `pii` — Personal information
Pattern matching for structured identifiers. Reliable for the formats below; it does **not** find names or free-text addresses.
| Setting | Default | Notes |
| --- | --- | --- |
| `entities` | `["EMAIL", "PHONE", "US_SSN", "CREDIT_CARD"]` | Non-empty subset of `EMAIL`, `PHONE`, `US_SSN`, `CREDIT_CARD`, `IPV4`, `IBAN` |
Card numbers must be 13–19 digits and pass a Luhn check before they count, which keeps order numbers and long IDs from matching. Each match is reported under its entity name, so the journal tells you *which* kind of PII appeared.
### `secrets` — Credentials and secrets
No settings. Detects by known formats: AWS access keys, GitHub tokens, OpenAI and Anthropic keys, Slack tokens, Google API keys, JWTs, PEM private-key blocks (the whole armored block, not just the header) and generic `password=` / `api_key:` style assignments where only the value is masked.
### `denylist` — Banned terms
| Setting | Default | Notes |
| --- | --- | --- |
| `terms` | — (required) | 1–500 terms, each ≤ 128 characters |
| `match` | `"word"` | `"word"` matches whole words only; `"substring"` matches anywhere |
| `case_sensitive` | `false` | |
Useful for competitor names, internal codenames, or phrases you never want an agent to repeat.
### `url` — Link policy
| Setting | Default | Notes |
| --- | --- | --- |
| `allow_hosts` | `[]` | Up to 200 hosts. When non-empty, any link whose host is not in the list (or a subdomain of one) is disallowed |
| `block_hosts` | `[]` | Up to 200 hosts. Links to these hosts (or their subdomains) are always disallowed |
At least one of the two lists is required. Hosts are matched against the parsed URL authority, so `https://allowed.com@evil.tld/` resolves to `evil.tld`. A URL that cannot be parsed is treated as disallowed.
### `injection` — Prompt injection (heuristic)
| Setting | Default | Notes |
| --- | --- | --- |
| `min_hits` | `1` | 1–10. Number of injection-like phrases needed before the check triggers |
Matches the phrasings that appear in real indirect-injection payloads: instruction overrides ("ignore previous instructions"), role hijacks ("you are now…"), system-prompt exfiltration ("reveal your instructions"), fake conversation turns (`system:` at the start of a line) and tool coercion ("you must immediately call the tool…"). It is most valuable at the `retrieval` and `tool_result` stages, where text an attacker may have planted arrives with the user's authority.
<Callout type="warning" emoji="⚠️">
This check catches unobfuscated payloads only. A motivated attacker can evade it trivially. Pair it with the `policy` judge if you need semantic coverage.
</Callout>
### `groundedness` — Grounding in sources
Output-only. Measures the lexical overlap between the answer and the retrieved sources using 4-word shingles and flags answers that fall below a threshold.
| Setting | Default | Notes |
| --- | --- | --- |
| `min_overlap` | `0.3` | 0–1. Fraction of the answer's shingles that must appear in the sources |
| `min_words` | `25` | 1–1000. Shorter answers are skipped |
| `require_retrieval` | `true` | When `true`, an answer produced with **no** retrieved sources triggers with category `NO_SOURCES` |
Lexical overlap is a proxy for support, not entailment. Keep this on `flag` until you have tuned the threshold against real traffic. When the sources contain no comparable text the check reports *not evaluated* rather than a verdict.
### `policy` — Custom policy (LLM judge)
Write a policy in plain language — a topic to stay off, a tone to hold, a rule to enforce — and a judge model decides whether the content breaks it.
| Setting | Default | Notes |
| --- | --- | --- |
| `policy` | — (required) | 10–2500 characters of policy text |
| `confidence_threshold` | `0.7` | 0–1. The judge must both report a violation **and** be at least this confident |
| `max_chars` | `8000` | 200–100000. Only the first `max_chars` of the content are sent to the judge |
| `model` | `null` | Optional model id override for this control |
The judge is the instance's own model provider, so a self-hosted deployment gets a semantic guardrail with no extra vendor account. The model used is, in order: the control's `model`, then the instance-wide `GUARDRAILS_JUDGE_MODEL`, then the model the agent is answering with. Judge calls are tagged `guardrail` in token usage so their cost shows up separately from the agent's own generation.
The content is passed to the judge as untrusted data inside a delimited envelope, with fences and the envelope's own tags neutralised, and the judge is instructed to ignore any directions it finds inside. If the judge times out, errors, or returns something unparsable, the control reports *not evaluated* and the fail-open policy below decides what happens.
## When a check cannot run
A timeout, a provider error, or a missing judge model is **not** a clean pass. The control reports `not_evaluated`, the journal records it, and the agent's failure policy applies:
| Setting | Default | Meaning |
| --- | --- | --- |
| `fail_open` | `true` | Let the turn continue when a check could not run. Set to `false` to block the turn instead whenever a `block` **or** `redact` control could not run — fail-closed exists so that unscanned text never reaches the user, and a broken PII detector would otherwise release exactly what it was there to remove |
| `timeout_ms` | `2000` | 100–60000. Deadline for the remote checks at one stage. Local pattern checks run inline and are not subject to it |
Remote controls at one stage run in parallel under a single deadline. At most 8 remote controls run per stage; any beyond that are reported as *not evaluated*.
## Configuring guardrails in the UI
Open the agent in the builder and expand the **Guardrails** section.
1. **Enable guardrails.** Nothing runs until this is on.
2. **Enforcement mode.** Leave it on *Monitor only* while you calibrate; switch to *Enforce everywhere* when the journal looks right.
3. **Checks.** Each check is a card with one chip per supported stage. Turning a chip on adds a control with the `flag` action; use the action selector on the chip to promote it to `redact` or `block`, and **Configure** to edit its settings. The card shows the approximate latency the check adds.
4. **Blocked-response message.** Up to 500 characters, shown to the user whenever a control blocks. Defaults to *"Sorry, I can't help with that request."*
5. **Continue if a check fails** and **Check timeout (ms)** map to `fail_open` and `timeout_ms`.
Checks that cannot run without settings (`denylist`, `url`, `policy`, and `pii` with no entities selected) are marked *Not configured* and block saving until they are filled in, so a half-configured control can never be published as if it were protecting you.
<Callout type="info" emoji="ℹ️">
Guardrails are the **agent owner's** policy. Team members with edit access can see the configuration but cannot change it — an editor who could clear a control would silently strip protection from everyone using the agent. Controls required by the instance floor (below) appear locked and cannot be removed.
</Callout>
Guardrails also apply to a **draft** agent in the builder preview, which is the natural place to try a control before publishing.
## Configuring guardrails via the API
Guardrails live under `guardrails` in the agent's `config` field. Pass `config` as a JSON string when creating or updating an agent through `POST /api/create_agent` or `PUT /api/update_agent/<agent_id>` (the same multipart form the builder uses). `update_agent` replaces the whole `config`; send the complete object each time.
```json
{
"guardrails": {
"enabled": true,
"mode": "scan_all",
"fail_open": true,
"timeout_ms": 2000,
"block_message": "Sorry, I can't help with that request.",
"controls": [
{ "check": "secrets", "stage": "output", "action": "redact" },
{ "check": "pii", "stage": "input", "action": "redact",
"settings": { "entities": ["EMAIL", "PHONE", "CREDIT_CARD"] } },
{ "check": "injection", "stage": "retrieval", "action": "block",
"settings": { "min_hits": 1 } },
{ "check": "denylist", "stage": "output", "action": "redact",
"settings": { "terms": ["Project Nimbus", "Acme Corp"], "match": "word" } },
{ "check": "url", "stage": "output", "action": "redact",
"settings": { "allow_hosts": ["docs.example.com", "example.com"] } },
{ "check": "policy", "stage": "output", "action": "block",
"settings": {
"policy": "Never give legal, medical or investment advice. Never quote pricing that is not in the retrieved sources.",
"confidence_threshold": 0.8
} },
{ "check": "groundedness", "stage": "output", "action": "flag",
"settings": { "min_overlap": 0.3, "min_words": 25 } }
]
}
}
```
<Tabs items={['cURL', 'Python']}>
<Tabs.Tab>
```bash
curl -X PUT http://localhost:7091/api/update_agent/<agent_id> \
-H "Authorization: Bearer <token>" \
-F 'config={"guardrails":{"enabled":true,"mode":"monitor_only","controls":[{"check":"secrets","stage":"output","action":"redact"}]}}'
```
</Tabs.Tab>
<Tabs.Tab>
```python
import json, requests
config = {"guardrails": {
"enabled": True,
"mode": "monitor_only",
"controls": [{"check": "secrets", "stage": "output", "action": "redact"}],
}}
requests.put(
"http://localhost:7091/api/update_agent/<agent_id>",
headers={"Authorization": "Bearer <token>"},
data={"config": json.dumps(config)},
).raise_for_status()
```
</Tabs.Tab>
</Tabs>
Each control accepts `check`, `stage`, `action` (default `flag`), `enabled` (default `true`) and `settings`. Omitted top-level fields take the defaults shown above; `enabled` defaults to `false`.
Writes are validated **strictly**. The request is rejected with HTTP 400 and the message *"Invalid config: one or more guardrail controls failed validation."* when:
- a `check` is unknown, or is not allowed by the instance's `GUARDRAILS_CHECKS_ENABLED`;
- a check is attached to a stage it does not support (for example `groundedness` at `input`);
- `redact` is requested on a check that reports no spans (`injection`, `groundedness`, `policy`);
- a control's settings are out of range, or a required setting is missing;
- the same `(check, stage)` pair appears twice, or there are more than 50 controls;
- `mode`, `timeout_ms` or `block_message` is out of bounds.
Reads are **lenient**: a stored control that has stopped validating (its check was disallowed by the operator, or removed in an upgrade) is dropped on its own and logged, and the remaining controls keep running. Agent export files carry `config`; on import an invalid guardrails block is dropped with a warning rather than failing the import.
### What a blocked turn looks like to a client
On the streaming endpoints, a block produces two final events. The first tells the client to retract anything it has rendered; the second carries the operator's block message as a user-facing error:
```json
{"type": "guardrail", "guardrail": {"stage": "output", "categories": ["AWS_ACCESS_KEY"], "checks": ["secrets"]}, "retract": true}
{"type": "error", "error": "Sorry, I can't help with that request."}
```
The DocsGPT chat UI and the React widget handle both. For webhook and scheduled runs, the run is recorded with the block message and none of the blocked text is stored.
## The audit journal
Every control that **triggers**, and every control that **could not run**, writes a row to the `guardrail_events` table — in both enforcement modes, and for `flag` actions as well as `redact` and `block`. A streamed answer that re-matches the same span on every chunk produces one row, not one per chunk. Rows also flow into the turn's activity log under the `guardrail` component, so a decision is visible next to the tool calls and retrieval it belongs to.
### Guardrail activity panel
Open an agent's **Logs** page. Below the usage logs, the **Guardrail activity** panel shows, for a trailing window of 7, 30 or 90 days:
- four totals — **Blocked**, **Redacted**, **Flagged** and **Not evaluated** — because "we refused to answer", "we masked something", "we noticed something" and "a check silently stopped working" are four different problems;
- a per-check breakdown, so you can see which control is doing the firing;
- the most recent 100 decisions, filterable by check and outcome, each with its stage, category (`EMAIL`, `INSTRUCTION_OVERRIDE`, `UNGROUNDED`, …) and the detector's one-line detail.
### Journal API
| Endpoint | Purpose |
| --- | --- |
| `GET /api/guardrails/catalog` | Available checks (with stages, redaction support, latency hint and remote flag), stages, modes, allowed actions per stage, PII entity names, the default block message, and which `(check, stage)` pairs the instance floor claims. |
| `GET /api/guardrails/events?agent_id=<id>&limit=100&offset=0` | Decisions for one agent, newest first. `limit` is capped at 500. |
| `GET /api/guardrails/summary?days=30&agent_id=<id>` | Totals and a `breakdown` grouped by check, stage, action, outcome and category. `agent_id` is optional; `days` is capped at 365. |
All three require a user token. Event rows are scoped to the **requesting user**: on a shared agent you see the decisions made on your own conversations, not other members'. Responses never include the agent's API key or the matched text.
### What is stored, and for how long
By default the journal records *that* something matched — check, stage, action, outcome, category, a score where the check produces one, a match count and a short detail string — but **not the text**. Pre-redaction text is exactly what a PII control exists to keep out of storage. Set `GUARDRAILS_STORE_SCANNED_TEXT=true` to persist a sample of the first matched value (up to 200 characters) alongside each row for forensic review; it is stored in the database but never returned by the API.
Rows older than `GUARDRAILS_EVENTS_RETENTION_DAYS` (default 30) are purged by a daily Celery beat task. The `message_id` link is set to `NULL` rather than cascading when a conversation is deleted, so the compliance trail outlives the conversation it came from.
## Instance settings
Operators control guardrails deployment-wide with these settings:
| Setting | Default | Purpose |
| --- | --- | --- |
| `GUARDRAILS_ENABLED` | `true` | Master switch. `false` disables every stage on every agent; the builder shows a notice explaining that nothing configured will run. |
| `GUARDRAILS_CHECKS_ENABLED` | `[]` | Allowlist of check keys. Empty means every registered check. A disallowed check cannot be saved through the API, and existing controls that use it are dropped on read. |
| `GUARDRAILS_FLOOR` | `{}` | A guardrails config fragment every agent inherits and cannot weaken. See below. |
| `GUARDRAILS_JUDGE_MODEL` | unset | Model id for `policy` controls that do not set their own `model`. Unset reuses the agent's model. |
| `GUARDRAILS_STORE_SCANNED_TEXT` | `false` | Persist a sample of matched text with each journal row. |
| `GUARDRAILS_EVENTS_RETENTION_DAYS` | `30` | Journal retention. Minimum 1. |
List and dict settings are read from the environment as JSON, for example:
```bash
GUARDRAILS_CHECKS_ENABLED='["pii", "secrets", "denylist", "url", "injection", "groundedness"]'
```
<Callout type="info" emoji="ℹ️">
Leaving `policy` out of `GUARDRAILS_CHECKS_ENABLED` is how an air-gapped or privacy-sensitive deployment guarantees that no user text is sent to a judge model, regardless of what agent owners configure.
</Callout>
### The instance floor
`GUARDRAILS_FLOOR` lets an operator impose a minimum policy on every agent. It uses the same shape as an agent's `guardrails` object and **must include `"enabled": true`** — without it the floor parses but applies to nothing, and a warning is logged.
```bash
GUARDRAILS_FLOOR='{
"enabled": true,
"mode": "scan_all",
"fail_open": false,
"controls": [
{"check": "secrets", "stage": "output", "action": "redact"},
{"check": "secrets", "stage": "tool_result", "action": "redact"},
{"check": "injection", "stage": "retrieval", "action": "block"}
]
}'
```
The floor is merged into each agent's own configuration at run time. An agent may tighten, never loosen:
- Guardrails are forced on for every agent, even one whose owner never enabled them.
- If the floor's mode is `scan_all`, the merged mode is `scan_all`. Otherwise the agent's mode stands.
- If the floor sets `fail_open: false`, the agent is fail-closed. The merged `timeout_ms` is the larger of the two.
- Floor controls the agent does not define are added.
- Where both define the same `(check, stage)`, the floor's **settings** are authoritative and the **stricter action** wins (`block` > `redact` > `flag`). The two settings dicts are deliberately not merged: adding to `denylist.terms` tightens, but adding to `url.allow_hosts` loosens, so an agent that could edit floor settings could always find a loosening edit. An agent that needs different settings attaches its own control at a stage the floor does not claim.
In the builder, floor controls appear as active and locked. The catalog exposes only which `(check, stage)` pairs the floor claims and their action — the floor's settings (banned-term lists, policy prompts) stay server-side so they cannot be read and evaded by any authenticated user.
The floor also applies to the individual AI Agent nodes inside a workflow, which do not otherwise carry per-agent controls (see below).
## Scope and limitations
- **Workflow agents** run the `input` stage with their own controls, but the AI Agent nodes inside the workflow run only the instance floor, not the parent agent's controls. Aggregate output guarding across a workflow is not yet wired.
- **Pattern checks match formats, not meaning.** `pii` does not find names; `injection` misses obfuscated payloads; `groundedness` measures word overlap, not truth. Use `policy` where you need a semantic judgement, and keep an eye on the *Not evaluated* count when you do.
- **Redaction is best-effort against structured identifiers.** A value that does not match a known format passes through. Treat `redact` as a safety net, not as a data-loss-prevention guarantee.
- **Latency.** Local checks add low single-digit milliseconds. A `policy` control adds a model round trip per scanned segment, and on the `output` stage it holds the stream until each sentence boundary has been judged.
## Extending: writing your own check
Checks are plain Python classes registered with a small registry, mirroring how chunkers and retrievers are pluggable. Subclass `GuardrailCheck` from `docsgpt.guardrails`, declare the stages you support, implement `scan`, and register it:
```python
from docsgpt.guardrails import GuardrailCheck, GuardrailCreator, Stage
from docsgpt.guardrails.types import CheckOutcome, Span
class TicketIdCheck(GuardrailCheck):
name = "ticket_id"
label = "Internal ticket ids"
description = "Masks references to internal tracker tickets."
supported_stages = {Stage.OUTPUT, Stage.TOOL_RESULT}
supports_redaction = True # scan() reports spans
latency_hint_ms = 1
max_match_chars = 16 # longest match; sizes the streaming window
def scan(self, text, stage, context):
import re
spans = [
Span(m.start(), m.end(), "TICKET_ID", replacement="[TICKET]")
for m in re.finditer(r"\bOPS-\d{3,6}\b", text)
]
if not spans:
return CheckOutcome.clean()
return CheckOutcome.hit(categories=["TICKET_ID"], spans=spans,
detail=f"{len(spans)} ticket id(s)")
GuardrailCreator.register(TicketIdCheck.name, TicketIdCheck)
```
Override `validate_settings` to strictly validate and normalise per-control settings on write, set `remote = True` for anything that makes a network call (so it runs under the stage deadline), and set `requires_complete_text = True` if the verdict is only meaningful over a finished answer. `max_match_chars` must cover the longest span the check can report, or the streaming guard may release the tail of a match before scanning it. Once registered, the check appears in the catalog and the builder automatically.
+3 -1
View File
@@ -88,6 +88,7 @@ Admins get a dashboard backed by a REST surface under `/api/admin` (every endpoi
| `GET` | `/api/admin/audit` | Authentication/admin audit feed. |
| `GET` | `/api/admin/devices/audit` | Remote-device audit feed. |
| `GET` | `/api/admin/teams` | Instance-wide oversight of all teams. |
| `GET` `PUT` `DELETE` | `/api/admin/quotas/...` | [Usage quotas](/Deploying/Usage-Quotas) for the instance, teams and users. |
<Callout type="info" emoji="ℹ️">
Deactivating a user via the dashboard works for any auth type, while OIDC deployments can also offboard through [SCIM](/Deploying/OIDC-SSO#scim-user-provisioning). Both revoke live sessions immediately.
@@ -141,9 +142,10 @@ Sharing rules:
## Audit log
Access-control actions are appended to the `auth_events` table alongside the [authentication events](/Deploying/OIDC-SSO#login-auditing). This includes admin actions — `admin_user_activated` / `admin_user_deactivated`, `admin_sessions_revoked`, `role_granted` / `role_revoked` (with `metadata.source` = `manual` or `oidc_group`) — and team events (`team.create`, `team.member_add`, `team.member_role`, `team.member_remove`, `team.share`, `team.unshare`, `team.transfer_owner`, `team.delete`). The acting admin is recorded in the event metadata.
Access-control actions are appended to the `auth_events` table alongside the [authentication events](/Deploying/OIDC-SSO#login-auditing). This includes admin actions — `admin_user_activated` / `admin_user_deactivated`, `admin_sessions_revoked`, `role_granted` / `role_revoked` (with `metadata.source` = `manual` or `oidc_group`), `quota_policy_set` / `quota_policy_deleted` — and team events (`team.create`, `team.member_add`, `team.member_role`, `team.member_remove`, `team.share`, `team.unshare`, `team.transfer_owner`, `team.delete`). The acting admin is recorded in the event metadata.
## Related
- [SSO with OIDC](/Deploying/OIDC-SSO) — sign-in, group allowlists, and the `auth_events` table.
- [Usage Quotas](/Deploying/Usage-Quotas) — token and cost limits per user and per team.
- [App Configuration](/Deploying/DocsGPT-Settings) — the full settings reference.
+3 -2
View File
@@ -39,13 +39,14 @@ On a machine with internet access, pull the images and save them to one file:
```bash
TAG=latest # or a release, e.g. 0.19.0
docker pull arc53/docsgpt:$TAG
docker pull arc53/docsgpt-fe:$TAG
docker pull redis:6-alpine
docker pull postgres:16-alpine
docker save -o docsgpt-images.tar \
arc53/docsgpt:$TAG arc53/docsgpt-fe:$TAG redis:6-alpine postgres:16-alpine
arc53/docsgpt:$TAG redis:6-alpine postgres:16-alpine
```
The backend image serves the web UI too, so the standalone stack needs no frontend image. Add `arc53/docsgpt-fe:$TAG` only if you run the checkout Compose files or Kubernetes, which use it.
Copy `docsgpt-images.tar` and the [standalone Compose file](/Deploying/Docker-Deploying#quickest-setup-pre-built-images-no-checkout) into the air-gapped network, then load the images (or push them to your internal registry):
```bash
@@ -78,7 +78,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
3. **Embedding Model (no action needed):**
The embedding model is downloaded automatically the first time you ingest a document, and cached under `models/` in the repository root for subsequent runs. Set `EMBEDDINGS_CACHE_DIR` to use another directory.
The embedding model is downloaded automatically the first time you ingest a document, and cached for subsequent runs under `models/` in the data home: the repository root, unless `DOCSGPT_HOME` points elsewhere. Set `EMBEDDINGS_CACHE_DIR` to use another directory.
For an offline or air-gapped machine, fetch it ahead of time instead:
@@ -119,7 +119,34 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
5. **Run the Backend:**
For local development, run the ASGI composition under uvicorn. It serves the **whole** application, hot-reloads on source changes, and matches the production runtime:
One command runs the API and the worker from this checkout, each restarting when you save a file:
```bash
docsgpt dev
```
Both run as children of that terminal, with their output interleaved and labelled, and Ctrl-C stops
them together. Useful flags:
| Flag | What it does |
| --- | --- |
| `--ui` | also start the Vite dev server, so the whole app runs from one command |
| `--mock-llm` | run `scripts/mock_llm.py` and point DocsGPT at it, so no API key is needed |
| `--no-worker` | leave the worker to you, for instance when debugging it in your editor |
| `--no-reload` | do not restart anything on save |
| `--port` | serve the API somewhere other than 7091 |
`docsgpt dev` is for a checkout. `docsgpt up --native`, by contrast, installs supervised services
that outlive the shell — see [Run it as services](/Deploying/Pip-Install#run-it-as-services-without-docker).
<Callout type="info">
`docsgpt doctor` checks the things that usually break a new setup: whether PostgreSQL answers and
its schema matches this version, whether Redis answers, whether a model provider is configured,
and whether the port is free. Run it first when something does not start.
</Callout>
To run the two processes yourself instead, start the ASGI composition under uvicorn. It serves the
**whole** application, hot-reloads on source changes, and matches the production runtime:
```bash
uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload
@@ -135,7 +162,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
But it serves **only** the WSGI Flask app and omits the native-async routes mounted on the ASGI shell in `docsgpt/asgi.py`: the `/mcp` FastMCP endpoint, the chat reconnect reader `GET /api/messages/<id>/events`, the notification stream `GET /api/events`, the remote-device command stream `GET /api/devices/sessions/<id>/events`, and artifact downloads `GET /api/artifacts/<id>/download`. Under `flask run` those paths return 404 — chat still works (`POST /stream` is a Flask route), but live notifications, stream auto-resume, paired devices and artifact downloads don't. Use `flask run` only when you don't need them.
6. **Start the Celery Worker:**
6. **Start the Celery Worker** (not needed if you used `docsgpt dev`)**:**
Open a new terminal window (and activate your virtual environment if you used one). Start the Celery worker to handle background tasks:
@@ -153,10 +180,14 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a
**Running in Debugger (VSCode):**
For easier debugging, you can launch the Flask app and Celery worker directly from VSCode's debugger.
For easier debugging, you can launch the API and the Celery worker directly from VSCode's debugger.
* Press <kbd>Shift</kbd> + <kbd>Cmd</kbd> + <kbd>D</kbd> (macOS) or <kbd>Shift</kbd> + <kbd>Windows</kbd> + <kbd>D</kbd> (Windows) to open the Run and Debug view.
* You should see configurations named "Flask" and "Celery". Select the desired configuration and click the "Start Debugging" button (green play icon).
* You should see configurations named "API (uvicorn)" and "Celery worker", and a compound "DocsGPT: Full Stack" that starts them with the frontend. Select one and click the "Start Debugging" button (green play icon).
The API configuration runs the same ASGI app as production, so the routes mounted on the ASGI shell
work under the debugger. It deliberately runs without `--reload`: the reloader restarts the server in
a child process, which your breakpoints would not be attached to.
## 3. Start the Frontend
@@ -207,3 +238,16 @@ To run the DocsGPT frontend locally, you'll need Node.js and npm (Node Package M
This command will start the Vite development server. The frontend application will typically be accessible at [http://localhost:5173/](http://localhost:5173/). The terminal will display the exact URL where the frontend is running.
With both the backend and frontend running, you should now have a fully functional DocsGPT development environment. You can access the application in your browser at [http://localhost:5173/](http://localhost:5173/) and start developing!
## Working on two branches at once
Each install keeps its own directory and its own services, so a second branch can run beside the
first as long as it gets its own port:
```bash
docsgpt dev --port 7092 # a second checkout, second terminal
docsgpt up --native --dir ~/.docsgpt/review --port 7092 # or a second installed copy
```
A native install in another directory gets its own service names, so the two never write over each
other's units. `docsgpt status --dir ~/.docsgpt/review` reports on that one alone.
+186 -5
View File
@@ -17,14 +17,122 @@ Docker is the recommended method for deploying DocsGPT, providing a consistent a
**Important Note for Windows Users:** Docker Desktop on Windows generally requires the WSL 2 backend to function correctly, especially when using features like host networking which are utilized in DocsGPT's Docker Compose setup. Ensure WSL 2 is enabled and configured in Docker Desktop settings.
## Run it with `docsgpt up`
The `docsgpt` Python package can set up and run the stack described below for
you. It needs Docker with Compose 2.24 or newer. The installer gets
[uv](https://docs.astral.sh/uv/), installs the package with it and runs
`docsgpt up`:
macOS and Linux:
```bash
curl -fsSL https://docs.ac/install | bash
```
Windows (PowerShell):
```powershell
irm https://docs.ac/install.ps1 | iex
```
Both scripts are attached to every [release](https://github.com/arc53/DocsGPT/releases)
as `install.sh` and `install.ps1`. To install the package yourself instead
(Python 3.12 or newer; uv brings one when it is missing):
```bash
uv tool install docsgpt # or: pipx install docsgpt
docsgpt up
```
`docsgpt up` keeps the stack in `~/.docsgpt/server` (`/opt/docsgpt` when run as
root on Linux; `--dir` or `DOCSGPT_HOME` choose another folder): the Compose
file of the installed version, a `.env` with your settings and the generated
secrets, and `install.json`. Data lives in named Docker volumes. The first run
asks two questions:
- **Who should reach DocsGPT:** only this computer; other machines on the
network (plain HTTP, with `AUTH_TYPE=simple_jwt` and an access token); or a
domain name with HTTPS (Caddy gets the certificate, access token as well).
- **Which model provider:** the DocsGPT public API (no key needed), OpenAI,
Anthropic, Google Gemini, OpenRouter, Groq, or an OpenAI-compatible server
such as Ollama or vLLM.
Flags answer the same questions, for scripts and servers:
```bash
docsgpt up --yes --domain docs.example.com --provider openai --api-key "$OPENAI_API_KEY"
```
Running `docsgpt up` again is safe: it keeps `.env` and the secrets and runs
the images of the installed package version. `docsgpt up --reconfigure` asks
the questions again.
| Command | What it does |
| --- | --- |
| `docsgpt status` | Version, address, containers, and whether the API answers |
| `docsgpt logs [-f] [service]` | Container logs |
| `docsgpt token` | The access token, for installs reachable beyond this computer |
| `docsgpt open` | Open DocsGPT in the browser |
| `docsgpt env set KEY=VALUE` | Change a setting; `docsgpt up` applies it |
| `docsgpt upgrade` | Upgrade the package (for `uv tool` installs) and restart on the new version |
| `docsgpt down` | Stop the stack; data and settings stay |
| `docsgpt uninstall [--purge]` | Remove the containers; `--purge` also deletes the settings and all data |
### Backups
`docsgpt backup` writes one archive holding a dump of the database and a tar of
each data volume (`indexes`, `inputs`, `vectors`):
```bash
docsgpt backup # into <stack>/backups
docsgpt backup --out /mnt/backups # somewhere else, e.g. a mounted disk
```
While the archive is made, the backend and the worker stop and start again, so
the database dump and the files in the volumes describe the same moment; Postgres
itself keeps running. On a small install that pause is seconds, but count on it
if you run `docsgpt backup` from cron. The archive is written readable only by
the user who took it.
The archive does **not** include `.env`, because that file holds the install's
secrets. `docsgpt backup --with-settings` puts it in, for when the archive
itself is stored somewhere private. Keep `.env` safe separately otherwise: the
database password in it is what an existing Postgres volume expects.
Restoring replaces the data in an install:
```bash
docsgpt restore ~/.docsgpt/server/backups/docsgpt-20260916-120000.tar.gz
```
It asks first, then stops the stack, puts the volumes and the database back, and
starts DocsGPT again. `--yes` skips the question for scripts. A backup taken
with a newer DocsGPT is refused, since its data may not fit this version's
schema; upgrade first, or pass `--force` if you know the two match.
The Postgres data directory itself is not archived: the dump is the database
backup, and copying a directory Postgres is writing to would capture a torn
copy. Caddy's certificates are not archived either, as it obtains them again.
More `docsgpt up` options: `--port`, `--docling` (the image with the docling
parser engine and OCR), `--image-tag develop` (follow the `main` branch) and
`--adopt` (manage a stack you started from the standalone Compose file in
another folder; both use the same data volumes). For Ollama on the same
machine, use the base URL `http://host.docker.internal:11434/v1`; on Linux,
also make Ollama listen beyond localhost (`OLLAMA_HOST=0.0.0.0`).
## Quickest Setup: Pre-built Images, No Checkout
Every release publishes ready-to-run images to Docker Hub (`arc53/docsgpt`,
`arc53/docsgpt-fe`) and GitHub Container Registry (`ghcr.io/arc53/docsgpt`,
`ghcr.io/arc53/docsgpt-fe`) for `linux/amd64` and `linux/arm64`. The images
contain everything the default configuration needs (embedding models,
tokenizers, tiktoken's encoding), so a fresh container makes no downloads on
first use. You do not need the source tree to run them:
`ghcr.io/arc53/docsgpt-fe`) for `linux/amd64` and `linux/arm64`.
`arc53/docsgpt` runs the API, serves the web UI and runs the worker;
`arc53/docsgpt-fe` is the separate frontend image the checkout Compose files
and Kubernetes use. The images contain everything the default configuration
needs (embedding models, tokenizers, tiktoken's encoding), so a fresh
container makes no downloads on first use. You do not need the source tree to
run them:
1. **Download the standalone Compose file** (also attached to every
[release](https://github.com/arc53/DocsGPT/releases)):
@@ -52,7 +160,9 @@ first use. You do not need the source tree to run them:
docker compose -f docker-compose-standalone.yaml up -d
```
Then open [http://localhost:5173/](http://localhost:5173/). Data lives in
Then open [http://localhost:7091/](http://localhost:7091/). The web UI and
the API share that port, which is published on `127.0.0.1`: only this
machine can reach it until you change `DOCSGPT_BIND` (below). Data lives in
named Docker volumes; `docker compose -f docker-compose-standalone.yaml down`
keeps it and `down -v` removes it.
@@ -66,6 +176,77 @@ the shell, e.g. `DOCSGPT_IMAGE_TAG=0.20.0 DOCSGPT_IMAGE_VARIANT=-docling`.
The same two variables drive `deployment/docker-compose-hub.yaml` in a
checkout.
### Opening it from other machines
Publish the port on every interface and turn on authentication in `.env`:
```bash
DOCSGPT_BIND=0.0.0.0
AUTH_TYPE=simple_jwt
JWT_SECRET_KEY=<a long random value, e.g. openssl rand -hex 32>
```
Then run `docker compose -f docker-compose-standalone.yaml up -d` again. The UI
takes its API address from the page it was loaded from, so
`http://<server-address>:7091/` works without further settings. Without
`AUTH_TYPE`, anyone who can reach the port can use DocsGPT.
With `simple_jwt` the UI asks for a token, which the backend prints when it
starts: `docker compose -f docker-compose-standalone.yaml logs backend | grep "Simple JWT"`.
The token is signed with `JWT_SECRET_KEY`. Without that setting each container
generates its own secret, and a re-created container (after `pull` or a
settings change) gets a new one and so a new token. Over plain HTTP the token
travels as readable text; outside a trusted network, use HTTPS as below.
`DOCSGPT_PORT` changes the host port (default `7091`). See
[Authentication Settings](/Deploying/DocsGPT-Settings#authentication-settings) for the other modes.
### HTTPS with your own domain
The Compose file has an optional Caddy service that obtains and renews a
Let's Encrypt certificate and proxies to the backend.
1. Point the domain's DNS records at the machine and open ports 80 and 443.
2. Add to `.env`:
```bash
COMPOSE_PROFILES=https
DOCSGPT_DOMAIN=docs.example.com
AUTH_TYPE=simple_jwt
JWT_SECRET_KEY=<a long random value, e.g. openssl rand -hex 32>
```
3. Run `docker compose -f docker-compose-standalone.yaml up -d` and open
`https://docs.example.com/`.
`COMPOSE_PROFILES=https` in `.env` makes every later `up`, `down` and `logs`
include Caddy. Leave `DOCSGPT_BIND` at its default: Caddy reaches the backend
over the Compose network.
### Database password
The Postgres password defaults to `docsgpt`; the database is only reachable
inside the Compose network. To use your own, set `POSTGRES_PASSWORD` in `.env`
before the first start, with URL-safe characters (e.g. `openssl rand -hex 24`).
Postgres reads it only when its volume is created, so changing it later does
not change the existing database's password.
### Upgrading from an earlier standalone file
Before this change the standalone file ran a separate frontend container on
port 5173 and published both ports on every interface. After downloading the
new file:
```bash
docker compose -f docker-compose-standalone.yaml pull
docker compose -f docker-compose-standalone.yaml up -d --remove-orphans
```
`--remove-orphans` removes the old frontend container. Open port 7091 instead
of 5173. Your data volumes are unchanged. If you opened DocsGPT from other
machines, follow [Opening it from other machines](#opening-it-from-other-machines),
and remove `VITE_API_HOST` from `.env` if it points at `localhost`: the UI
would otherwise keep calling the visitor's own machine.
## Using the Source Checkout
With a clone of the repository, `deployment/docker-compose-hub.yaml` runs the
+9 -11
View File
@@ -27,13 +27,13 @@ API_KEY=YOUR_OPENAI_API_KEY
LLM_NAME=gpt-4o
```
### 2. Configuration via `settings.py` file (Advanced)
### 2. Configuration in code (Advanced)
For more advanced configurations or if you prefer to manage settings directly in code, you can modify the `settings.py` file. This file is located in the `docsgpt/core` directory of your DocsGPT project.
Every setting is defined in the `docsgpt/core/settings/` package, one module per domain (`auth.py`, `llm.py`, `embeddings.py`, ...). If you prefer to manage defaults directly in code, change them there; the `.env` file and the process environment still override whatever the code says.
While modifying `settings.py` offers more flexibility, it's generally recommended to use the `.env` file for basic settings and reserve `settings.py` for more complex adjustments or when you need to configure settings programmatically.
Using the `.env` file is recommended for day-to-day configuration. Reserve code changes for new settings or for defaults you want every deployment of your fork to share.
**Location of `settings.py`:** `docsgpt/core/settings.py`
The [Settings Reference](/Deploying/Settings-Reference) lists every setting with its type, default and description, generated from those definitions.
## Basic Settings Explained
@@ -289,7 +289,7 @@ DocsGPT includes a JWT (JSON Web Token) based authentication feature for managin
### `AUTH_TYPE` Overview
The `AUTH_TYPE` setting in your `.env` file or `settings.py` determines the authentication method used by DocsGPT. This allows you to control how users authenticate with your DocsGPT instance.
The `AUTH_TYPE` setting in your `.env` file determines the authentication method used by DocsGPT. This allows you to control how users authenticate with your DocsGPT instance.
| Value | Description |
| ------------- | ------------------------------------------------------------------------------------------- |
@@ -300,7 +300,7 @@ The `AUTH_TYPE` setting in your `.env` file or `settings.py` determines the auth
#### How to Configure
Add the following to your `.env` file (or set in `settings.py`):
Add the following to your `.env` file:
```env
# Shared signing key (required in production for every authentication mode)
@@ -461,7 +461,6 @@ These control how sources are retrieved and whether the advanced RAG features ar
| Setting | Default | Description |
| --- | --- | --- |
| `RETRIEVERS_ENABLED` | `["classic", "default"]` | Allow-list of retrievers usable instance-wide. Valid keys: `classic`, `default`, `hybrid`, `graphrag`. A per-source `retriever` must be within this list. |
| `PER_SOURCE_RETRIEVAL_ENABLED` | `true` | Master switch for per-source retrieval config. When `false`, all sources fall back to the classic retriever regardless of their stored config. |
| `GRAPHRAG_ENABLED` | `false` | Enable [GraphRAG](/Sources/GraphRAG). Requires `VECTOR_STORE=pgvector`. |
| `GRAPHRAG_EXTRACTION_MODEL` | unset | Model used for ingest-time graph extraction. Unset reuses the instance default model. |
@@ -547,11 +546,10 @@ recovers.
## Exploring More Settings
These are just the basic settings to get you started. The `settings.py` file contains many more advanced options that you can explore to further customize DocsGPT, such as:
These are just the basic settings to get you started. DocsGPT has many more advanced options, such as:
- Vector store configuration (`VECTOR_STORE`, Qdrant, Milvus, LanceDB settings) If you're looking for an easy way to set up a vector store with pgvector, try [Neon](https://get.neon.com/docsgpt).
- Retriever settings (`RETRIEVERS_ENABLED`)
- Cache settings (`CACHE_REDIS_URL`)
- And many more!
- Sandbox, scheduler, guardrails and event-stream tuning
For a complete list of available settings and their descriptions, refer to the `settings.py` file in `docsgpt/core`. Remember to restart your Docker containers after making changes to your `.env` file or `settings.py` for the changes to take effect.
The [Settings Reference](/Deploying/Settings-Reference) lists every setting with its type, default and description. Remember to restart your Docker containers after making changes to your `.env` file for the changes to take effect.
+39 -2
View File
@@ -13,6 +13,10 @@ DocsGPT is on PyPI as [`docsgpt`](https://pypi.org/project/docsgpt/): the API se
`docsgpt api` serves the web UI on the same port as the API. Set `SERVE_UI=false` to run the API alone, for example behind the frontend Docker image or a UI you host yourself.
</Callout>
<Callout type="info">
With Docker available, the same package can run the whole stack for you, including Postgres and Redis: `docsgpt up`. See [Run it with `docsgpt up`](/Deploying/Docker-Deploying#run-it-with-docsgpt-up).
</Callout>
## Requirements
- Python 3.12 or newer
@@ -49,7 +53,11 @@ pipx runpip docsgpt install --force-reinstall --no-deps --index-url https://down
## Configure
DocsGPT keeps its runtime files in a **data home**: the `.env` file it reads settings from, uploaded files under `inputs/`, vector indexes under `indexes/` and downloaded embedding models under `models/`. The data home is the directory you run the commands from, or the directory `DOCSGPT_HOME` points to. `DOCSGPT_ENV_FILE` points at a `.env` kept somewhere else. Both variables must be set in the process environment, not in `.env`: they decide where `.env` is read from.
DocsGPT keeps its runtime files in a **data home**: the `.env` file it reads settings from, uploaded files under `inputs/`, vector indexes under `indexes/` and downloaded embedding models under `models/`. The data home is `~/.docsgpt/server` (`/opt/docsgpt` when you run as root on Linux), whatever directory you run the commands from. `DOCSGPT_HOME` moves it, and `DOCSGPT_ENV_FILE` points at a `.env` kept somewhere else. Both variables must be set in the process environment, not in `.env`: they decide where `.env` is read from. In a source checkout the data home is the checkout.
<Callout type="warning">
Up to 0.20 the data home of an installed package was the directory you ran the command from. If you kept `.env` and your data there, move them to `~/.docsgpt/server` or set `DOCSGPT_HOME` to that directory. `docsgpt api` and `docsgpt worker` point out a `.env` in the working directory that they no longer read.
</Callout>
Create a `.env` in the data home. The minimum for a hosted LLM:
@@ -80,7 +88,7 @@ docsgpt worker # in a second terminal: the Celery worker, with the scheduler
The API applies pending migrations when it starts (`AUTO_MIGRATE`), so `docsgpt migrate` is the explicit step for deployments that want the schema in place before the first request or that run the API with a restricted database role.
Both commands print the data home they resolved on start-up. Run them from the same directory, or set `DOCSGPT_HOME` for both, so the worker finds the files the API stores and the API finds the indexes the worker builds.
Both commands print the data home they resolved on start-up. They share it as long as `DOCSGPT_HOME` is the same for both (or unset), so the worker finds the files the API stores and the API finds the indexes the worker builds.
The worker is not optional: query embedding runs on it, so search fails without one. `docsgpt worker --help` lists the queue, concurrency and pool options; `--no-beat` starts a worker without the scheduler when another worker already runs it. On Windows the scheduler cannot be embedded, so run `docsgpt beat` in a third terminal.
@@ -91,6 +99,35 @@ Other commands:
- `docsgpt verify-offline`: check that a prepared install starts with networking off.
- `docsgpt reembed`: re-embed every index after changing `EMBEDDINGS_NAME` (see [Upgrading](/upgrading)).
## Run it as services, without Docker
`docsgpt up --native` runs the API and the worker as services on the machine itself: launchd agents on macOS, systemd user units on Linux. It does not start PostgreSQL or Redis; point it at ones you already run.
```bash
docsgpt up --native \
--postgres-uri postgresql://docsgpt:<password>@localhost:5432/docsgpt \
--redis-url redis://localhost:6379
```
Without a terminal only `--postgres-uri` is required, since Redis defaults to `redis://localhost:6379`; with one, it asks for both and for the model provider. It writes the same `.env` a Docker install uses (minus the image settings), generates `INTERNAL_KEY` and `JWT_SECRET_KEY` on the first run, applies the migrations, then starts `docsgpt-api` and `docsgpt-worker` and waits for the API to answer.
One Redis URL covers all three uses: the Celery broker, its result backend and the cache go on databases 0, 1 and 2 of it. Name a database in the URL and the three start there instead, so `redis://localhost:6379/5` puts them on 5, 6 and 7 — that is how you share a Redis that already holds something else.
The same commands manage it:
| Command | In native mode |
| --- | --- |
| `docsgpt status` | Which services run, the address, and whether the API answers |
| `docsgpt logs [api\|worker]` | The service log files under `<stack>/logs` |
| `docsgpt down` | Stops both services; settings stay |
| `docsgpt uninstall [--purge]` | Removes the services; `--purge` also deletes the stack directory |
`uninstall` never touches the database or Redis: they were yours to begin with.
<Callout type="info">
Windows has neither launchd nor systemd, so native mode is macOS and Linux only. On Windows, run DocsGPT on Docker with `docsgpt up`, or start `docsgpt api`, `docsgpt worker` and `docsgpt beat` yourself — the worker cannot run the scheduler in-process there, so `docsgpt beat` has to run alongside it for scheduled tasks to fire.
</Callout>
## Upgrade
```bash
File diff suppressed because it is too large. Load diff
+109
View File
@@ -0,0 +1,109 @@
---
title: Usage Quotas
description: Cap how many tokens or dollars each user may spend per day, week or month, with an instance default, per-team allowances and per-user overrides.
---
import { Callout } from 'nextra/components'
# Usage Quotas
An instance admin can limit how much each user spends on language models. A quota has two independent budgets:
- **Tokens** — prompt plus generated tokens. Works for every model, including local ones.
- **Cost (USD)** — tokens priced at the model's catalog rate. Only sees models that declare a price.
Set either, both or neither. Quotas are managed from **Admin → Quotas**, or through the [API](#api). With no quota set, nothing is limited.
## Layers
Limits are set at three layers. For each budget, the first layer that says something wins:
1. **User override** — one user's own limit.
2. **Team allowance** — what each member of a team gets.
3. **Instance default** — everyone else.
At each layer a budget is either *not set* (defer to the next layer), a *limit*, or *unlimited*. A limit of `0` blocks the user. The two budgets resolve separately, so a user's token limit can come from their team while their cost limit comes from the instance default.
### Teams
A team allowance is **per member**, not a pool the team shares: if the allowance is 2M tokens, each member may use 2M.
A user in several teams gets the **most generous** allowance among them, and allowances are never added together. Usage is always counted per user, whichever teams they belong to. To hold one person below their team's allowance, give them a user override.
<Callout type="info" emoji="ℹ️">
Team membership can change without an instance admin — team admins, OIDC group sync and SCIM all add members — so joining a team can only raise a user's allowance to what you granted that team, never lower it. Only instance admins set allowances; team admins cannot.
</Callout>
## Windows and enforcement
Usage is counted over a calendar window in UTC, chosen for the whole instance with [`QUOTA_PERIOD`](/Deploying/Settings-Reference#quotas): `day` (from 00:00), `week` (from Monday) or `month` (from the 1st, the default). Windows are worked out when a request arrives, so there is no reset job to run.
The quota is checked **before** a request starts. The request that crosses a limit completes; the next one is refused with HTTP `429`:
```json
{
"success": false,
"error_code": "quota-exceeded",
"message": "Usage quota reached (1,000,000 of 1,000,000 tokens). It resets at 2026-10-01T00:00:00+00:00.",
"dimension": "tokens",
"unit": "tokens",
"usage": 1000000,
"limit": 1000000,
"bucket": "all",
"source": "instance",
"resets_at": "2026-10-01T00:00:00+00:00"
}
```
The response carries a `Retry-After` header. The check covers chat, the agent and OpenAI-compatible APIs, scheduled runs (recorded as `budget_exceeded`) and webhook runs. If the quota check itself fails, the request is allowed.
Who is charged:
| Traffic | Charged to |
| --- | --- |
| Chat without an agent | The user |
| A user's own agent, its API key, webhooks and schedules | The agent's owner |
| An agent shared with the user | The user |
Per-agent token and request limits still apply on top of the owner's quota.
Users with a quota see their usage and the reset time under **Settings → Analytics**.
## Pricing
Cost budgets use the rates in the [model catalog](/Models/cloud-providers), in USD per million tokens:
```yaml
models:
- id: my-model
input_cost_per_million: 3.0
output_cost_per_million: 15.0
cached_input_cost_per_million: 0.3 # optional, prompt-cache reads
cache_write_cost_per_million: 3.75 # optional, prompt-cache writes
```
The built-in catalogs ship list prices for hosted models. Override or add rates by dropping a YAML with the same model `id` into `MODELS_CONFIG_DIR`. The cost of each call is stored with its usage row when the call is made, so later price changes do not rewrite history.
<Callout type="warning" emoji="⚠️">
A model with no declared price is recorded at $0, so a cost budget cannot see it. The Quotas tab lists such models once they have been used. Either limit them with a token budget, declare their rates, or set [`QUOTA_UNPRICED_RATE_PER_MILLION`](/Deploying/Settings-Reference#quotas) to charge a fallback rate. Models a user adds with their own API key are always $0, but their tokens still count.
</Callout>
## API
Every admin endpoint requires the admin role, and every change is written to the [audit log](/Deploying/Access-Control#audit-log) as `quota_policy_set` or `quota_policy_deleted`.
| Method | Path | Description |
| --- | --- | --- |
| `GET` | `/api/admin/quotas` | All policies by layer, the current window, and used models without a price. |
| `PUT` `DELETE` | `/api/admin/quotas/instance` | The instance default. |
| `GET` `PUT` `DELETE` | `/api/admin/quotas/teams/<team_id>` | A team's per-member allowance. |
| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/<user_id>` | A user's override; the user must already exist (SCIM-provisioned, or signed in once), otherwise `404`. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. |
| `GET` | `/api/user/quota` | The caller's own limits, usage and reset time. |
A `PUT` body sets, per budget, a limit or the unlimited flag; leave both out to defer to the next layer:
```json
{ "token_limit": 2000000, "cost_unlimited": true, "note": "Research team" }
```
`bucket` (default `all`) narrows a policy to `direct` traffic (chat without an agent) or `agent` traffic (anything that runs through an agent, whether or not the agent has an API key). A request must fit both its own bucket and `all`. The dashboard edits `all`.
+8
View File
@@ -3,6 +3,10 @@ export default {
"title": "⚙️ App Configuration",
"href": "/Deploying/DocsGPT-Settings"
},
"Settings-Reference": {
"title": "📖 Settings Reference",
"href": "/Deploying/Settings-Reference"
},
"OIDC-SSO": {
"title": "🔐 SSO with OIDC",
"href": "/Deploying/OIDC-SSO"
@@ -11,6 +15,10 @@ export default {
"title": "👥 Access Control & Teams",
"href": "/Deploying/Access-Control"
},
"Usage-Quotas": {
"title": "📊 Usage Quotas",
"href": "/Deploying/Usage-Quotas"
},
"Docker-Deploying": {
"title": "🛳️ Docker Setup",
"href": "/Deploying/Docker-Deploying"
+4
View File
@@ -3,6 +3,10 @@ export default {
"title": "🔑 Getting API key",
"href": "/Extensions/api-key-guide"
},
"personal-access-tokens": {
"title": "🎟️ Personal Access Tokens",
"href": "/Extensions/personal-access-tokens"
},
"chat-widget": {
"title": "💬️ Chat Widget",
"href": "/Extensions/chat-widget"
@@ -0,0 +1,203 @@
---
title: Personal Access Tokens
description: Scoped, revocable API tokens for managing agents, sources and other DocsGPT resources from the CLI, scripts and CI/CD pipelines.
---
# Personal Access Tokens
A personal access token (PAT) lets a script, the [DocsGPT CLI](https://github.com/arc53/DocsGPT-cli) or a CI/CD pipeline act on your account without a browser session. Unlike an [agent API key](/Extensions/api-key-guide), which can only talk to one agent, a PAT manages resources: it can create and update agents, upload sources, edit prompts and tools, and run agents for benchmarking.
Every token is limited in three ways:
- **Scopes** decide which parts of the API the token may call.
- **Resource restrictions** (optional) narrow a token to specific agents, sources, prompts, tools or workflows.
- **Expiry** ends the token's life automatically.
## Creating a token
1. Open **Settings → Access Tokens** in the DocsGPT web app.
2. Choose **Create token**, give it a name, and select the scopes it needs.
3. Optionally restrict it to specific resources and pick an expiry.
4. Copy the token. It starts with `dgpt_pat_` and is shown **once**. DocsGPT stores only a hash of it, so a lost token cannot be recovered. Revoke it and create a new one.
Tokens can only be created, regenerated and revoked from a signed-in session. A token cannot create, list, regenerate or revoke tokens, so a leaked token cannot mint a replacement for itself.
## Using a token
Send the token as a bearer credential:
```bash
export DOCSGPT_URL=https://docsgpt.example.com
export DOCSGPT_TOKEN=dgpt_pat_...
curl -H "Authorization: Bearer $DOCSGPT_TOKEN" "$DOCSGPT_URL/api/user/me"
```
`GET /api/user/me` works with any valid token and reports what the token may do, which makes it a convenient first step in a pipeline:
```json
{
"success": true,
"user_id": "alice@example.com",
"roles": ["user"],
"auth_method": "pat",
"token": {
"id": "0b6c...",
"name": "ci-deploy",
"scopes": ["agents:read", "agents:write"],
"resource_filter": {}
}
}
```
### Applying agent definitions
Agents can be exported to YAML and applied back, which makes them reviewable and deployable like any other configuration. With the CLI:
```bash
docsgpt-cli agents export <agent-id> -o support-bot.agent.yaml
docsgpt-cli agents apply -f support-bot.agent.yaml --dry-run
docsgpt-cli agents apply -f support-bot.agent.yaml
```
Or with the API directly (`agents:write`):
```bash
curl -X POST "$DOCSGPT_URL/api/import_agent/plan" \
-H "Authorization: Bearer $DOCSGPT_TOKEN" \
-H "Content-Type: application/json" \
-d "$(jq -Rs '{yaml: .}' support-bot.agent.yaml)"
```
Sources are matched by **name**, and when several of your sources share a name the oldest one wins. A pipeline that re-uploads documentation on every push should therefore upload with `docsgpt-cli sources upload ... --wait --replace` (which removes the older same-named sources) and run `agents apply` afterwards, or the agent stays bound to the first upload.
`/api/import_agent/plan` is a dry run that reports whether the agent would be created or updated and how each referenced source, tool and prompt resolves. `/api/import_agent` applies it. An agent is matched by `metadata.id`, then `metadata.slug`; when nothing matches, a new draft agent is created.
### GitHub Actions example
```yaml
jobs:
deploy-agents:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Apply agent definitions
env:
DOCSGPT_URL: ${{ vars.DOCSGPT_URL }}
DOCSGPT_TOKEN: ${{ secrets.DOCSGPT_TOKEN }}
run: |
docsgpt-cli sources upload docs/*.md --name "Product docs" --wait --replace --idempotency-key "docs-${{ github.sha }}"
docsgpt-cli agents apply -f agents/
```
## Scopes
A `write` scope includes the matching `read` scope.
| Scope | Allows |
| --- | --- |
| `agents:read` | View agents, folders, guardrail events and export agent definitions |
| `agents:write` | Create, update, delete, share and import (apply) agents and folders |
| `agents:keys` | Regenerate agent API keys and read incoming webhook URLs |
| `sources:read` | View sources, their files, chunks and ingestion task status |
| `sources:write` | Upload, ingest, sync, edit and delete sources and chunks |
| `prompts:read` / `prompts:write` | View / create, update and delete prompts |
| `tools:read` / `tools:write` | View / create, update and delete tools and MCP servers |
| `models:read` / `models:write` | View models / manage custom models |
| `workflows:read` / `workflows:write` | View / create, update and delete workflows |
| `schedules:read` / `schedules:write` | View / create, update, run and delete agent schedules |
| `conversations:read` / `conversations:write` | View / rename, delete and rate conversations |
| `analytics:read` | View usage analytics and logs |
| `teams:read` | View teams, members and resource shares |
| `chat:run` | Ask agents and search sources (`/api/answer`, `/stream`, `/api/search`); used for benchmarking |
`agents:keys` is separate from `agents:write` on purpose. Creating, publishing or adopting an agent mints its API key; a token without `agents:keys` gets the key back masked (`1234...90ab`), because an agent key keeps working after the token that saw it is revoked. Put differently: a deployment token that updates agents does not need to be able to read or rotate the secrets other systems use to call them.
Some parts of the API are never available to a token, whatever its scopes: token management, the admin API, team management, sign-in flows, device pairing, and the interactive OAuth handshakes used by connectors and MCP servers. A token also never carries the `admin` role, even when its owner is an admin.
Authorization is deny by default. An endpoint that is not explicitly mapped to a scope cannot be called with a token, and answers `403` with `"error": "not_available_to_tokens"`. A mapped endpoint called without the scope answers `403` with `"error": "insufficient_scope"` and names the `required_scope`.
## Resource restrictions
A token can be narrowed to specific resources in any of these families: `agents`, `sources`, `prompts`, `tools`, `workflows`. A family that is not listed stays unrestricted within the token's scopes.
```json
{
"name": "support-bot-deploy",
"scopes": ["agents:write", "chat:run"],
"resource_filter": { "agents": ["3f0e8f0c-5a53-4f0e-9a39-0e5f4f8d2c11"] },
"expires_in_days": 30
}
```
For a restricted family the token:
- can read, update and delete only the listed resources, and listings show only those;
- **cannot create** new resources of that family, since a new resource would be outside the list;
- cannot attach a resource outside the list to something else, for example set an agent's source to a source the token may not use;
- is refused (`403`, `"error": "resource_not_allowed"`) wherever DocsGPT cannot prove the request stays inside the list:
- **Conversations, analytics and message replay** (`/api/messages/<id>/tail`, `/api/messages/<id>/events`) are closed to every restricted token. They span all agents and contain cited source text and tool output.
- **Schedules** run an agent with a free-form instruction and store the output. A token restricted to agents can list and create schedules for its agents; every other schedule route, and schedules altogether for tokens restricted on another family, are closed.
- **Workflow writes** are closed to tokens restricted on sources, tools or prompts, because a workflow graph names those inside its nodes. Such a token also cannot attach a workflow to an agent unless it is restricted on workflows too, in which case only the listed workflows can be attached.
- `/api/sources/paginated` is closed to source-restricted tokens (use `/api/sources`).
A restriction covers what the token *asks for*, not what an allowed resource already contains: an agent on the list runs with its own sources, prompt and tools even when the token is also restricted on those families. List an agent only if you are happy for the token to use everything that agent uses.
Restrictions and chat (`chat:run`):
- A token restricted to specific **agents** must pass exactly one `agent_id` in the request body, and it must be one of the listed agents. An agent `api_key` or an inline workflow in the body is refused.
- A token restricted to specific **sources** only may chat against those sources with `active_docs`. It cannot run agents, because an agent brings its own sources. Restrict the token to agents instead to allow that.
- A token restricted on **prompts** or **workflows** must also be restricted to agents to chat.
- A token restricted on **tools** cannot use chat at all, and a tools restriction cannot be combined with `chat:run` when the token is created. Chat executes tools (an agent's own, or your default tools when there is no agent) and those cannot be held to a list.
- A `conversation_id` must belong to the agent being run (or to no agent, for agent-less chat). Otherwise the server would continue, append to, or resume pending tool calls of another agent's conversation.
`agents:write` and import: applying an agent definition can create the prompt and tools it references and rewrite the agent's workflow, all under `agents:write` alone. It does not need `prompts:write`, `tools:write` or `workflows:write`, so treat `agents:write` as able to create those through an import.
Restrictions and `agents apply`: a token restricted to specific agents can apply a definition only when it updates one of those agents. A token restricted on sources, prompts, tools or workflows cannot import agents at all, because an import resolves those references by name and may create them.
## Expiry and revocation
- A token created without an explicit lifetime expires after `PAT_DEFAULT_LIFETIME_DAYS` (90 by default). Users can choose any lifetime up to `PAT_MAX_LIFETIME_DAYS` (365 by default).
- Non-expiring tokens are available only when the operator sets `PAT_ALLOW_NON_EXPIRING=true`.
- **Regenerate** in **Settings → Access Tokens** issues a new secret for the same token and resets its expiry. The name, scopes and restrictions stay; the old secret stops working immediately, so update whatever uses it. The new lifetime defaults to the one the token was last issued with, and an expired token can be renewed this way (a revoked one cannot). This is the way to rotate a secret or extend a token without rebuilding its scopes.
- Revoking a token in **Settings → Access Tokens** takes effect on the next request.
- Admins can list a user's tokens with `GET /api/admin/users/<user_id>/tokens` and revoke any token with `DELETE /api/admin/tokens/<token_id>`. The admin **revoke sessions** action also revokes all of that user's tokens.
- Tokens of a deactivated user (through the admin API or SCIM) stop working immediately and work again if the user is reactivated.
- `GET /api/user/tokens` reports a token past its expiry as `"status": "expired"`.
- Token creation and revocation are recorded in the authentication audit log (`pat_created`, `pat_regenerated`, `pat_revoked`), visible to admins.
An expired token's name can be reused: creating a token with that name retires the expired one.
Each user may hold up to `PAT_MAX_PER_USER` live tokens (25 by default). The token list shows when and from which IP address each token was last used.
## Operator settings
| Setting | Default | Purpose |
| --- | --- | --- |
| `PAT_ENABLED` | `true` | Master switch. When `false`, tokens cannot be created **and every existing token stops authenticating immediately** |
| `PAT_DEFAULT_LIFETIME_DAYS` | `90` | Lifetime of a token created without an explicit expiry |
| `PAT_MAX_LIFETIME_DAYS` | `365` | Longest lifetime a user may request |
| `PAT_ALLOW_NON_EXPIRING` | `false` | Let users create tokens that never expire |
| `PAT_MAX_PER_USER` | `25` | Maximum number of live tokens per user |
Personal access tokens need a stable user identity, so they are available with `AUTH_TYPE=oidc` and with authentication disabled (single-user self-hosting). They are not available with `simple_jwt` or `session_jwt`.
Turning `PAT_ENABLED` off, or switching `AUTH_TYPE` to `simple_jwt` or `session_jwt`, is not limited to the settings page: every pipeline that uses a token starts getting `401` right away. Tokens are not deleted and work again once the setting is restored. See the [Settings Reference](/Deploying/Settings-Reference) for details.
## Management API
These endpoints need a signed-in session and cannot be called with a token.
| Endpoint | Purpose |
| --- | --- |
| `GET /api/user/tokens` | List your tokens, the scope catalog and the server's token policy |
| `POST /api/user/tokens` | Create a token. Body: `name`, `scopes`, optional `resource_filter`, optional `expires_in_days` (`0` = never, when allowed). The response carries the plaintext `token` once |
| `POST /api/user/tokens/<id>/regenerate` | New secret and new expiry for the same token. Optional body `expires_in_days`; omitted = the lifetime it was last issued with. The response carries the plaintext `token` once |
| `DELETE /api/user/tokens/<id>` | Revoke a token |
## Good practice
- Give each pipeline its own token with the narrowest scopes that work, and name it after where it is used.
- Store tokens in your CI system's secret store. Never commit them. The `dgpt_pat_` prefix lets secret scanners recognise them.
- Prefer short lifetimes for tokens used by automation you can easily re-provision.
- Revoke a token as soon as it is no longer needed or may have been exposed.
+3 -1
View File
@@ -98,9 +98,11 @@ The DocsGPT Search Bar Widget offers a range of customizable properties that all
|-----------------|-----------|-------------------------------------|--------------------------------------------------------------------------------------------------|
| **`apiKey`** | `string` | `"your-api-key"` | API key for authentication with your DocsGPT API. Leave empty if no authentication is required. |
| **`apiHost`** | `string` | `"https://gptcloud.arc53.com"` | **Required.** The URL of your DocsGPT API backend. This endpoint handles vector similarity search queries. |
| **`theme`** | `"dark" \| "light"` | `"dark"` | Color theme of the search bar. Options: `"dark"` or `"light"`. Defaults to `"dark"`. |
| **`theme`** | `"dark" \| "light"` | `"dark"` | Color theme of the search bar and of the chat it opens. Options: `"dark"` or `"light"`. Defaults to `"dark"`. |
| **`placeholder`** | `string` | `"Search or Ask AI..."` | Placeholder text displayed in the search input field. |
| **`width`** | `string` | `"256px"` | Width of the search bar. Accepts any valid CSS width value (e.g., `"300px"`, `"100%"`, `"20rem"`). |
| **`allowedFileExtensions`** | `string[]` | _unset_ | Passed to the chat opened from "Ask the AI". File extensions its composer accepts, e.g. `['.pdf', '.md']`; attachments stay off while unset. See the [chat widget page](/Extensions/chat-widget). |
| **`showMicButton`** | `boolean` | `false` | Adds a microphone to the search field for dictating a query, and passes the same option to the chat opened from "Ask the AI". Uses the browser's Web Speech API. See the [chat widget page](/Extensions/chat-widget). |
---
+1 -1
View File
@@ -251,4 +251,4 @@ The main extension points in [arc53/DocsGPT](https://github.com/arc53/DocsGPT) a
| Parsing and workers | [`docsgpt/parser/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/parser), [`docsgpt/worker.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/worker.py), [`docsgpt/api/user/tasks.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/api/user/tasks.py) |
| Models and vector stores | [`docsgpt/llm/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/llm), [`docsgpt/vectorstore/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/vectorstore) |
| Storage and events | [`docsgpt/storage/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/storage), [`docsgpt/streaming/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/streaming), [`docsgpt/events/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/events) |
| Configuration, UI, and deployment | [`docsgpt/core/settings.py`](https://github.com/arc53/DocsGPT/blob/main/docsgpt/core/settings.py), [`frontend/`](https://github.com/arc53/DocsGPT/tree/main/frontend), [`deployment/`](https://github.com/arc53/DocsGPT/tree/main/deployment) |
| Configuration, UI, and deployment | [`docsgpt/core/settings/`](https://github.com/arc53/DocsGPT/tree/main/docsgpt/core/settings), [`frontend/`](https://github.com/arc53/DocsGPT/tree/main/frontend), [`deployment/`](https://github.com/arc53/DocsGPT/tree/main/deployment) |
+1 -1
View File
@@ -19,7 +19,7 @@ The compression system operates on a "summarize and truncate" principle:
## Configuration
You can configure the compression behavior in your `.env` file or `docsgpt/core/settings.py`:
You can configure the compression behavior in your `.env` file or `docsgpt/core/settings/agents.py`:
| Setting | Default | Description |
| :--- | :--- | :--- |
@@ -115,8 +115,6 @@ Requests are also bounded: `chunks` is clamped to 0–500, and `0` still means
Keyword search for the **hybrid** retriever is currently implemented only for the **pgvector** vector store. On other stores (FAISS, Qdrant, Milvus, etc.) the keyword half returns nothing, so `hybrid` quietly behaves like `classic` (vector-only).
</Callout>
Operators can restrict which retrievers are usable instance-wide with the `RETRIEVERS_ENABLED` setting; a per-source `retriever` value must be within that allow-list.
### Exposure: prefetch vs. agentic tool
`exposure` controls *how* a source's content is delivered to the model:
+67
View File
@@ -11,6 +11,73 @@ The notable changes in each release. Every release on GitHub also carries
[auto-generated notes](https://github.com/arc53/DocsGPT/releases) listing every merged pull
request, and [Upgrading](/upgrading) covers the steps an existing deployment has to take.
## Unreleased
### A development loop in one command
`docsgpt dev` runs this checkout's API and worker as children of one terminal, both restarting when
you save, with their output interleaved and Ctrl-C stopping them together. `--ui` adds the Vite dev
server and `--mock-llm` runs the bundled mock model, so a working loop needs no API key.
`docsgpt doctor` checks PostgreSQL, its schema version, Redis, the model provider and the port;
`docsgpt restart` bounces the services without touching settings; `docsgpt logs -f` now follows a
native install; and `docsgpt env set` applies itself to a running native install instead of asking
you to run `docsgpt up` again. See
[Setting up a development environment](/Deploying/Development-Environment).
### Run DocsGPT without Docker
`docsgpt up --native` runs the API and the worker as services on the machine itself, launchd on
macOS and systemd user units on Linux, against a PostgreSQL and Redis you already have
(`--postgres-uri`, `--redis-url`). `status`, `logs`, `down` and `uninstall` work on such an install
the same way they do on a Docker one, and never touch the database or Redis. See
[Run it as services, without Docker](/Deploying/Pip-Install#run-it-as-services-without-docker).
### Back up and restore an install
`docsgpt backup` writes a dump of the database and a tar of each data volume into one archive, and
`docsgpt restore <archive>` puts them back. The settings file is left out unless
`--with-settings` asks for it, since it holds the install's secrets, and a backup from a newer
DocsGPT is refused unless you pass `--force`. The archive is written readable only by its owner,
and the backend and the worker pause while it is made so the dump and the volume tars match. See
[Backups](/Deploying/Docker-Deploying#backups).
## 0.21.0
### Install with one command
`curl -fsSL https://docs.ac/install | bash` on macOS and Linux, or `irm https://docs.ac/install.ps1 | iex`
in Windows PowerShell, installs uv and the `docsgpt` package and runs `docsgpt up`. Running it again
upgrades and keeps your settings. On Linux it offers to install Docker when it is missing. Both
scripts are attached to every release. See the [Quickstart](/quickstart).
### `docsgpt up` runs DocsGPT on Docker
The Python package now sets up and runs the Docker stack: `uv tool install docsgpt`, then
`docsgpt up`. The first run asks who should reach DocsGPT (this computer, the network with an
access token, or a domain with HTTPS) and which model provider to use, writes the settings and
secrets to `~/.docsgpt/server/.env`, and starts the images of the installed version. `docsgpt status`,
`logs`, `token`, `upgrade`, `down` and `uninstall` manage it afterwards. See
[Run it with `docsgpt up`](/Deploying/Docker-Deploying#run-it-with-docsgpt-up).
### An installed package keeps its data in `~/.docsgpt/server`
Outside a source checkout, the data home (`.env`, uploads, indexes, models) was the directory
the command ran from, so starting `docsgpt api` from another folder silently used other
settings. It is now `~/.docsgpt/server`, or `/opt/docsgpt` for root on Linux; `DOCSGPT_HOME`
still overrides it. See [Upgrading](/upgrading#pip-installs-data-home-moved).
### The standalone Docker stack runs on one port
The `arc53/docsgpt` image now serves the web UI next to the API, the way `docsgpt api` does from
the Python package. `docker-compose-standalone.yaml` no longer runs a frontend container: the UI
and the API share port 7091, published on `127.0.0.1` by default. The UI takes its API address
from the page it was loaded from, so opening the stack from another machine works without
setting `VITE_API_HOST`. New Compose settings: `DOCSGPT_BIND` and `DOCSGPT_PORT` for where the
port is published, `POSTGRES_PASSWORD`, and an `https` profile that puts Caddy with an automatic
certificate in front of a public domain. The `arc53/docsgpt-fe` image is still published for the
checkout Compose files and Kubernetes. See
[Upgrading from an earlier standalone file](/Deploying/Docker-Deploying#upgrading-from-an-earlier-standalone-file).
## 0.20.0
### DocsGPT installs from PyPI
+74 -67
View File
@@ -1,40 +1,91 @@
---
title: Quickstart - Launching DocsGPT Web App
description: Get started with DocsGPT quickly by launching the web application using the setup script.
description: Install and start DocsGPT with one command, or from a clone with the setup script.
---
import { Callout } from 'nextra/components'
# Quickstart
**Prerequisites:**
* **Docker:** Ensure you have Docker installed and running on your system.
* **Docker:** DocsGPT runs on Docker, with Docker Compose 2.24 or newer. On macOS and Windows install [Docker Desktop](https://docs.docker.com/desktop/) (or [OrbStack](https://orbstack.dev) on macOS); on Linux, [Docker Engine](https://docs.docker.com/engine/install/). On Linux the installer offers to install Docker for you.
## Launching DocsGPT (macOS and Linux)
## Install with one command
The easiest way to launch DocsGPT is using the provided `setup.sh` script. This script automates the configuration process and offers several setup options.
**macOS and Linux:**
**Steps:**
```bash
curl -fsSL https://docs.ac/install | bash
```
1. **Download the DocsGPT Repository:**
**Windows (PowerShell):**
First, you need to download the DocsGPT repository to your local machine. You can do this using Git:
```powershell
irm https://docs.ac/install.ps1 | iex
```
The installer:
1. Checks for Docker.
2. Installs [uv](https://docs.astral.sh/uv/) if it is missing or too old. uv installs Python packages and brings its own Python when the system has none.
3. Installs the `docsgpt` Python package with `uv tool install`.
4. Runs `docsgpt up`, which asks two questions:
* **Who should reach DocsGPT:** only this computer; other machines on your network (plain HTTP, with an access token); or a domain name with HTTPS (a certificate from Let's Encrypt, with an access token).
* **Which model provider:** the DocsGPT public API (no key needed), OpenAI, Anthropic, Google Gemini, OpenRouter, Groq, or an OpenAI-compatible server such as Ollama or vLLM.
It then starts DocsGPT and prints its address, [http://localhost:7091](http://localhost:7091) for a local install. Settings and generated secrets are in `~/.docsgpt/server/.env`, or `/opt/docsgpt/.env` when the installer runs as root on Linux; `docsgpt up --dir` or `DOCSGPT_HOME` choose another folder.
<Callout type="info">
To read the script before running it, download it first: `curl -fsSL https://docs.ac/install -o install.sh`, then `bash install.sh`. On Windows: `irm https://docs.ac/install.ps1 -OutFile install.ps1`, then `.\install.ps1`.
</Callout>
**Options.** Arguments after `bash -s --` go to `docsgpt up`, so a server can be set up without questions:
```bash
curl -fsSL https://docs.ac/install | bash -s -- --yes --domain docs.example.com --provider openai --api-key "$OPENAI_API_KEY"
```
`DOCSGPT_VERSION` installs a specific release, and `DOCSGPT_NO_MODIFY_PATH=1` leaves your shell profile alone. `docsgpt up --help` lists every option.
**Afterwards:**
| Command | What it does |
| --- | --- |
| `docsgpt status` | Version, address, and whether DocsGPT answers |
| `docsgpt logs -f` | Follow the logs |
| `docsgpt token` | The access token, for installs reachable beyond this computer |
| `docsgpt up --reconfigure` | Ask the setup questions again |
| `docsgpt upgrade` | Upgrade to the latest release, keeping your settings and data |
| `docsgpt down` | Stop DocsGPT |
| `docsgpt uninstall` | Remove it; `--purge` also deletes settings and data |
Running the install command again also upgrades. See [Run it with `docsgpt up`](/Deploying/Docker-Deploying#run-it-with-docsgpt-up) for the details.
## From a clone, with the setup script
To work from the source tree, for example to build the images yourself, use `setup.sh` (macOS and Linux) or `setup.ps1` (Windows).
1. **Clone the repository:**
```bash
git clone https://github.com/arc53/DocsGPT.git
cd DocsGPT
```
2. **Run the `setup.sh` script:**
Navigate to the DocsGPT directory in your terminal and execute the `setup.sh` script:
2. **Run the setup script:**
```bash
./setup.sh
```
3. **Follow the interactive setup:**
On Windows:
The `setup.sh` script will guide you through an interactive menu with the following options:
```powershell
PowerShell -ExecutionPolicy Bypass -File .\setup.ps1
```
3. **Follow the interactive setup:**
```
Welcome to DocsGPT Setup!
@@ -47,76 +98,32 @@ The easiest way to launch DocsGPT is using the provided `setup.sh` script. This
Choose option (1-5):
```
Let's break down each option:
* **1) Use DocsGPT Public API Endpoint (simple and free):** This is the simplest option to get started. It utilizes the DocsGPT public API, requiring no API keys or local model downloads.
* **1) Use DocsGPT Public API Endpoint (simple and free):** This is the simplest option to get started. It utilizes the DocsGPT public API, requiring no API keys or local model downloads. Choose this for a quick and easy setup.
* **2) Serve Local (with Ollama):** Runs a Large Language Model locally using [Ollama](https://ollama.com/). You'll be prompted to choose between CPU or GPU for Ollama and select a model to download.
* **2) Serve Local (with Ollama):** This option allows you to run a Large Language Model locally using [Ollama](https://ollama.com/). You'll be prompted to choose between CPU or GPU for Ollama and select a model to download. This is a good option for local processing and experimentation.
* **3) Connect Local Inference Engine:** If you already run a local inference engine like Llama.cpp, Text Generation Inference (TGI), vLLM, or others, choose this option and provide the connection details.
* **3) Connect Local Inference Engine:** If you are already running a local inference engine like Llama.cpp, Text Generation Inference (TGI), vLLM, or others, choose this option. You'll be asked to select your engine and provide the necessary connection details. This is for users with existing local LLM infrastructure.
* **4) Connect Cloud API Provider:** Connect DocsGPT to a Cloud API provider such as OpenAI, Google (Vertex AI/Gemini), Anthropic (Claude), Groq, HuggingFace Inference API, or Azure OpenAI. You will need an API key from your chosen provider.
* **4) Connect Cloud API Provider:** This option lets you connect DocsGPT to a commercial Cloud API provider such as OpenAI, Google (Vertex AI/Gemini), Anthropic (Claude), Groq, HuggingFace Inference API, or Azure OpenAI. You will need an API key from your chosen provider. Select this if you prefer to use a powerful cloud-based LLM.
* **5) Modify DocsGPT's source code and rebuild the Docker images locally.** Instead of pulling prebuilt images from Docker Hub, you build the backend and frontend from source, to customize how DocsGPT works internally or to run it in an environment without internet access.
* **5) Modify DocsGPT's source code and rebuild the Docker images locally.** Instead of pulling prebuilt images from Docker Hub or using the hosted/public API, you build the entire backend and frontend from source, customizing how DocsGPT works internally, or run it in an environment without internet access.
After selecting an option and providing any required information (like API keys or model names), the script configures your `.env` file and starts DocsGPT using Docker Compose.
After selecting an option and providing any required information (like API keys or model names), the script will configure your `.env` file and start DocsGPT using Docker Compose.
4. **Access DocsGPT in your browser:** open [http://localhost:5173/](http://localhost:5173/).
4. **Access DocsGPT in your browser:**
Once the setup is complete and Docker containers are running, navigate to [http://localhost:5173/](http://localhost:5173/) in your web browser to access the DocsGPT web application.
5. **Stopping DocsGPT:**
To stop DocsGPT, simply open a new terminal in the `DocsGPT` directory and run:
5. **Stopping DocsGPT:** in the `DocsGPT` directory, run the `docker compose down` command the script printed at the end, for example:
```bash
docker compose -f deployment/docker-compose-hub.yaml down
```
(or the specific `docker compose` command shown at the end of the `setup.sh` execution, which may include optional compose files depending on your choices).
## Launching DocsGPT (Windows)
**Important for Windows:** Ensure Docker Desktop is installed and running before you start. The script tries to start Docker if it is not running, but you may need to start it manually.
For Windows users, we provide a PowerShell script that offers the same functionality as the macOS/Linux setup script.
**Steps:**
1. **Download the DocsGPT Repository:**
First, you need to download the DocsGPT repository to your local machine. You can do this using Git:
```powershell
git clone https://github.com/arc53/DocsGPT.git
cd DocsGPT
```
2. **Run the `setup.ps1` script:**
Execute the PowerShell setup script:
```powershell
PowerShell -ExecutionPolicy Bypass -File .\setup.ps1
```
3. **Follow the interactive setup:**
Just like the Linux/macOS script, the PowerShell script will guide you through setting DocsGPT.
The script will handle environment configuration and start DocsGPT based on your selections.
4. **Access DocsGPT in your browser:**
Once the setup is complete and Docker containers are running, navigate to [http://localhost:5173/](http://localhost:5173/) in your web browser to access the DocsGPT web application.
5. **Stopping DocsGPT:**
To stop DocsGPT run the Docker Compose down command displayed at the end of the setup script's execution.
**Important for Windows:** Ensure Docker Desktop is installed and running correctly on your Windows system before proceeding. The script will attempt to start Docker if it's not running, but you may need to start it manually if there are issues.
**Alternative Method:**
If you prefer a more manual approach, you can follow our [Docker Deployment documentation](/Deploying/Docker-Deploying) for detailed instructions on setting up DocsGPT on Windows using Docker commands directly.
**Alternative Method:** To run the pre-built images with Docker Compose yourself, follow the [Docker Deployment documentation](/Deploying/Docker-Deploying).
## Advanced Configuration
For more advanced customization of DocsGPT settings, such as configuring vector stores, embedding models, and other parameters, please refer to the [DocsGPT Settings documentation](/Deploying/DocsGPT-Settings). This guide explains how to modify the `.env` file or `settings.py` for deeper configuration.
For more advanced customization of DocsGPT settings, such as configuring vector stores, embedding models, and other parameters, please refer to the [DocsGPT Settings documentation](/Deploying/DocsGPT-Settings). This guide explains how to configure DocsGPT through the `.env` file, and links to the full settings reference.
Enjoy using DocsGPT!
+11 -2
View File
@@ -11,6 +11,14 @@ import { Callout } from 'nextra/components'
**Upgrading from 0.16.x?** User data moved from MongoDB to Postgres in 0.17.0. Follow the [Postgres Migration guide](/Deploying/Postgres-Migration) before running `docker compose pull` or `git pull` — existing deployments will not start cleanly without it.
</Callout>
## pip installs: data home moved
An installed package (`pip install docsgpt`, pipx, `uv tool`) used to keep its data home, meaning `.env`, `inputs/`, `indexes/` and `models/`, in the directory you ran `docsgpt api` and `docsgpt worker` from. It is now `~/.docsgpt/server` (`/opt/docsgpt` for root on Linux). Either move those files there, or set `DOCSGPT_HOME` to the old directory in the environment of both commands. Both commands print the data home they use, and point out a `.env` in the working directory that they no longer read. Source checkouts and the Docker images are not affected.
## Standalone Compose file: one port
The standalone Compose file (`docker-compose-standalone.yaml`) no longer runs a frontend container. The backend image serves the web UI on port 7091, and the port is published on `127.0.0.1` unless you set `DOCSGPT_BIND`. After downloading the new file, start it with `--remove-orphans` to remove the old frontend container, then open port 7091 instead of 5173. If you opened DocsGPT from other machines, see [Upgrading from an earlier standalone file](/Deploying/Docker-Deploying#upgrading-from-an-earlier-standalone-file). The checkout Compose files and the Kubernetes manifests are unchanged.
## Embedding models
DocsGPT now runs embeddings through [FastEmbed](https://github.com/qdrant/fastembed) (ONNX Runtime) instead of SentenceTransformer. The models are the same and the vectors are identical, so **your existing index needs no action** — `all-mpnet-base-v2` keeps working exactly as before.
@@ -113,8 +121,9 @@ alias, so nothing breaks on upgrade, but update these before the alias goes:
and vectors under `application/` in the checkout, where they already are.
- The backend is also a package now (`pip install docsgpt`, see
[Install with pip](/Deploying/Pip-Install)). Runtime data lives in a data
home: `DOCSGPT_HOME`, else the checkout, else the directory the process
starts in. One consequence for a source checkout: the embedded Milvus
home: `DOCSGPT_HOME`, else the checkout, else `~/.docsgpt/server`
(`/opt/docsgpt` for root on Linux; see
[pip installs: data home moved](#pip-installs-data-home-moved)). One consequence for a source checkout: the embedded Milvus
(`MILVUS_URI`) and LanceDB (`LANCEDB_PATH`) default paths now resolve under
the checkout instead of the start directory. If you use either store at its
default path and start DocsGPT from another directory, the old data is at
+1 -1
View File
@@ -320,7 +320,7 @@ redis-cli -n 2 DEL user:<id>:stream
## Settings reference
Everything in `docsgpt/core/settings.py`:
Everything in `docsgpt/core/settings/events.py`:
| Setting | Default | Purpose |
| --------------------------------------------- | ------- | --------------------------------------------- |
+16 -2
View File
@@ -1,4 +1,4 @@
# DocsGPT backend image.
# DocsGPT image: the API, the web UI it serves, and the Celery worker.
#
# Build args:
# EXTRAS comma-separated optional extras to bake in, matching the
@@ -15,6 +15,19 @@
# docling's layout/table/OCR models. `python -m docsgpt.scripts.verify_offline`
# under `docker run --network none` proves it.
# The web UI, built by the same script the Python package build runs. The API
# serves it from docsgpt/static (docsgpt/ui.py). The output is static files, so
# this stage runs on the build machine's platform whatever the target is.
FROM --platform=$BUILDPLATFORM node:22-bookworm-slim AS ui
WORKDIR /src
COPY frontend/package.json frontend/package-lock.json frontend/
RUN cd frontend && npm ci --include=dev --no-audit --no-fund
COPY frontend/ frontend/
COPY scripts/build_frontend.sh scripts/
RUN mkdir docsgpt && bash scripts/build_frontend.sh
FROM ubuntu:24.04 AS builder
ENV DEBIAN_FRONTEND=noninteractive
@@ -87,7 +100,7 @@ RUN if [ "$INSTALL_TESSERACT" = "true" ]; then \
LABEL org.opencontainers.image.source="https://github.com/arc53/DocsGPT" \
org.opencontainers.image.title="DocsGPT" \
org.opencontainers.image.description="DocsGPT backend: API and Celery worker" \
org.opencontainers.image.description="DocsGPT: API, web UI and Celery worker" \
org.opencontainers.image.licenses="MIT"
WORKDIR /app
@@ -136,6 +149,7 @@ RUN if python -c "import docling" 2>/dev/null; then \
fi
COPY --chown=appuser:appuser docsgpt /app/docsgpt
COPY --from=ui --chown=appuser:appuser /src/docsgpt/static /app/docsgpt/static
# One-release alias so `-A application.app.celery` style entry points keep working.
COPY --chown=appuser:appuser application/__init__.py /app/application/__init__.py
+8
View File
@@ -0,0 +1,8 @@
"""``python -m docsgpt`` runs what the ``docsgpt`` command runs."""
import sys
from docsgpt.cli import main
if __name__ == "__main__":
sys.exit(main())
+2
View File
@@ -2,6 +2,7 @@ import logging
from typing import Dict, Generator, Optional
from docsgpt.agents.base import BaseAgent
from docsgpt.agents.tools.graph_search import add_graph_search_tool
from docsgpt.agents.tools.internal_search import add_internal_search_tool
from docsgpt.agents.tools.wiki import add_wiki_tool
from docsgpt.logging import LogContext
@@ -33,6 +34,7 @@ class AgenticAgent(BaseAgent):
) -> Generator[Dict, None, None]:
tools_dict = self.tool_executor.get_tools()
add_internal_search_tool(tools_dict, self.retriever_config)
add_graph_search_tool(tools_dict, self.retriever_config)
if self.wiki_config:
add_wiki_tool(tools_dict, self.wiki_config)
self._prepare_tools(tools_dict)
+23 -9
View File
@@ -382,7 +382,7 @@ class BaseAgent(ABC):
when the conversation was compressed after that turn was produced —
the compressed local history is the context then, not the server's.
"""
if not getattr(settings, "OPENAI_RESPONSES_CHAIN_ACROSS_TURNS", True):
if not settings.OPENAI_RESPONSES_CHAIN_ACROSS_TURNS:
return None
if not self.chat_history:
return None
@@ -422,7 +422,7 @@ class BaseAgent(ABC):
# No provider-reported usage on the previous turn (older rows,
# estimate-only providers): nothing to bound against.
return meta["response_id"]
budget = getattr(settings, "OPENAI_RESPONSES_CHAIN_BUDGET_TOKENS", None)
budget = settings.OPENAI_RESPONSES_CHAIN_BUDGET_TOKENS
if not budget:
from docsgpt.core.model_utils import get_token_limit
@@ -955,16 +955,30 @@ class BaseAgent(ABC):
)
self.retrieved_docs = scrubbed
def _collect_internal_sources(self) -> None:
"""Merge the cached InternalSearchTool's docs into ``retrieved_docs``,
deduped, preserving any pre-fetched docs so a mixed-exposure agent cites
both pre-fetched and tool-retrieved sources (not just the tool's)."""
def _search_tool_docs(self) -> List[Dict]:
"""Documents this run's search tools read: internal search and the graph tool.
Both record what they surface in ``retrieved_docs``; a page read from
the graph carries the answer as much as a search hit does, so both are
cited. Tools are looked up the way the executor caches them.
"""
from docsgpt.agents.tools.graph_search import GRAPH_TOOL_ID
from docsgpt.agents.tools.internal_search import INTERNAL_TOOL_ID
executor = getattr(self, "tool_executor", None)
loaded = getattr(executor, "_loaded_tools", None) or {}
tool = loaded.get(f"internal_search:{INTERNAL_TOOL_ID}:{self.user or ''}")
if not (tool and getattr(tool, "retrieved_docs", None)):
docs: List[Dict] = []
for name, tool_id in (("internal_search", INTERNAL_TOOL_ID), ("graph_search", GRAPH_TOOL_ID)):
tool = loaded.get(f"{name}:{tool_id}:{self.user or ''}")
docs.extend(getattr(tool, "retrieved_docs", None) or [])
return docs
def _collect_internal_sources(self) -> None:
"""Merge the search tools' docs into ``retrieved_docs``, deduped,
preserving any pre-fetched docs so a mixed-exposure agent cites both
pre-fetched and tool-retrieved sources (not just the tools')."""
tool_docs = self._search_tool_docs()
if not tool_docs:
return
def _key(d):
@@ -974,7 +988,7 @@ class BaseAgent(ABC):
merged = list(self.retrieved_docs or [])
seen = {_key(d) for d in merged}
for doc in tool.retrieved_docs:
for doc in tool_docs:
k = _key(doc)
if k not in seen:
seen.add(k)
+2
View File
@@ -2,6 +2,7 @@ import logging
from typing import Dict, Generator, Optional
from docsgpt.agents.base import BaseAgent
from docsgpt.agents.tools.graph_search import add_graph_search_tool
from docsgpt.agents.tools.internal_search import add_internal_search_tool
from docsgpt.agents.tools.wiki import add_wiki_tool
from docsgpt.logging import LogContext
@@ -38,6 +39,7 @@ class ClassicAgent(BaseAgent):
tools_dict = self.tool_executor.get_tools()
if self.retriever_config:
add_internal_search_tool(tools_dict, self.retriever_config)
add_graph_search_tool(tools_dict, self.retriever_config)
if self.wiki_config:
add_wiki_tool(tools_dict, self.wiki_config)
self._prepare_tools(tools_dict)
+11 -1
View File
@@ -15,6 +15,7 @@ from docsgpt.api.answer.services.prompt_renderer import (
)
from docsgpt.api.answer.services.stream_processor import get_prompt
from docsgpt.core.settings import settings
from docsgpt.quotas.service import QuotaExceededError, QuotaService
from docsgpt.retriever.retriever_creator import RetrieverCreator
from docsgpt.storage.db.repositories.sources import SourcesRepository
from docsgpt.storage.db.session import db_readonly
@@ -69,7 +70,11 @@ def run_agent_headless(
chat_history: Optional[List[Dict[str, Any]]] = None,
conversation_id: Optional[str] = None,
) -> Dict[str, Any]:
"""Run an agent with no live client; returns a structured outcome dict."""
"""Run an agent with no live client; returns a structured outcome dict.
Raises:
QuotaExceededError: If the agent owner's usage quota is exhausted.
"""
from docsgpt.core.model_utils import (
get_api_key_for_provider,
get_default_model_id,
@@ -82,6 +87,11 @@ def run_agent_headless(
if not owner:
raise ValueError("Agent config is missing user_id; cannot run headless.")
decoded_token = {"sub": owner}
# An agent run is agent traffic whether or not the agent has a key yet.
is_agent_run = bool(agent_config.get("key") or _resolve_agent_id(agent_config))
exceeded = QuotaService.check(owner, "agent" if is_agent_run else "direct")
if exceeded is not None:
raise QuotaExceededError(exceeded)
retriever_kind = agent_config.get("retriever", "classic")
source_id = agent_config.get("source_id") or agent_config.get("source")
+6 -10
View File
@@ -6,10 +6,8 @@ from typing import Dict, Generator, List, Optional
from docsgpt.agents.base import BaseAgent
from docsgpt.agents.tool_executor import ToolExecutor
from docsgpt.agents.tools.internal_search import (
INTERNAL_TOOL_ID,
add_internal_search_tool,
)
from docsgpt.agents.tools.graph_search import add_graph_search_tool
from docsgpt.agents.tools.internal_search import add_internal_search_tool
from docsgpt.agents.tools.wiki import add_wiki_tool
from docsgpt.agents.tools.think import THINK_TOOL_ENTRY, THINK_TOOL_ID
from docsgpt.logging import LogContext
@@ -277,6 +275,7 @@ class ResearchAgent(BaseAgent):
tools_dict = self.tool_executor.get_tools()
add_internal_search_tool(tools_dict, self.retriever_config)
add_graph_search_tool(tools_dict, self.retriever_config)
if self.wiki_config:
add_wiki_tool(tools_dict, self.wiki_config)
@@ -620,12 +619,9 @@ class ResearchAgent(BaseAgent):
return messages, search_returned_empty
def _collect_step_sources(self):
"""Collect sources from InternalSearchTool and register with CitationManager."""
cache_key = f"internal_search:{INTERNAL_TOOL_ID}:{self.user or ''}"
tool = self.tool_executor._loaded_tools.get(cache_key)
if tool and hasattr(tool, "retrieved_docs"):
for doc in tool.retrieved_docs:
self.citations.add(doc)
"""Register the search tools' docs (internal search and graph pages) with CitationManager."""
for doc in self._search_tool_docs():
self.citations.add(doc)
# ------------------------------------------------------------------
# Phase 3: Synthesis
+1 -1
View File
@@ -53,7 +53,7 @@ def _dedupable_tool_names() -> frozenset:
"""
from docsgpt.core.settings import settings
return frozenset(BUILTIN_AGENT_TOOLS) | frozenset(getattr(settings, "DEFAULT_CHAT_TOOLS", None) or [])
return frozenset(BUILTIN_AGENT_TOOLS) | frozenset(settings.DEFAULT_CHAT_TOOLS or [])
def _requires_approval(tool: Dict, action: Dict) -> bool:
+1 -1
View File
@@ -829,7 +829,7 @@ class ArtifactGeneratorTool(Tool):
spec_path = f"{token_dir}/spec.json"
out_path = f"{token_dir}/out.{_KIND_INFO[kind]['ext']}"
program = _RENDERERS[kind].format(spec_path=spec_path, out_path=out_path)
timeout = float(getattr(settings, "SANDBOX_EXEC_TIMEOUT", 60))
timeout = float(settings.SANDBOX_EXEC_TIMEOUT)
manager = SandboxCreator.get_manager()
try:
+1 -1
View File
@@ -113,7 +113,7 @@ def bridge_attachment(
# Reject oversize attachments BEFORE buffering them: the authoritative ``size``
# column lets us avoid pulling a multi-hundred-MB file fully into worker memory,
# and the bounded read below backstops a missing/lying ``size``.
max_bytes = int(getattr(settings, "ARTIFACT_MAX_BYTES", 0) or 0)
max_bytes = int(settings.ARTIFACT_MAX_BYTES or 0)
declared_size = attachment.get("size")
if max_bytes and isinstance(declared_size, (int, float)) and declared_size > max_bytes:
raise AttachmentBridgeError(
+4 -4
View File
@@ -87,9 +87,9 @@ class CodeExecutorTool(Tool):
baked in. Keep the package lists in sync with deployment/sandbox/Dockerfile
(jupyter) and scripts/build_daytona_snapshot.py (daytona snapshot).
"""
backend = str(getattr(settings, "SANDBOX_BACKEND", "jupyter") or "jupyter").lower()
backend = str(settings.SANDBOX_BACKEND or "jupyter").lower()
if backend == "daytona":
if getattr(settings, "DAYTONA_SNAPSHOT", None):
if settings.DAYTONA_SNAPSHOT:
return (
"Preinstalled beyond the stdlib: python-pptx, python-docx, openpyxl, "
"reportlab, lxml, pillow. pip install anything else from within the code "
@@ -330,7 +330,7 @@ class CodeExecutorTool(Tool):
# Reject an oversize input BEFORE buffering it: the declared ``size``
# avoids pulling a huge file into worker memory, and the bounded read
# below backstops a missing/lying size column.
max_bytes = int(getattr(settings, "SANDBOX_MAX_INPUT_BYTES", 0) or 0)
max_bytes = int(settings.SANDBOX_MAX_INPUT_BYTES or 0)
declared_size = version.get("size")
if max_bytes and isinstance(declared_size, (int, float)) and declared_size > max_bytes:
return {"error": f"input artifact {artifact_id} exceeds the {max_bytes}-byte sandbox input limit."}
@@ -484,7 +484,7 @@ class CodeExecutorTool(Tool):
@staticmethod
def _exec_timeout() -> float:
"""Return the fixed per-run wall-clock cap (SANDBOX_EXEC_TIMEOUT; not caller-adjustable)."""
return float(getattr(settings, "SANDBOX_EXEC_TIMEOUT", 60))
return float(settings.SANDBOX_EXEC_TIMEOUT)
@staticmethod
def _is_timeout(result: ExecResult) -> bool:
+299
View File
@@ -0,0 +1,299 @@
"""Let the model search the knowledge graph itself, one edge at a time.
Graph retrieval normally runs as a ranker: seed a walk from the question,
diffuse mass over a subgraph, hand back the highest-scoring chunks. Measured
across five corpora that never beat plain vector search, because a question
whose answer lives two documents away has nothing in it for the seeding step to
match — the bridging entity is named in the *first* document, not the question.
Exposing the graph as tools removes the guess. The model can look up the
service, read which store it names, then fetch that store's page: the chain
followed deliberately rather than approximated by a diffusion. On a corpus built
so that vector search cannot shortcut the chain, this took two-hop answers from
1/8 to 8/8, against 0.40 for vector and 0.47 for one-shot graph retrieval.
It is not a general win, and is deliberately not a default. On ordinary prose
documentation it *lost* to plain vector search (0.50 against 0.90): it answers
well when a question names an entity and wanders when the question is a task
description. It also costs several model round-trips per answer instead of one.
So it is offered only where a source owner has already chosen search over
prefetch — the per-source exposure setting, or an agentic/research agent — and
suits content that is genuinely chain-structured: runbooks, service catalogues,
infrastructure inventories.
"""
from __future__ import annotations
import logging
from typing import Any, Dict, List, Optional
from docsgpt.agents.tools.base import Tool
from docsgpt.graphrag import graphrag_available
from docsgpt.retriever.labels import labels_from_metadata
logger = logging.getLogger(__name__)
GRAPH_TOOL_ID = "graph_search"
MAX_PAGE_CHARS = 1500
class GraphSearchTool(Tool):
"""Entity lookup, relationship traversal and page reads over a source's graph."""
internal = True
def __init__(self, config: Dict):
self.config = config or {}
self._store = None
self.retrieved_docs: List[Dict] = []
# -- plumbing ------------------------------------------------------------
def _sources(self) -> List[str]:
source = self.config.get("source") or {}
active = source.get("active_docs") or []
if isinstance(active, str):
active = [active]
return [str(s) for s in active if s]
def _get_store(self):
if self._store is None:
from docsgpt.graphrag.store import GraphStore
self._store = GraphStore()
return self._store
def _release_store(self) -> None:
"""Hand the pooled connection back at the end of an action.
The executor caches this tool for the whole agent run, so a store kept
between actions pins one connection of the shared pgvector pool across
every LLM round trip of that run -- minutes at a time, and enough
concurrent runs exhaust the pool. ``GraphRAGRetriever`` releases its
store before falling back for the same reason. Checking one back out
costs a pool acquire.
"""
store, self._store = self._store, None
if store is None:
return
try:
store.close()
except Exception as exc: # noqa: BLE001 -- releasing must not fail an action
logger.debug(f"Graph tool could not release its store: {exc}")
def _embed(self, text: str) -> Optional[List[float]]:
try:
from docsgpt.vectorstore.base import get_embeddings
return get_embeddings().embed_query(text)
except Exception as e: # noqa: BLE001
logger.error(f"Graph tool could not embed the query: {e}")
return None
# -- actions -------------------------------------------------------------
def execute_action(self, action_name: str, **kwargs):
# The graph lives in the pgvector store, so the flag alone is not
# enough: under another vector store there is no graph to read.
if not graphrag_available():
return "The knowledge graph is not enabled for this deployment."
if not self._sources():
return "No graph-backed sources are configured."
try:
if action_name == "search_entities":
return self._search_entities(**kwargs)
if action_name == "get_relationships":
return self._get_relationships(**kwargs)
if action_name == "read_entity_pages":
return self._read_entity_pages(**kwargs)
except Exception as e: # noqa: BLE001
logger.error(f"Graph tool action {action_name} failed: {e}", exc_info=True)
return "The graph lookup failed."
finally:
self._release_store()
return f"Unknown action: {action_name}"
def _search_entities(self, **kwargs) -> str:
query = str(kwargs.get("query") or "").strip()
if not query:
return "Error: 'query' parameter is required."
limit = max(1, min(int(kwargs.get("k") or 8), 25))
embedding = self._embed(query)
if embedding is None:
return "Entity search is unavailable."
store = self._get_store()
lines: List[str] = []
for source_id in self._sources():
for row in store.search_nodes_by_embedding(source_id, embedding, k=limit):
similarity = 1.0 - float(row.get("distance") or 0.0)
description = (row.get("description") or "").strip()
suffix = f" — {description[:160]}" if description else ""
lines.append(f"- {row['name']} (match {similarity:.2f}){suffix}")
if not lines:
return f"No entities found for {query!r}."
return "Entities:\n" + "\n".join(lines[:limit])
def _get_relationships(self, **kwargs) -> str:
entity = str(kwargs.get("entity") or "").strip()
if not entity:
return "Error: 'entity' parameter is required."
store = self._get_store()
lines: List[str] = []
for source_id in self._sources():
for edge in store.entity_relationships(source_id, entity):
relation = edge.get("type") or "related to"
lines.append(f"- {edge['source']} --{relation}--> {edge['target']}")
if not lines:
return (
f"No relationships found for {entity!r}. Try search_entities first "
"to get the exact name used in the graph."
)
return f"Relationships for {entity!r}:\n" + "\n".join(lines)
def _read_entity_pages(self, **kwargs) -> str:
entity = str(kwargs.get("entity") or "").strip()
if not entity:
return "Error: 'entity' parameter is required."
store = self._get_store()
parts: List[str] = []
for source_id in self._sources():
for page in store.entity_pages(source_id, entity):
text = page.get("text") or ""
# The retrievers' own labelling: a page read here and the same
# chunk retrieved by internal_search are one document, and
# citations key on (source, title). Labelling it differently
# gives that document two citation numbers.
labels = labels_from_metadata(page.get("metadata"), text, source_id)
doc = {**labels, "text": text}
if doc not in self.retrieved_docs:
self.retrieved_docs.append(doc)
header = labels["filename"] or labels["title"]
parts.append(f"--- {header} ---\n{text[:MAX_PAGE_CHARS]}")
if not parts:
return f"No documents mention {entity!r}."
return "\n\n".join(parts)
# -- metadata ------------------------------------------------------------
def get_actions_metadata(self):
return [
{
"name": "search_entities",
"description": (
"Find named things in the knowledge graph — services, components, "
"settings, people — whose names resemble a query. Use this first to "
"learn the exact name the graph uses before asking for its "
"relationships."
),
"parameters": {
"properties": {
"query": {
"type": "string",
"description": "What to look for, e.g. a service or component name.",
"filled_by_llm": True,
"required": True,
},
"k": {
"type": "integer",
"description": "How many entities to return (default 8).",
"filled_by_llm": True,
"required": False,
},
}
},
},
{
"name": "get_relationships",
"description": (
"List what an entity is connected to, as 'source --relation--> target'. "
"This is how you answer a question about something the question does not "
"name: look up what it points at, then read that thing's pages."
),
"parameters": {
"properties": {
"entity": {
"type": "string",
"description": "Exact entity name, as returned by search_entities.",
"filled_by_llm": True,
"required": True,
}
}
},
},
{
"name": "read_entity_pages",
"description": (
"Read the documentation an entity appears in, the page it is about "
"first. Use this once you know which entity holds the answer."
),
"parameters": {
"properties": {
"entity": {
"type": "string",
"description": "Exact entity name, as returned by search_entities.",
"filled_by_llm": True,
"required": True,
}
}
},
},
]
def get_config_requirements(self):
return {}
def build_graph_tool_entry() -> Dict:
"""The synthetic ``tools_dict`` entry for the graph tool."""
tool = GraphSearchTool({})
actions = []
for action in tool.get_actions_metadata():
entry = dict(action)
entry["active"] = True
actions.append(entry)
return {"name": "graph_search", "actions": actions}
def sources_have_graph(source: Dict) -> bool:
"""Whether any active source actually has a graph to search."""
active = source.get("active_docs") or []
if isinstance(active, str):
active = [active]
if not active:
return False
try:
from docsgpt.graphrag.store import GraphStore
counts = GraphStore().count_nodes_many([str(a) for a in active])
return any(count > 0 for count in counts.values())
except Exception as e: # noqa: BLE001
logger.debug(f"Could not check for graphs: {e}")
return False
def add_graph_search_tool(tools_dict: Dict, retriever_config: Dict) -> None:
"""Add the graph tool when the agent's search-tool sources include a graph.
No setting of its own: ``retriever_config`` already carries exactly the
sources the agent may *search* — the ones a source owner exposed as a
search tool, or every source for an agentic/research agent — so the graph
tool follows that same per-source exposure choice. A graph source left at
``prefetch`` in a classic agent is used for ranking only.
"""
if not graphrag_available():
return
source = retriever_config.get("source") or {}
if not source.get("active_docs") or not sources_have_graph(source):
return
entry = build_graph_tool_entry()
# The executor resolves tools by ``id``; this one is synthetic (no DB row).
entry["id"] = GRAPH_TOOL_ID
entry["config"] = {"source": source}
tools_dict[GRAPH_TOOL_ID] = entry
def build_graph_tool_config(source: Dict, **_ignored: Any) -> Dict:
"""Config for :class:`GraphSearchTool` — it only needs the source ids."""
return {"source": source}
+2 -2
View File
@@ -108,11 +108,11 @@ class MCPTool(Tool):
if configured_redirect_uri:
return configured_redirect_uri.rstrip("/")
explicit = getattr(settings, "MCP_OAUTH_REDIRECT_URI", None)
explicit = settings.MCP_OAUTH_REDIRECT_URI
if explicit:
return explicit.rstrip("/")
connector_base = getattr(settings, "CONNECTOR_REDIRECT_BASE_URI", None)
connector_base = settings.CONNECTOR_REDIRECT_BASE_URI
if connector_base:
parsed = urlparse(connector_base)
if parsed.scheme and parsed.netloc:
+5 -6
View File
@@ -17,8 +17,6 @@ import signal
import threading
from typing import Any, Callable, Dict, List, Optional
from celery import current_task
from docsgpt.agents.tools.artifact_ref import resolve_artifact_id
from docsgpt.agents.tools.attachment_bridge import (
AttachmentBridgeError,
@@ -26,6 +24,7 @@ from docsgpt.agents.tools.attachment_bridge import (
match_attachment,
)
from docsgpt.agents.tools.base import Tool
from docsgpt.celery_init import in_worker
from docsgpt.core.json_schema_utils import (
JsonSchemaValidationError,
normalize_json_schema_payload,
@@ -229,9 +228,9 @@ class ReadDocumentTool(Tool):
# (floored at DOCUMENT_PARSE_TIMEOUT).
timeout = parse_timeout_for_size(self._input_size)
# ``current_task`` is a Celery proxy: truthy only while this runs inside a worker task,
# falsy in the web process (the bare proxy is NOT identity-None, so test truthiness).
if current_task:
# Process-wide, not the thread-local ``current_task``: a thread a task starts has no
# task of its own, and dispatching from there is the self-deadlock described above.
if in_worker():
from docsgpt.worker import run_parse_document
try:
@@ -255,7 +254,7 @@ class ReadDocumentTool(Tool):
# The task's per-call time limits are raised to match the awaited window: bound to
# the base timeout at import, the worker would otherwise self-terminate a large
# parse long before this await gives up.
queue = getattr(settings, "DOCUMENT_PARSE_QUEUE", "parsing")
queue = settings.DOCUMENT_PARSE_QUEUE
try:
async_result = parse_document.apply_async(
args=[artifact_id, parent, self.user_id, options],
+1 -1
View File
@@ -331,7 +331,7 @@ class WorkflowAgent(BaseAgent):
from docsgpt.storage.storage_creator import StorageCreator
storage = StorageCreator.get_storage()
max_bytes = int(getattr(settings, "ARTIFACT_MAX_BYTES", 0) or 0)
max_bytes = int(settings.ARTIFACT_MAX_BYTES or 0)
dropped: List[str] = []
if len(self.attachments) > _MAX_INPUT_DOCUMENTS:
over = len(self.attachments) - _MAX_INPUT_DOCUMENTS
+9 -7
View File
@@ -393,6 +393,8 @@ class WorkflowEngine:
"prompt": node_prompt,
"chat_history": self.agent.chat_history,
"decoded_token": self.agent.decoded_token,
# Attributes the node's token usage to the workflow agent.
"agent_id": getattr(self.agent, "agent_id", None),
"json_schema": node_json_schema,
"retrieved_docs": node_docs,
# A template that interpolates the documents itself already carries
@@ -639,7 +641,7 @@ class WorkflowEngine:
raw_ids = self._resolve_input_artifact_ids(inputs)
if not raw_ids:
return loaded
max_bytes = int(getattr(settings, "SANDBOX_MAX_INPUT_BYTES", 0) or 0)
max_bytes = int(settings.SANDBOX_MAX_INPUT_BYTES or 0)
storage = StorageCreator.get_storage()
# Two inputs whose current versions share a filename would clobber each other at the
# same ``inputs/{name}`` path; track used paths and disambiguate deterministically.
@@ -749,15 +751,15 @@ class WorkflowEngine:
supported = set(supported_types)
supports_images = any(t.startswith("image/") for t in supported)
max_files = int(getattr(settings, "WORKFLOW_NODE_NATIVE_MAX_FILES", 5))
extract_max = int(getattr(settings, "WORKFLOW_NODE_EXTRACT_MAX_FILES", 5))
max_files = int(settings.WORKFLOW_NODE_NATIVE_MAX_FILES)
extract_max = int(settings.WORKFLOW_NODE_EXTRACT_MAX_FILES)
# One wall clock for every blocking parse this node issues. The cap
# above bounds how MANY parses run; this bounds how LONG they take in
# total, so N documents cannot serialize N size-scaled windows.
parse_deadline = time.monotonic() + float(
getattr(settings, "WORKFLOW_NODE_EXTRACT_BUDGET_SECONDS", 900)
settings.WORKFLOW_NODE_EXTRACT_BUDGET_SECONDS
)
max_bytes = int(getattr(settings, "SANDBOX_MAX_INPUT_BYTES", 25 * 1024 * 1024))
max_bytes = int(settings.SANDBOX_MAX_INPUT_BYTES)
# One read-only connection for the whole batch; the resolved-version
# rows are collected, then storage reads happen outside the DB context.
@@ -976,7 +978,7 @@ class WorkflowEngine:
if not user_id:
return None
options = {"output": "markdown", "include_tables": False, "persist": False}
queue = getattr(settings, "DOCUMENT_PARSE_QUEUE", "parsing")
queue = settings.DOCUMENT_PARSE_QUEUE
# OCR cost scales with pages, so the window grows with the document's size
# (floored at DOCUMENT_PARSE_TIMEOUT); the task's per-call time limits are
# raised to match, else the worker would self-terminate mid-parse.
@@ -1084,7 +1086,7 @@ class WorkflowEngine:
"""Return the stricter of the node's requested timeout and the sandbox cap."""
from docsgpt.core.settings import settings
cap = float(getattr(settings, "SANDBOX_EXEC_TIMEOUT", 60))
cap = float(settings.SANDBOX_EXEC_TIMEOUT)
if requested is None:
return cap
try:
@@ -0,0 +1,74 @@
"""0032 personal access tokens — scoped, user-level API credentials.
A personal access token (PAT) authenticates its owner against the management
API for CLI and CI/CD use. Only the SHA-256 of the secret is stored, mirroring
``devices.token_hash``: the plaintext is shown once at creation and a database
leak cannot reconstruct it. ``token_prefix`` keeps the first characters so a
user can tell their tokens apart in the UI.
``scopes`` is the server-side grant list (never read from the credential
itself). ``resource_filter`` optionally narrows a resource family to specific
ids, e.g. ``{"agents": ["<uuid>"]}``; an absent family is unrestricted within
the token's scopes. ``expires_at`` is NULL only when the operator allows
non-expiring tokens. Regenerating a token swaps its secret in place and stamps
``regenerated_at``; the row, its name, scopes and restrictions stay.
``user_id`` is the auth ``sub``; no FK or trigger, mirroring ``devices`` and
``user_roles`` so a token row never blocks user deletion.
Revision ID: 0032_personal_access_tokens
Revises: 0031_token_usage_cache_tokens
"""
from typing import Sequence, Union
from alembic import op
revision: str = "0032_personal_access_tokens"
down_revision: Union[str, None] = "0031_token_usage_cache_tokens"
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
op.execute(
"""
CREATE TABLE IF NOT EXISTS personal_access_tokens (
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
user_id TEXT NOT NULL,
name TEXT NOT NULL,
token_hash TEXT NOT NULL,
token_prefix TEXT NOT NULL,
scopes TEXT[] NOT NULL DEFAULT '{}',
resource_filter JSONB NOT NULL DEFAULT '{}'::jsonb,
status TEXT NOT NULL DEFAULT 'active'
CHECK (status IN ('active', 'revoked')),
expires_at TIMESTAMPTZ,
last_used_at TIMESTAMPTZ,
last_used_ip TEXT,
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
regenerated_at TIMESTAMPTZ,
revoked_at TIMESTAMPTZ,
revoke_reason TEXT
);
"""
)
# Looked up on every PAT-authenticated request.
op.execute(
"CREATE UNIQUE INDEX IF NOT EXISTS personal_access_tokens_hash_uidx "
"ON personal_access_tokens(token_hash);"
)
# Names are unique among a user's live tokens; a revoked name can be reused.
op.execute(
"CREATE UNIQUE INDEX IF NOT EXISTS personal_access_tokens_user_name_uidx "
"ON personal_access_tokens(user_id, name) WHERE status = 'active';"
)
op.execute(
"CREATE INDEX IF NOT EXISTS personal_access_tokens_user_idx "
"ON personal_access_tokens(user_id, created_at DESC);"
)
def downgrade() -> None:
op.execute("DROP TABLE IF EXISTS personal_access_tokens;")
+102
View File
@@ -0,0 +1,102 @@
"""0033 quotas — admin-set usage limits and a per-call cost.
``quota_policies`` holds the limits an instance admin sets at three layers:
the instance default (``subject_id`` NULL), a team's per-member allowance
(``subject_id`` = ``teams.id``) and a single user's override (``subject_id`` =
the auth ``sub``). Each row carries a token budget and a USD budget; per
budget a row either sets a limit (0 blocks), marks it unlimited, or leaves
both empty to defer to the next layer. ``bucket`` narrows a row to chat without
an agent (``direct``) or traffic through an agent (``agent``); ``all`` covers both.
``subject_id`` is polymorphic, so there is no FK: an AFTER DELETE trigger on
``teams`` scrubs a deleted team's rows, and user rows follow the ``user_roles``
convention of never blocking user deletion.
``token_usage.cost`` is the USD cost of the call at write time (see
``docsgpt/pricing.py``); 0 for unpriced and bring-your-own models.
Idempotent both ways.
Revision ID: 0033_quotas
Revises: 0032_personal_access_tokens
"""
from typing import Sequence, Union
from alembic import op
revision: str = "0033_quotas"
down_revision: Union[str, None] = "0032_personal_access_tokens"
branch_labels: Union[str, Sequence[str], None] = None
depends_on: Union[str, Sequence[str], None] = None
def upgrade() -> None:
op.execute(
"ALTER TABLE token_usage ADD COLUMN IF NOT EXISTS cost NUMERIC(12,8) NOT NULL DEFAULT 0;"
)
op.execute(
"""
CREATE TABLE IF NOT EXISTS quota_policies (
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
scope TEXT NOT NULL CHECK (scope IN ('instance', 'team', 'user')),
subject_id TEXT,
bucket TEXT NOT NULL DEFAULT 'all'
CHECK (bucket IN ('all', 'direct', 'agent')),
token_limit BIGINT CHECK (token_limit >= 0),
token_unlimited BOOLEAN NOT NULL DEFAULT false,
cost_limit_usd NUMERIC(12,4) CHECK (cost_limit_usd >= 0),
cost_unlimited BOOLEAN NOT NULL DEFAULT false,
enabled BOOLEAN NOT NULL DEFAULT true,
note TEXT,
created_by TEXT,
updated_by TEXT,
created_at TIMESTAMPTZ NOT NULL DEFAULT now(),
updated_at TIMESTAMPTZ NOT NULL DEFAULT now(),
CONSTRAINT quota_policies_subject_chk
CHECK ((scope = 'instance') = (subject_id IS NULL)),
CONSTRAINT quota_policies_token_chk
CHECK (NOT (token_unlimited AND token_limit IS NOT NULL)),
CONSTRAINT quota_policies_cost_chk
CHECK (NOT (cost_unlimited AND cost_limit_usd IS NOT NULL))
);
"""
)
# One row per (layer subject, bucket); the instance row's NULL subject folds to ''.
op.execute(
"CREATE UNIQUE INDEX IF NOT EXISTS quota_policies_subject_uidx "
"ON quota_policies (scope, COALESCE(subject_id, ''), bucket);"
)
op.execute("DROP TRIGGER IF EXISTS quota_policies_set_updated_at ON quota_policies;")
op.execute(
"""
CREATE TRIGGER quota_policies_set_updated_at
BEFORE UPDATE ON quota_policies
FOR EACH ROW EXECUTE FUNCTION set_updated_at();
"""
)
op.execute(
"""
CREATE OR REPLACE FUNCTION cleanup_team_quota_policies() RETURNS trigger AS $$
BEGIN
DELETE FROM quota_policies WHERE scope = 'team' AND subject_id = OLD.id::text;
RETURN OLD;
END;
$$ LANGUAGE plpgsql;
"""
)
op.execute("DROP TRIGGER IF EXISTS teams_cleanup_quota_policies ON teams;")
op.execute(
"""
CREATE TRIGGER teams_cleanup_quota_policies
AFTER DELETE ON teams
FOR EACH ROW EXECUTE FUNCTION cleanup_team_quota_policies();
"""
)
def downgrade() -> None:
op.execute("DROP TRIGGER IF EXISTS teams_cleanup_quota_policies ON teams;")
op.execute("DROP FUNCTION IF EXISTS cleanup_team_quota_policies();")
op.execute("DROP TABLE IF EXISTS quota_policies;")
op.execute("ALTER TABLE token_usage DROP COLUMN IF EXISTS cost;")
+1
View File
@@ -1,3 +1,4 @@
from .routes import admin_ns
from . import quotas # noqa: F401 (registers the quota resources on admin_ns)
__all__ = ["admin_ns"]
+273
View File
@@ -0,0 +1,273 @@
"""Admin endpoints for usage quotas (RBAC ``admin`` role required).
Policies are set at three layers: the instance default, a team's per-member
allowance and a single user's override. Every write is audited to
``auth_events`` with the acting admin recorded.
"""
from __future__ import annotations
import math
from typing import Any, Optional
from flask import jsonify, make_response, request
from flask_restx import Resource
from docsgpt.api.admin.routes import _actor, admin_ns
from docsgpt.api.user.authz import admin_required
from docsgpt.core.settings import settings
from docsgpt.pricing import is_priced
from docsgpt.quotas.service import REQUEST_BUCKETS, QuotaService
from docsgpt.quotas.windows import window_bounds
from docsgpt.storage.db.base_repository import looks_like_uuid
from docsgpt.storage.db.repositories.auth_events import AuthEventsRepository
from docsgpt.storage.db.repositories.quota_policies import BUCKETS, QuotaPoliciesRepository
from docsgpt.storage.db.repositories.teams import TeamsRepository
from docsgpt.storage.db.repositories.token_usage import TokenUsageRepository
from docsgpt.storage.db.repositories.users import UsersRepository
from docsgpt.storage.db.session import db_readonly, db_session
_MAX_TOKEN_LIMIT = 2**62
_MAX_COST_LIMIT = 99_999_999.0
_MAX_NOTE_LENGTH = 500
_FLAG_DEFAULTS = {"token_unlimited": False, "cost_unlimited": False, "enabled": True}
_BUCKET_MESSAGE = f"bucket must be one of: {', '.join(BUCKETS)}"
def _policy_json(row: dict) -> dict:
cost = row.get("cost_limit_usd")
return {
"scope": row["scope"],
"subject_id": row.get("subject_id"),
"bucket": row["bucket"],
"token_limit": row.get("token_limit"),
"token_unlimited": bool(row.get("token_unlimited")),
"cost_limit_usd": float(cost) if cost is not None else None,
"cost_unlimited": bool(row.get("cost_unlimited")),
"enabled": bool(row.get("enabled", True)),
"note": row.get("note"),
"updated_by": row.get("updated_by"),
"updated_at": row.get("updated_at"),
}
def _error(message: str, status: int):
return make_response(jsonify({"success": False, "message": message}), status)
def _limit_error(value: Any, name: str, whole: bool, maximum: float) -> Optional[str]:
"""Return why ``value`` is not a valid limit, or ``None``."""
if value is None:
return None
number = (int,) if whole else (int, float)
if isinstance(value, bool) or not isinstance(value, number):
return f"{name} must be a {'whole number' if whole else 'number'} or null"
# Range first: ``isfinite`` overflows on an int too large for a float.
if not 0 <= value <= maximum or not math.isfinite(value):
return f"{name} is out of range"
return None
def _parse_policy(data: Any) -> tuple[Optional[dict], Optional[str]]:
"""Validate a policy body.
Returns:
``(fields, None)`` with ``QuotaPoliciesRepository.upsert`` kwargs, or
``(None, message)`` describing the first problem.
"""
if not isinstance(data, dict):
return None, "Body must be a JSON object"
token_limit, cost_limit = data.get("token_limit"), data.get("cost_limit_usd")
problem = _limit_error(token_limit, "token_limit", True, _MAX_TOKEN_LIMIT) or _limit_error(
cost_limit, "cost_limit_usd", False, _MAX_COST_LIMIT
)
if problem:
return None, problem
flags = {key: data.get(key, default) for key, default in _FLAG_DEFAULTS.items()}
for key, value in flags.items():
if not isinstance(value, bool):
return None, f"{key} must be a boolean"
bucket = data.get("bucket", "all")
if bucket not in BUCKETS:
return None, _BUCKET_MESSAGE
note = data.get("note")
if note is not None and not isinstance(note, str):
return None, "note must be a string"
if flags["token_unlimited"] and token_limit is not None:
return None, "Set token_limit or token_unlimited, not both"
if flags["cost_unlimited"] and cost_limit is not None:
return None, "Set cost_limit_usd or cost_unlimited, not both"
if token_limit is None and cost_limit is None and not flags["token_unlimited"] and not flags["cost_unlimited"]:
return None, "Set a limit or mark a budget unlimited; delete the policy to remove it"
return {
"bucket": bucket,
"token_limit": token_limit,
"token_unlimited": flags["token_unlimited"],
"cost_limit_usd": round(float(cost_limit), 4) if cost_limit is not None else None,
"cost_unlimited": flags["cost_unlimited"],
"enabled": flags["enabled"],
"note": (note.strip()[:_MAX_NOTE_LENGTH] or None) if note else None,
}, None
def _audit(conn, event: str, scope: str, subject_id: Optional[str], detail: dict) -> None:
actor = _actor()
AuthEventsRepository(conn).insert(
# A user policy is filed under that user; the rest under the acting admin.
subject_id if scope == "user" else (actor or "unknown"),
event,
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={"by": actor, "via": "admin_api", "scope": scope, "subject_id": subject_id, **detail},
)
def _put_policy(scope: str, subject_id: Optional[str]):
fields, problem = _parse_policy(request.get_json(silent=True))
if fields is None:
return _error(problem or "Invalid policy", 400)
with db_session() as conn:
row = QuotaPoliciesRepository(conn).upsert(
scope=scope, subject_id=subject_id, actor=_actor(), **fields
)
_audit(conn, "quota_policy_set", scope, subject_id, fields)
return make_response(jsonify({"success": True, "policy": _policy_json(row)}), 200)
def _delete_policy(scope: str, subject_id: Optional[str]):
bucket = request.args.get("bucket")
if bucket is not None and bucket not in BUCKETS:
return _error(_BUCKET_MESSAGE, 400)
with db_session() as conn:
deleted = QuotaPoliciesRepository(conn).delete(scope, subject_id, bucket)
if deleted:
_audit(conn, "quota_policy_deleted", scope, subject_id, {"bucket": bucket or "*"})
return make_response(jsonify({"success": True, "deleted": deleted}), 200)
def _unpriced_models(conn) -> list[dict]:
"""Models used this period whose calls were all recorded at $0 for want of a price."""
start, _ = window_bounds(settings.QUOTA_PERIOD)
return [
row
for row in TokenUsageRepository(conn).tokens_by_model(start=start)
# Judged by what was recorded, so a priced model whose provider has since
# been disabled is not listed. BYOM ids are UUIDs and $0 by design; a
# model explicitly priced at $0 is free, not unpriced.
if row["cost"] == 0 and not looks_like_uuid(row["model_id"]) and not is_priced(row["model_id"])
]
@admin_ns.route("/admin/quotas")
class AdminQuotasResource(Resource):
@admin_required
def get(self):
"""Every stored policy, grouped by layer, plus the models cost limits cannot see."""
start, resets_at = window_bounds(settings.QUOTA_PERIOD)
with db_readonly() as conn:
repo = QuotaPoliciesRepository(conn)
teams = {str(t["id"]): t for t in TeamsRepository(conn).list_all()}
team_policies = []
for row in repo.list_by_scope("team"):
team = teams.get(str(row["subject_id"]), {})
team_policies.append(
{
**_policy_json(row),
"team_name": team.get("name"),
"team_slug": team.get("slug"),
"member_count": team.get("member_count"),
}
)
body = {
"success": True,
"period": settings.QUOTA_PERIOD,
"period_start": start.isoformat(),
"resets_at": resets_at.isoformat(),
"instance": [_policy_json(r) for r in repo.list_by_scope("instance")],
"teams": team_policies,
"users": [_policy_json(r) for r in repo.list_by_scope("user")],
"unpriced_models": _unpriced_models(conn),
}
return make_response(jsonify(body), 200)
@admin_ns.route("/admin/quotas/instance")
class AdminInstanceQuotaResource(Resource):
@admin_required
def put(self):
"""Set the instance default for one bucket."""
return _put_policy("instance", None)
@admin_required
def delete(self):
"""Remove the instance default for ``?bucket=``, or for every bucket."""
return _delete_policy("instance", None)
@admin_ns.route("/admin/quotas/teams/<string:team_id>")
class AdminTeamQuotaResource(Resource):
@admin_required
def get(self, team_id):
"""The per-member allowance of one team."""
if not looks_like_uuid(team_id):
return _error("Team not found", 404)
with db_readonly() as conn:
if TeamsRepository(conn).get(team_id) is None:
return _error("Team not found", 404)
rows = QuotaPoliciesRepository(conn).list_for_subject("team", team_id)
return make_response(
jsonify({"success": True, "policies": [_policy_json(r) for r in rows]}), 200
)
@admin_required
def put(self, team_id):
"""Set the allowance each member of the team gets."""
if not looks_like_uuid(team_id):
return _error("Team not found", 404)
with db_readonly() as conn:
if TeamsRepository(conn).get(team_id) is None:
return _error("Team not found", 404)
return _put_policy("team", team_id)
@admin_required
def delete(self, team_id):
"""Remove the team's allowance for ``?bucket=``, or for every bucket."""
if not looks_like_uuid(team_id):
return _error("Team not found", 404)
return _delete_policy("team", team_id)
@admin_ns.route("/admin/quotas/users/<string:user_id>")
class AdminUserQuotaResource(Resource):
@admin_required
def get(self, user_id):
"""A user's overrides and the limits and usage they resolve to."""
with db_readonly() as conn:
if UsersRepository(conn).get(user_id) is None:
return _error("User not found", 404)
rows = QuotaPoliciesRepository(conn).list_for_subject("user", user_id)
statuses = QuotaService.status(user_id, ("all", *REQUEST_BUCKETS))
return make_response(
jsonify(
{
"success": True,
"period": settings.QUOTA_PERIOD,
"policies": [_policy_json(r) for r in rows],
"effective": [s.to_dict() for s in statuses],
}
),
200,
)
@admin_required
def put(self, user_id):
"""Set one user's override, which beats team allowances and the default."""
with db_readonly() as conn:
if UsersRepository(conn).get(user_id) is None:
return _error("User not found", 404)
return _put_policy("user", user_id)
@admin_required
def delete(self, user_id):
"""Remove the user's override for ``?bucket=``, or for every bucket."""
return _delete_policy("user", user_id)
+22 -1
View File
@@ -23,6 +23,9 @@ from docsgpt.api.user.authz import ROLE_ADMIN, admin_required
from docsgpt.storage.db.repositories.admin_stats import AdminStatsRepository
from docsgpt.storage.db.repositories.auth_events import AuthEventsRepository
from docsgpt.storage.db.repositories.device_audit_log import DeviceAuditLogRepository
from docsgpt.storage.db.repositories.personal_access_tokens import (
PersonalAccessTokensRepository,
)
from docsgpt.storage.db.repositories.token_usage import TokenUsageRepository
from docsgpt.storage.db.repositories.user_roles import UserRolesRepository
from docsgpt.storage.db.repositories.users import UsersRepository
@@ -246,12 +249,30 @@ class AdminUserSessionsResource(Resource):
"""Force-logout: revoke the user's live OIDC sessions (best-effort)."""
ok = denylist.deny_user(user_id)
with db_session() as conn:
# A forced logout that left API credentials alive would not be one.
revoked_token_ids = PersonalAccessTokensRepository(conn).revoke_all_for_user(
user_id, reason="admin_sessions_revoked"
)
# One pat_revoked event per token, like every other revocation path.
for token_id in revoked_token_ids:
AuthEventsRepository(conn).insert(
user_id,
"pat_revoked",
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={"token_id": token_id, "by": _actor(), "via": "admin_sessions_revoked"},
)
AuthEventsRepository(conn).insert(
user_id,
"admin_sessions_revoked",
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={"by": _actor(), "via": "admin_api", "persisted": ok},
metadata={
"by": _actor(),
"via": "admin_api",
"persisted": ok,
"personal_access_tokens_revoked": len(revoked_token_ids),
},
)
return make_response(jsonify({"success": True, "revoked": ok}), 200)
+8 -2
View File
@@ -103,7 +103,9 @@ class AnswerResource(Resource, BaseAnswerResource):
)
if not processor.decoded_token:
return make_response({"error": "Unauthorized"}, 401)
if error := self.check_usage(processor.agent_config):
if error := self.check_usage_on_resume(
processor, data["conversation_id"]
):
return error
stream = self.complete_stream(
question="",
@@ -129,7 +131,11 @@ class AnswerResource(Resource, BaseAnswerResource):
if not processor.decoded_token:
return make_response({"error": "Unauthorized"}, 401)
if error := self.check_usage(processor.agent_config):
if error := self.check_usage(
processor.agent_config,
processor.decoded_token,
agent_id=processor.agent_id,
):
return error
should_persist, visibility = resolve_persistence(
+49 -2
View File
@@ -23,6 +23,8 @@ from docsgpt.core.model_utils import (
from docsgpt.core.settings import settings
from docsgpt.error import sanitize_api_error
from docsgpt.llm.llm_creator import LLMCreator
from docsgpt.quotas.http import quota_exceeded_response
from docsgpt.quotas.service import QuotaService
from docsgpt.storage.db.repositories.agents import AgentsRepository
from docsgpt.storage.db.repositories.conversations import (
HeartbeatState,
@@ -112,17 +114,34 @@ class BaseAnswerResource:
prepared.append(item)
return prepared
def check_usage(self, agent_config: Dict) -> Optional[Response]:
"""Check if there is a usage limit and if it is exceeded
def check_usage(
self,
agent_config: Dict,
decoded_token: Optional[Dict] = None,
agent_id: Optional[str] = None,
) -> Optional[Response]:
"""Refuse the request when a usage limit is exhausted.
The billable user's quota is checked first, for every request; the
agent's own 24h token and request limits then apply to traffic that
runs through an agent.
Args:
agent_config: The config dict of agent instance
decoded_token: The request's resolved identity; its ``sub`` is the
billable user.
agent_id: The agent the request runs through. A draft agent has no
key, but its usage rows carry the agent id, so it is agent traffic.
Returns:
None or Response if either of limits exceeded.
"""
api_key = agent_config.get("user_api_key")
user_id = (decoded_token or {}).get("sub") or agent_config.get("user_id")
exceeded = QuotaService.check(user_id, "agent" if api_key or agent_id else "direct")
if exceeded is not None:
return quota_exceeded_response(exceeded)
if not api_key:
return None
with db_readonly() as conn:
@@ -197,6 +216,34 @@ class BaseAnswerResource:
)
return None
def check_usage_on_resume(self, processor: Any, conversation_id: Any) -> Optional[Response]:
"""Run ``check_usage`` for a tool continuation, releasing its claim on refusal.
``resume_from_tool_actions`` has already claimed the paused turn by the
time the limits can be checked (the agent config comes from the claimed
state). A refusal returns before ``complete_stream`` and its cleanup, so
the claim is released here; otherwise retries get a 409 until the stale
claim is reverted.
Args:
processor: The ``StreamProcessor`` that resumed the turn.
conversation_id: The conversation whose pending state was claimed.
Returns:
None, or the refusal Response.
"""
error = self.check_usage(
processor.agent_config, processor.decoded_token, agent_id=processor.agent_id
)
if error is None or not conversation_id:
return error
user = processor.initial_user_id or (processor.decoded_token or {}).get("sub")
try:
ContinuationService().release_claim(str(conversation_id), user)
except Exception:
logger.exception("Failed to release resume claim after a usage refusal")
return error
def complete_stream(
self,
question: str,
+8 -2
View File
@@ -115,7 +115,9 @@ class StreamResource(Resource, BaseAnswerResource):
status=401,
mimetype="text/event-stream",
)
if error := self.check_usage(processor.agent_config):
if error := self.check_usage_on_resume(
processor, data["conversation_id"]
):
return error
return Response(
with_sse_keepalive(
@@ -151,7 +153,11 @@ class StreamResource(Resource, BaseAnswerResource):
mimetype="text/event-stream",
)
if error := self.check_usage(processor.agent_config):
if error := self.check_usage(
processor.agent_config,
processor.decoded_token,
agent_id=processor.agent_id,
):
return error
should_persist, visibility = resolve_persistence(
visibility_flag=data.get("visibility"),
@@ -367,7 +367,7 @@ class CompressionService:
never mutated.
"""
max_tokens = int(
getattr(settings, "COMPRESSION_RECENT_FIELD_MAX_TOKENS", 8000) or 0
settings.COMPRESSION_RECENT_FIELD_MAX_TOKENS or 0
)
if max_tokens <= 0:
return queries
+32 -3
View File
@@ -9,34 +9,63 @@ from __future__ import annotations
import uuid
from contextvars import Token
from typing import Optional, Tuple
from typing import Optional, Sequence, Tuple, Union
import anyio
from starlette.requests import Request
from starlette.responses import JSONResponse
from docsgpt.api.oidc.denylist import is_denied as oidc_session_denied
from docsgpt.api.pat.tokens import is_pat
from docsgpt.auth import handle_auth
from docsgpt.core import log_context
from docsgpt.core.settings import settings
async def authenticate(request: Request) -> Tuple[Optional[dict], Optional[JSONResponse]]:
async def authenticate(
request: Request, *, pat_scope: Union[str, Sequence[str], None] = None
) -> Tuple[Optional[dict], Optional[JSONResponse]]:
"""Decode the caller's JWT the way Flask's ``authenticate_request`` does.
Args:
request: The incoming Starlette request.
pat_scope: Scope (or any-of scopes) a personal access token needs for this route. Left
unset, the route rejects PATs outright (deny by default, matching
the Flask rule table in ``docsgpt/api/pat/rules.py``). A token with
a resource filter is always rejected.
Returns:
tuple: ``(claims, None)`` for an authenticated caller, ``(None, None)``
when no token was sent (the route decides whether that is allowed), or
``(None, response)`` carrying the 401 to return.
"""
decoded = handle_auth(request)
# A personal access token resolves against Postgres; keep that sync read off the event loop.
decoded = await anyio.to_thread.run_sync(handle_auth, request)
if not decoded:
return None, None
if "error" in decoded:
return None, JSONResponse(decoded, status_code=401)
if is_pat(decoded):
# A PAT lookup already excludes revoked tokens and deactivated users,
# so the session denylist below does not apply to it.
accepted = (pat_scope,) if isinstance(pat_scope, str) else tuple(pat_scope or ())
if not set(accepted).intersection(decoded.get("scopes") or []):
return None, JSONResponse(
{"success": False, "message": "Token lacks the required scope", "error": "insufficient_scope"},
status_code=403,
)
if decoded.get("resource_filter"):
# These routes sit outside the Flask rule table and cannot tie what
# they serve to an allowlist, so a restricted token is kept out.
return None, JSONResponse(
{
"success": False,
"message": "This endpoint is not available to a resource-restricted token",
"error": "resource_not_allowed",
},
status_code=403,
)
return decoded, None
# The denylist is a sync Redis read; keep it off the event loop.
if settings.AUTH_TYPE == "oidc" and await anyio.to_thread.run_sync(oidc_session_denied, decoded):
return None, JSONResponse(
+4 -3
View File
@@ -25,6 +25,7 @@ from starlette.responses import Response
from starlette.routing import Route
from docsgpt.api.asgi_auth import authenticate, bind_log_context, json_error
from docsgpt.api.pat.rules import MESSAGE_REPLAY_SCOPES
from docsgpt.api.asgi_stream import sse_response
from docsgpt.core.settings import settings
from docsgpt.storage.db.session import db_readonly
@@ -33,7 +34,6 @@ from docsgpt.streaming.async_event_replay import (
)
from docsgpt.streaming.async_redis import get_async_redis_instance
from docsgpt.streaming.event_replay import (
DEFAULT_KEEPALIVE_SECONDS,
DEFAULT_POLL_TIMEOUT_SECONDS,
)
from docsgpt.streaming.sse_leases import StreamCapExceeded, acquire_stream_lease
@@ -95,7 +95,8 @@ async def stream_message_events(request: Request) -> Response:
"""
# Same JWT decoder and OIDC revocation check as the Flask routes. With
# AUTH_TYPE unset the caller resolves to ``{"sub": "local"}``.
decoded, error = await authenticate(request)
# Same scopes as its Flask sibling GET /api/messages/<id>/tail.
decoded, error = await authenticate(request, pat_scope=MESSAGE_REPLAY_SCOPES)
if error is not None:
return error
user_id = decoded.get("sub") if isinstance(decoded, dict) else None
@@ -127,7 +128,7 @@ async def stream_message_events(request: Request) -> Response:
)
last_event_id = _normalise_last_event_id(raw_cursor)
keepalive_seconds = float(
getattr(settings, "SSE_KEEPALIVE_SECONDS", DEFAULT_KEEPALIVE_SECONDS)
settings.SSE_KEEPALIVE_SECONDS
)
logger.info(
View File
Whitespace-only changes.
+313
View File
@@ -0,0 +1,313 @@
"""Personal access token management.
``/api/user/tokens`` lets a signed-in user list, create and revoke their own
tokens; ``/api/admin/...`` lets an admin inspect and revoke anyone's. None of
these routes accept a PAT (see ``docsgpt/api/pat/rules.py``), so a leaked
token can neither mint a replacement nor widen itself.
"""
from __future__ import annotations
import uuid
from datetime import datetime, timezone
from flask import jsonify, make_response, request
from flask_restx import Namespace, Resource
from sqlalchemy.exc import IntegrityError
from docsgpt.api.pat.tokens import (
FILTERABLE_FAMILIES,
SCOPES,
auth_type_supports_pats,
generate_token,
is_pat,
normalize_resource_filter,
normalize_scopes,
renewal_lifetime_days,
resolve_expiry,
)
from docsgpt.api.user.authz import admin_required
from docsgpt.core.settings import settings
from docsgpt.storage.db.repositories.auth_events import AuthEventsRepository
from docsgpt.storage.db.repositories.personal_access_tokens import (
PersonalAccessTokensRepository,
)
from docsgpt.storage.db.session import db_readonly, db_session
pat_ns = Namespace("tokens", description="Personal access tokens", path="/api")
_MAX_NAME_LENGTH = 100
def _error(message: str, status: int):
return make_response(jsonify({"success": False, "message": message}), status)
def _session_user_id():
"""The caller's id, or ``None`` for anonymous and PAT callers alike."""
decoded = getattr(request, "decoded_token", None)
if not decoded or is_pat(decoded):
return None
return decoded.get("sub")
def _valid_uuid(value: str) -> bool:
"""Canonical form only: ``uuid.UUID`` also accepts ``urn:uuid:…`` and braces, which Postgres does not."""
try:
return str(uuid.UUID(str(value))) == str(value).lower()
except (ValueError, AttributeError, TypeError):
return False
def _is_expired(expires_at) -> bool:
if not expires_at:
return False
try:
moment = expires_at if isinstance(expires_at, datetime) else datetime.fromisoformat(str(expires_at))
except ValueError:
return False
if moment.tzinfo is None:
moment = moment.replace(tzinfo=timezone.utc)
return moment <= datetime.now(timezone.utc)
def serialize_token(row: dict) -> dict:
# The row keeps status 'active' until someone revokes it; report what is true for a caller.
status = "expired" if row["status"] == "active" and _is_expired(row.get("expires_at")) else row["status"]
return {
"id": str(row["id"]),
"name": row["name"],
"token_prefix": row["token_prefix"],
"scopes": list(row.get("scopes") or []),
"resource_filter": row.get("resource_filter") or {},
"status": status,
"expires_at": row.get("expires_at"),
"last_used_at": row.get("last_used_at"),
"last_used_ip": row.get("last_used_ip"),
"created_at": row.get("created_at"),
"regenerated_at": row.get("regenerated_at"),
"revoked_at": row.get("revoked_at"),
}
def _policy() -> dict:
return {
"enabled": auth_type_supports_pats(),
"default_lifetime_days": settings.PAT_DEFAULT_LIFETIME_DAYS,
"max_lifetime_days": settings.PAT_MAX_LIFETIME_DAYS,
"allow_non_expiring": settings.PAT_ALLOW_NON_EXPIRING,
"max_per_user": settings.PAT_MAX_PER_USER,
"filterable_families": list(FILTERABLE_FAMILIES),
}
@pat_ns.route("/user/tokens")
class PersonalAccessTokens(Resource):
def get(self):
"""List the caller's tokens with the scope catalog and the server's token policy."""
user_id = _session_user_id()
if not user_id:
return _error("Authentication required", 401)
with db_readonly() as conn:
rows = PersonalAccessTokensRepository(conn).list_for_user(user_id)
return make_response(
jsonify(
{
"success": True,
"tokens": [serialize_token(r) for r in rows],
"scopes": [{"name": k, "description": v} for k, v in SCOPES.items()],
"policy": _policy(),
}
),
200,
)
def post(self):
"""Create a token. The plaintext ``token`` is returned here and never again."""
user_id = _session_user_id()
if not user_id:
return _error("Authentication required", 401)
if not auth_type_supports_pats():
return _error("Personal access tokens are not available on this server", 403)
body = request.get_json(silent=True)
if body is None:
body = {}
if not isinstance(body, dict):
return _error("Request body must be a JSON object", 400)
name = body.get("name")
if not isinstance(name, str) or not name.strip():
return _error("name is required", 400)
name = name.strip()
if len(name) > _MAX_NAME_LENGTH:
return _error(f"name must be at most {_MAX_NAME_LENGTH} characters", 400)
try:
scopes = normalize_scopes(body.get("scopes"))
resource_filter = normalize_resource_filter(body.get("resource_filter"), scopes)
expires_at = resolve_expiry(body.get("expires_in_days"))
except ValueError as exc:
return _error(str(exc), 400)
token, token_hash, token_prefix = generate_token()
try:
with db_session() as conn:
repo = PersonalAccessTokensRepository(conn)
# Serialise this user's creates so concurrent requests cannot both pass the cap check.
repo.lock_user(user_id)
if repo.count_active(user_id) >= settings.PAT_MAX_PER_USER:
return _error(
f"Token limit reached ({settings.PAT_MAX_PER_USER}); revoke one first", 409
)
repo.retire_expired_name(user_id, name)
if repo.name_in_use(user_id, name):
return _error("A token with this name already exists", 409)
row = repo.create(
user_id,
name,
token_hash=token_hash,
token_prefix=token_prefix,
scopes=scopes,
resource_filter=resource_filter,
expires_at=expires_at,
)
AuthEventsRepository(conn).insert(
user_id,
"pat_created",
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={
"token_id": str(row["id"]),
"name": name,
"scopes": scopes,
"resource_filter": resource_filter,
"expires_at": row.get("expires_at"),
},
)
except IntegrityError:
# Lost a race against a concurrent create with the same name.
return _error("A token with this name already exists", 409)
return make_response(
jsonify({"success": True, "token": token, "personal_access_token": serialize_token(row)}),
201,
)
@pat_ns.route("/user/tokens/<string:token_id>")
class PersonalAccessToken(Resource):
def delete(self, token_id):
"""Revoke one of the caller's tokens. Takes effect on the next request."""
user_id = _session_user_id()
if not user_id:
return _error("Authentication required", 401)
if not _valid_uuid(token_id):
return _error("Token not found", 404)
with db_session() as conn:
revoked = PersonalAccessTokensRepository(conn).revoke(token_id, user_id)
if revoked:
AuthEventsRepository(conn).insert(
user_id,
"pat_revoked",
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={"token_id": token_id, "by": user_id},
)
if not revoked:
return _error("Token not found", 404)
return make_response(jsonify({"success": True}), 200)
@pat_ns.route("/user/tokens/<string:token_id>/regenerate")
class PersonalAccessTokenRegenerate(Resource):
def post(self, token_id):
"""Issue a new secret for a token and reset its expiry.
Name, scopes and restrictions stay; the old secret stops working at
once. ``expires_in_days`` is optional and defaults to the lifetime the
token was last issued with. An expired token can be renewed this way; a
revoked one cannot. The plaintext ``token`` is returned here and never again.
"""
user_id = _session_user_id()
if not user_id:
return _error("Authentication required", 401)
if not auth_type_supports_pats():
return _error("Personal access tokens are not available on this server", 403)
if not _valid_uuid(token_id):
return _error("Token not found", 404)
body = request.get_json(silent=True)
if body is None:
body = {}
if not isinstance(body, dict):
return _error("Request body must be a JSON object", 400)
token, token_hash, token_prefix = generate_token()
with db_session() as conn:
repo = PersonalAccessTokensRepository(conn)
current = repo.get(token_id, user_id)
if not current or current["status"] != "active":
return _error("Token not found", 404)
requested = body.get("expires_in_days")
if requested is None:
requested = renewal_lifetime_days(current)
try:
expires_at = resolve_expiry(requested)
except ValueError as exc:
return _error(str(exc), 400)
row = repo.regenerate(
token_id, user_id, token_hash=token_hash, token_prefix=token_prefix, expires_at=expires_at
)
if not row:
# Revoked between the read and the write.
return _error("Token not found", 404)
AuthEventsRepository(conn).insert(
user_id,
"pat_regenerated",
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={
"token_id": token_id,
"name": row["name"],
"expires_at": row.get("expires_at"),
"previous_expires_at": current.get("expires_at"),
},
)
return make_response(
jsonify({"success": True, "token": token, "personal_access_token": serialize_token(row)}),
200,
)
@pat_ns.route("/admin/users/<string:user_id>/tokens")
class AdminUserTokens(Resource):
@admin_required
def get(self, user_id):
"""List a user's tokens, revoked ones included."""
with db_readonly() as conn:
rows = PersonalAccessTokensRepository(conn).list_for_user(user_id, include_revoked=True)
return make_response(
jsonify({"success": True, "tokens": [serialize_token(r) for r in rows]}), 200
)
@pat_ns.route("/admin/tokens/<string:token_id>")
class AdminToken(Resource):
@admin_required
def delete(self, token_id):
"""Revoke any user's token."""
if not _valid_uuid(token_id):
return _error("Token not found", 404)
actor = (getattr(request, "decoded_token", None) or {}).get("sub")
with db_session() as conn:
repo = PersonalAccessTokensRepository(conn)
row = repo.get(token_id)
revoked = bool(row) and repo.revoke(token_id, reason="admin_revoked")
if revoked:
AuthEventsRepository(conn).insert(
row["user_id"],
"pat_revoked",
ip=request.remote_addr,
user_agent=request.headers.get("User-Agent"),
metadata={"token_id": token_id, "by": actor, "via": "admin_api"},
)
if not revoked:
return _error("Token not found", 404)
return make_response(jsonify({"success": True}), 200)
+537
View File
@@ -0,0 +1,537 @@
"""What a personal access token may call: the scope and resource rule table.
Authorization for PATs is central and deny by default. ``RULES`` maps a Flask
route (its rule string and method) to the scope it needs; a PAT request to a
route that is not listed is refused, so a new endpoint is unreachable by token
until someone classifies it here. ``tests/api/test_pat_rules.py`` fails when a
registered route is in neither ``RULES`` nor ``DENIED``.
A token may also carry a resource filter (``{"agents": [ids]}``). For a
restricted family the rule must be able to prove the request stays inside the
allowlist: it names where the id travels (``ids``), or declares that the route
filters its own listing (``listing``), or delegates to the route
(``in_route``). Anything else, creation included, is refused. ``refs`` cover
ids of *other* families a route accepts (an agent update naming a source), and
``blocked_by`` closes routes whose rows hang off a family the rule cannot see
(a schedule belongs to an agent).
Session (JWT) callers never pass through here.
"""
from __future__ import annotations
import json
import uuid
from dataclasses import dataclass
from typing import Any, Callable, Iterable, Optional
from docsgpt.api.pat.tokens import is_pat
VIEW, QUERY, JSON, FORM, BODY = "view", "query", "json", "form", "body"
Locator = tuple[str, str]
@dataclass(frozen=True)
class Rule:
"""Requirement for one route+method. ``scopes`` is any-of; empty means any valid token."""
scopes: tuple[str, ...] = ()
family: Optional[str] = None
ids: tuple[Locator, ...] = ()
refs: tuple[tuple[str, Locator], ...] = ()
listing: bool = False
open: bool = False
in_route: bool = False
blocked_by: tuple[str, ...] = ()
check: Optional[Callable[[Any, dict, Optional[str]], Optional[str]]] = None
def _rule(scope: Optional[str] = None, *ids: Locator, any_of: tuple[str, ...] = (), **kwargs) -> Rule:
scopes = any_of or ((scope,) if scope else ())
family = kwargs.pop("family", None)
if family is None and scope:
family = scope.partition(":")[0]
return Rule(scopes=scopes, family=family, ids=tuple(ids), **kwargs)
#: Scopes that admit a token to message replay. Shared by the Flask tail route
#: below and its ASGI sibling GET /api/messages/<id>/events (docsgpt/api/async_sse.py),
#: which sits outside this table.
MESSAGE_REPLAY_SCOPES = ("conversations:read", "chat:run")
_ALL_FAMILIES = ("agents", "sources", "prompts", "tools", "workflows")
_WORKFLOW_CONTENT_FAMILIES = ("sources", "tools", "prompts")
# A route reached through an agent id can prove the agent; nothing else about it.
_NON_AGENT_FAMILIES = ("sources", "prompts", "tools", "workflows")
# Ids of other families that agent create/update accept in their JSON-or-form body.
_AGENT_BODY_REFS = (
("sources", (BODY, "source")),
("sources", (BODY, "sources")),
("prompts", (BODY, "prompt_id")),
("tools", (BODY, "tools")),
("workflows", (BODY, "workflow")),
)
def _conversation_agent_id(conversation_id: str, user_id: Optional[str]) -> tuple[bool, str]:
"""``(found, agent_id)`` for a conversation the user can reach; ``agent_id`` is "" when it has none."""
from docsgpt.storage.db.repositories.conversations import ConversationsRepository
from docsgpt.storage.db.session import db_readonly
if not user_id:
return False, ""
try:
with db_readonly() as conn:
row = ConversationsRepository(conn).get_any(str(conversation_id), user_id)
except Exception:
return False, ""
if not row:
return False, ""
return True, str(row.get("agent_id") or "")
def _chat_check(request, resource_filter: dict, user_id: Optional[str]) -> Optional[str]:
"""Keep a restricted token's chat traffic inside its allowlists.
An agent brings its own sources, prompt and tools, which this table cannot
see, so a restricted token must name an allowed agent. The one exception is
a token restricted on sources only, which may chat against allowed sources
directly. Everything that could swap in another agent or another set of
resources is refused: an agent ``api_key``, an inline workflow, and a
``conversation_id`` that belongs to a different agent (the server would
otherwise continue, append to, or resume tool calls of that conversation).
Chat executes tools: an agent's own, or the user's defaults when there is
no agent. Neither can be held to a tools allowlist from here, so a token
restricted on tools cannot chat at all.
"""
body = _json_body(request)
if "tools" in resource_filter:
return "A token restricted to specific tools cannot use chat endpoints"
if body.get("api_key"):
return "A restricted token cannot chat with an agent API key; pass agent_id"
if body.get("workflow"):
# An inline workflow graph (builder preview) can reference any resource.
return "A restricted token cannot run an inline workflow"
agent_ids = _as_ids(body.get("agent_id"))
if "agents" in resource_filter:
if len(agent_ids) != 1:
return "This token is restricted to specific agents; pass agent_id"
# the agent id itself is verified through ``refs``
elif set(resource_filter) - {"sources"}:
return "Restrict this token to specific agents to use chat endpoints"
elif agent_ids:
return "This token is restricted to specific sources and cannot run agents"
conversation_id = body.get("conversation_id")
if conversation_id:
found, conversation_agent = _conversation_agent_id(conversation_id, user_id)
expected = agent_ids[0] if agent_ids else ""
if not found or _canonical(conversation_agent) != _canonical(expected):
return "This conversation does not belong to the agent this token may use"
return None
def _agent_body_check(request, resource_filter: dict, user_id: Optional[str]) -> Optional[str]:
"""A workflow pulls in its own sources, tools and prompts, which a reference check cannot see.
A token restricted on any of those may attach a workflow to an agent only
when it is also restricted on workflows, so the workflow is one its owner
chose (``refs`` then verifies the id).
"""
if "workflows" in resource_filter or not set(resource_filter) & {"sources", "tools", "prompts"}:
return None
if _read(request, (BODY, "workflow")):
return "A token restricted to specific sources, tools or prompts cannot attach a workflow to an agent"
return None
_CHAT = dict(
family=None,
refs=(
("agents", (JSON, "agent_id")),
("sources", (JSON, "active_docs")),
("prompts", (JSON, "prompt_id")),
("workflows", (JSON, "workflow_id")),
),
check=_chat_check,
)
RULES: dict[tuple[str, str], Rule] = {
# Identity and public metadata: any valid token.
("/api/user/me", "GET"): _rule(open=True),
("/api/user/quota", "GET"): _rule(open=True),
("/api/health", "GET"): _rule(open=True),
("/api/config", "GET"): _rule(open=True),
# Agents
("/api/get_agent", "GET"): _rule("agents:read", (QUERY, "id")),
("/api/get_agents", "GET"): _rule("agents:read", listing=True),
("/api/pinned_agents", "GET"): _rule("agents:read"),
("/api/shared_agents", "GET"): _rule("agents:read"),
("/api/template_agents", "GET"): _rule("agents:read", open=True),
("/api/export_agent", "GET"): _rule("agents:read", (QUERY, "id")),
("/api/guardrails/catalog", "GET"): _rule("agents:read", open=True),
("/api/guardrails/events", "GET"): _rule("agents:read", (QUERY, "agent_id")),
("/api/guardrails/summary", "GET"): _rule("agents:read", (QUERY, "agent_id")),
("/api/agents/folders/", "GET"): _rule("agents:read", open=True),
("/api/agents/folders/<string:folder_id>", "GET"): _rule("agents:read"),
("/api/create_agent", "POST"): _rule("agents:write", refs=_AGENT_BODY_REFS, check=_agent_body_check),
("/api/update_agent/<string:agent_id>", "PUT"): _rule(
"agents:write", (VIEW, "agent_id"), refs=_AGENT_BODY_REFS, check=_agent_body_check
),
("/api/delete_agent", "DELETE"): _rule("agents:write", (QUERY, "id")),
("/api/adopt_agent", "POST"): _rule("agents:write"),
("/api/pin_agent", "POST"): _rule("agents:write", (QUERY, "id")),
("/api/remove_shared_agent", "DELETE"): _rule("agents:write", (QUERY, "id")),
("/api/share_agent", "PUT"): _rule("agents:write", (JSON, "id")),
("/api/import_agent/plan", "POST"): _rule("agents:write", in_route=True),
("/api/import_agent", "POST"): _rule("agents:write", in_route=True),
("/api/agents/folders/", "POST"): _rule("agents:write"),
("/api/agents/folders/<string:folder_id>", "PUT"): _rule("agents:write"),
("/api/agents/folders/<string:folder_id>", "DELETE"): _rule("agents:write"),
("/api/agents/folders/move_agent", "POST"): _rule("agents:write", (JSON, "agent_id")),
("/api/agents/folders/bulk_move", "POST"): _rule("agents:write", (JSON, "agent_ids")),
("/api/regenerate_agent_key/<string:agent_id>", "POST"): _rule("agents:keys", (VIEW, "agent_id")),
("/api/agent_webhook", "GET"): _rule("agents:keys", (QUERY, "id")),
# Schedules hang off an agent, and a schedule runs that agent with a free-form
# instruction and stores the output. The agent id proves the agent and nothing
# else, so tokens restricted on any other family are kept out; routes that
# carry only a schedule id prove nothing and are closed to every restricted token.
("/api/agents/<string:agent_id>/schedules", "GET"): _rule(
"schedules:read", refs=(("agents", (VIEW, "agent_id")),), blocked_by=_NON_AGENT_FAMILIES
),
("/api/agents/<string:agent_id>/schedules", "POST"): _rule(
"schedules:write", refs=(("agents", (VIEW, "agent_id")),), blocked_by=_NON_AGENT_FAMILIES
),
("/api/schedules/<string:schedule_id>", "GET"): _rule("schedules:read", blocked_by=_ALL_FAMILIES),
("/api/schedules/<string:schedule_id>/runs", "GET"): _rule("schedules:read", blocked_by=_ALL_FAMILIES),
("/api/schedules/<string:schedule_id>/runs/<string:run_id>", "GET"): _rule(
"schedules:read", blocked_by=_ALL_FAMILIES
),
("/api/schedules/<string:schedule_id>", "PUT"): _rule("schedules:write", blocked_by=_ALL_FAMILIES),
("/api/schedules/<string:schedule_id>", "PATCH"): _rule("schedules:write", blocked_by=_ALL_FAMILIES),
("/api/schedules/<string:schedule_id>", "DELETE"): _rule("schedules:write", blocked_by=_ALL_FAMILIES),
("/api/schedules/<string:schedule_id>/run", "POST"): _rule("schedules:write", blocked_by=_ALL_FAMILIES),
# Sources
("/api/sources", "GET"): _rule("sources:read", listing=True),
# Counted and paged in SQL, so it cannot be narrowed here; restricted tokens use /api/sources.
("/api/sources/paginated", "GET"): _rule("sources:read"),
("/api/directory_structure", "GET"): _rule("sources:read", (QUERY, "id")),
("/api/get_chunks", "GET"): _rule("sources:read", (QUERY, "id")),
("/api/sources/<string:source_id>/wiki/pages", "GET"): _rule("sources:read", (VIEW, "source_id")),
("/api/sources/<string:source_id>/wiki/page", "GET"): _rule("sources:read", (VIEW, "source_id")),
("/api/sources/<string:source_id>/graph", "GET"): _rule("sources:read", (VIEW, "source_id")),
("/api/sources/<string:source_id>/graph/node/<string:node_id>", "GET"): _rule(
"sources:read", (VIEW, "source_id")
),
# Ingestion and attachment extraction both report through this poll.
("/api/task_status", "GET"): _rule(any_of=("sources:read", "sources:write", "chat:run"), open=True),
("/api/upload", "POST"): _rule("sources:write"),
("/api/remote", "POST"): _rule("sources:write"),
("/api/sources/wiki", "POST"): _rule("sources:write"),
("/api/delete_old", "GET"): _rule("sources:write", (QUERY, "source_id")),
("/api/manage_sync", "POST"): _rule("sources:write", (JSON, "source_id")),
("/api/sync_source", "POST"): _rule("sources:write", (JSON, "source_id")),
("/api/sources/reingest", "POST"): _rule("sources:write", (JSON, "source_id")),
("/api/manage_source_files", "POST"): _rule("sources:write", (FORM, "source_id")),
("/api/sources/<string:source_id>/config", "PATCH"): _rule("sources:write", (VIEW, "source_id")),
("/api/sources/<string:source_id>/wiki/page", "PUT"): _rule("sources:write", (VIEW, "source_id")),
("/api/sources/<string:source_id>/wiki/convert", "POST"): _rule("sources:write", (VIEW, "source_id")),
("/api/sources/<string:source_id>/graphrag/enable", "POST"): _rule(
"sources:write", (VIEW, "source_id")
),
("/api/add_chunk", "POST"): _rule("sources:write", (JSON, "id")),
("/api/update_chunk", "PUT"): _rule("sources:write", (JSON, "id")),
("/api/delete_chunk", "DELETE"): _rule("sources:write", (QUERY, "id")),
# Prompts
("/api/get_prompts", "GET"): _rule("prompts:read", listing=True),
("/api/get_single_prompt", "GET"): _rule("prompts:read", (QUERY, "id")),
("/api/create_prompt", "POST"): _rule("prompts:write"),
("/api/update_prompt", "POST"): _rule("prompts:write", (JSON, "id")),
("/api/delete_prompt", "POST"): _rule("prompts:write", (JSON, "id")),
# Tools
("/api/available_tools", "GET"): _rule("tools:read", open=True),
("/api/get_tools", "GET"): _rule("tools:read", listing=True),
("/api/create_tool", "POST"): _rule("tools:write"),
("/api/parse_spec", "POST"): _rule("tools:write", open=True),
("/api/update_tool", "POST"): _rule("tools:write", (JSON, "id")),
("/api/update_tool_config", "POST"): _rule("tools:write", (JSON, "id")),
("/api/update_tool_actions", "POST"): _rule("tools:write", (JSON, "id")),
("/api/update_tool_status", "POST"): _rule("tools:write", (JSON, "id")),
("/api/delete_tool", "POST"): _rule("tools:write", (JSON, "id")),
("/api/mcp_server/test", "POST"): _rule("tools:write", open=True),
("/api/mcp_server/save", "POST"): _rule("tools:write", (JSON, "id")),
# Models
("/api/models", "GET"): _rule(any_of=("models:read", "chat:run"), open=True),
("/api/user/models", "GET"): _rule("models:read"),
("/api/user/models/<string:model_id>", "GET"): _rule("models:read"),
("/api/user/models", "POST"): _rule("models:write"),
("/api/user/models/<string:model_id>", "PATCH"): _rule("models:write"),
("/api/user/models/<string:model_id>", "DELETE"): _rule("models:write"),
("/api/user/models/test", "POST"): _rule("models:write"),
("/api/user/models/<string:model_id>/test", "POST"): _rule("models:write"),
# Workflows
# A workflow graph names sources, tools and prompts inside its nodes, out of reach of ``refs``.
("/api/workflows", "POST"): _rule("workflows:write", blocked_by=_WORKFLOW_CONTENT_FAMILIES),
("/api/workflows/<string:workflow_id>", "GET"): _rule("workflows:read", (VIEW, "workflow_id")),
("/api/workflows/<string:workflow_id>", "PUT"): _rule(
"workflows:write", (VIEW, "workflow_id"), blocked_by=_WORKFLOW_CONTENT_FAMILIES
),
("/api/workflows/<string:workflow_id>", "DELETE"): _rule("workflows:write", (VIEW, "workflow_id")),
# Conversations and analytics span every agent and carry cited source text and tool
# output, so they are closed to every restricted token.
("/api/get_conversations", "GET"): _rule("conversations:read", blocked_by=_ALL_FAMILIES),
("/api/search_conversations", "GET"): _rule("conversations:read", blocked_by=_ALL_FAMILIES),
("/api/get_single_conversation", "GET"): _rule("conversations:read", blocked_by=_ALL_FAMILIES),
# A message cannot be tied to an allowlist from here, so any restricted token is kept out.
("/api/messages/<string:message_id>/tail", "GET"): _rule(
any_of=MESSAGE_REPLAY_SCOPES, family=None, blocked_by=_ALL_FAMILIES
),
("/api/delete_conversation", "POST"): _rule("conversations:write", blocked_by=_ALL_FAMILIES),
("/api/delete_all_conversations", "GET"): _rule("conversations:write", blocked_by=_ALL_FAMILIES),
("/api/update_conversation_name", "POST"): _rule("conversations:write", blocked_by=_ALL_FAMILIES),
("/api/feedback", "POST"): _rule("conversations:write", blocked_by=_ALL_FAMILIES),
("/api/get_message_analytics", "POST"): _rule("analytics:read", blocked_by=_ALL_FAMILIES),
("/api/get_token_analytics", "POST"): _rule("analytics:read", blocked_by=_ALL_FAMILIES),
("/api/get_feedback_analytics", "POST"): _rule("analytics:read", blocked_by=_ALL_FAMILIES),
("/api/get_tool_analytics", "POST"): _rule("analytics:read", blocked_by=_ALL_FAMILIES),
("/api/get_schedule_analytics", "POST"): _rule("analytics:read", blocked_by=_ALL_FAMILIES),
("/api/get_user_logs", "POST"): _rule("analytics:read", blocked_by=_ALL_FAMILIES),
# Teams (read only)
("/api/teams", "GET"): _rule("teams:read"),
("/api/teams/<string:team_id>", "GET"): _rule("teams:read"),
("/api/teams/<string:team_id>/members", "GET"): _rule("teams:read"),
("/api/teams/<string:team_id>/grants", "GET"): _rule("teams:read"),
("/api/resource_shares", "GET"): _rule("teams:read"),
# Chat
("/api/answer", "POST"): _rule("chat:run", **_CHAT),
("/stream", "POST"): _rule("chat:run", **_CHAT),
("/api/search", "POST"): _rule("chat:run", **_CHAT),
("/api/store_attachment", "POST"): _rule("chat:run", family=None),
("/api/sources/<string:source_id>/search", "POST"): _rule(
"chat:run", family=None, refs=(("sources", (VIEW, "source_id")),)
),
}
#: Routes a PAT may never call, by exact rule string ("*" = every method) or prefix.
#: Token management, admin, login flows and interactive OAuth handshakes need a
#: signed-in session; the rest have no scope yet. Listing them keeps the
#: classification test honest: a new route must land here or in ``RULES``.
DENIED: dict[str, tuple[str, ...]] = {
"/": ("*",),
"/api/user/tokens": ("*",),
"/api/user/tokens/<string:token_id>": ("*",),
"/api/user/tokens/<string:token_id>/regenerate": ("*",),
"/api/generate_token": ("*",),
"/api/combine": ("*",),
"/api/download": ("*",),
"/api/upload_index": ("*",),
"/api/share": ("*",),
"/api/shared_agent": ("*",),
"/api/shared_conversation/<string:identifier>": ("*",),
"/api/webhooks/agents/<string:webhook_token>": ("*",),
"/api/images/<string:agent_id>/<string:capability>": ("*",),
"/api/mcp_server/callback": ("*",),
"/api/mcp_server/auth_status": ("*",),
"/api/artifact/<artifact_id>": ("*",),
"/api/artifacts": ("*",),
"/api/artifacts/<artifact_id>": ("*",),
"/api/artifacts/<artifact_id>/restore": ("*",),
"/api/artifacts/<artifact_id>/versions/<int:version>": ("*",),
"/api/stt": ("*",),
"/api/stt/live/start": ("*",),
"/api/stt/live/chunk": ("*",),
"/api/stt/live/finish": ("*",),
"/api/tts": ("*",),
"/api/teams": ("POST",),
"/api/teams/<string:team_id>": ("PUT", "DELETE"),
"/api/teams/<string:team_id>/members": ("POST",),
"/api/teams/<string:team_id>/members/<string:member_id>": ("*",),
"/api/teams/<string:team_id>/grants": ("POST", "DELETE"),
"/api/teams/<string:team_id>/transfer_owner": ("*",),
"/swagger.json": ("*",),
}
DENIED_PREFIXES = (
"/api/admin/",
"/api/auth/oidc/",
"/api/connectors/",
"/api/devices",
"/scim/",
"/static/",
"/swaggerui/",
"/v1/",
)
def is_denied(rule: str, method: str) -> bool:
if rule.startswith(DENIED_PREFIXES):
return True
methods = DENIED.get(rule)
return bool(methods) and ("*" in methods or method in methods)
def _json_body(request) -> dict:
body = request.get_json(silent=True)
return body if isinstance(body, dict) else {}
def _as_ids(value: Any) -> list[str]:
"""Flatten whatever a route accepts as ids: a string, a JSON-encoded or plain list, or an ``{id}`` dict."""
if value is None or value == "":
return []
if isinstance(value, dict):
return _as_ids(value.get("id") or value.get("_id") or value.get("workflow_id"))
if isinstance(value, (list, tuple)):
out: list[str] = []
for item in value:
out.extend(_as_ids(item))
return out
text = str(value).strip()
if text[:1] in "[{":
try:
return _as_ids(json.loads(text))
except ValueError:
return [text]
return [text]
def _read(request, locator: Locator) -> list[str]:
where, key = locator
if where == VIEW:
return _as_ids((request.view_args or {}).get(key))
if where == QUERY:
return _as_ids(request.args.get(key))
if where == JSON:
return _as_ids(_json_body(request).get(key))
if where == FORM:
return _as_ids(request.form.get(key))
# BODY: routes that accept JSON or a multipart form interchangeably.
if request.is_json:
return _as_ids(_json_body(request).get(key))
return _as_ids(request.form.get(key))
def _canonical(value: str) -> str:
try:
return str(uuid.UUID(value))
except (ValueError, AttributeError, TypeError):
return value
def _all_allowed(ids: Iterable[str], allowed: Iterable[str]) -> bool:
allowlist = {_canonical(a) for a in allowed}
return all(_canonical(i) in allowlist for i in ids)
def authorize(request, decoded_token: dict) -> Optional[tuple[dict, int]]:
"""Check a PAT request against the table. ``None`` allows; otherwise ``(body, status)``."""
url_rule = getattr(request, "url_rule", None)
if url_rule is None:
# Routing failed (unknown path or wrong method): no view will run, so
# let Flask answer 404/405 instead of masking it with a 403.
return None
rule = RULES.get((url_rule.rule, request.method))
if rule is None:
return (
{
"success": False,
"error": "not_available_to_tokens",
"message": "This endpoint cannot be called with a personal access token",
},
403,
)
granted = set(decoded_token.get("scopes") or [])
if rule.scopes and not granted.intersection(rule.scopes):
return (
{
"success": False,
"error": "insufficient_scope",
"message": f"Token lacks the required scope: {' or '.join(rule.scopes)}",
"required_scope": rule.scopes[0],
},
403,
)
resource_filter = decoded_token.get("resource_filter") or {}
if not resource_filter:
return None
reason = _check_resources(request, rule, resource_filter, decoded_token.get("sub"))
if reason is None:
return None
return ({"success": False, "error": "resource_not_allowed", "message": reason}, 403)
def _check_resources(request, rule: Rule, resource_filter: dict, user_id: Optional[str] = None) -> Optional[str]:
for family in rule.blocked_by:
if family in resource_filter:
return f"This endpoint is not available to a token restricted to specific {family}"
if rule.check is not None:
reason = rule.check(request, resource_filter, user_id)
if reason:
return reason
for family, locator in rule.refs:
if family not in resource_filter:
continue
ids = _read(request, locator)
if ids and not _all_allowed(ids, resource_filter[family]):
return f"Token is not allowed to use one of the referenced {family}"
family = rule.family
if family is None or family not in resource_filter or rule.open or rule.listing or rule.in_route:
return None
ids = [i for locator in rule.ids for i in _read(request, locator)]
if not ids:
return f"This token is restricted to specific {family} and cannot use this endpoint"
if not _all_allowed(ids, resource_filter[family]):
return f"Token is not allowed to access this resource ({family})"
return None
def allowed_ids(request, family: str) -> Optional[set[str]]:
"""The caller's allowlist for ``family``, or ``None`` when unrestricted (or not a PAT).
Used by listing routes (``listing=True``) and by ``in_route`` handlers.
"""
decoded = getattr(request, "decoded_token", None)
if not is_pat(decoded):
return None
ids = (decoded.get("resource_filter") or {}).get(family)
if ids is None:
return None
return {_canonical(str(i)) for i in ids}
def filter_listing(request, family: str, items: list, key: str = "id") -> list:
"""Drop rows outside the caller's allowlist. Rows without a UUID id (built-in presets) are kept."""
allowed = allowed_ids(request, family)
if allowed is None:
return items
kept = []
for item in items:
value = str(item.get(key, ""))
if not _is_uuid(value) or _canonical(value) in allowed:
kept.append(item)
return kept
def _is_uuid(value: str) -> bool:
try:
uuid.UUID(value)
except (ValueError, AttributeError, TypeError):
return False
return True
def may_see_agent_keys(request) -> bool:
"""False for a token without ``agents:keys``: it must not receive a plaintext agent API key.
Create, first publish and adopt all mint a key and used to return it, which
handed a deploy token a secret that outlives the token's own revocation.
"""
decoded = getattr(request, "decoded_token", None)
if not is_pat(decoded):
return True
return "agents:keys" in (decoded.get("scopes") or [])
def mask_agent_key(key: Optional[str]) -> str:
return f"{key[:4]}...{key[-4:]}" if key else ""
+264
View File
@@ -0,0 +1,264 @@
"""Personal access tokens: format, scope catalog and the per-request verifier.
A PAT is ``dgpt_pat_`` + 32 random bytes (urlsafe). Only its SHA-256 is stored
(``personal_access_tokens.token_hash``), the same shape as device session
tokens. Scopes and the resource filter are always read from the database row;
nothing about a token's authority is encoded in the credential itself.
"""
from __future__ import annotations
import hashlib
import logging
import secrets
import uuid
from datetime import datetime, timedelta, timezone
from typing import Any, Optional
from docsgpt.core.settings import settings
from docsgpt.storage.db.repositories.personal_access_tokens import (
PersonalAccessTokensRepository,
)
from docsgpt.storage.db.session import db_readonly, db_session
logger = logging.getLogger(__name__)
TOKEN_PREFIX = "dgpt_pat_"
# Characters of the secret kept in ``token_prefix`` so users can tell tokens apart.
_DISPLAY_CHARS = 6
AUTH_METHOD_PAT = "pat"
#: Every grantable scope with the description shown in the UI and docs.
SCOPES: dict[str, str] = {
"agents:read": "View agents, folders, guardrail events and export agent definitions",
"agents:write": "Create, update, delete, share and import (apply) agents and folders",
"agents:keys": "Regenerate agent API keys and read incoming webhook URLs",
"sources:read": "View sources, their files, chunks and ingestion task status",
"sources:write": "Upload, ingest, sync, edit and delete sources and chunks",
"prompts:read": "View prompts",
"prompts:write": "Create, update and delete prompts",
"tools:read": "View configured tools",
"tools:write": "Create, update and delete tools and MCP servers",
"models:read": "View available and custom models",
"models:write": "Create, update, test and delete custom models",
"workflows:read": "View workflows",
"workflows:write": "Create, update and delete workflows",
"schedules:read": "View agent schedules and their runs",
"schedules:write": "Create, update, run and delete agent schedules",
"conversations:read": "View conversations and messages",
"conversations:write": "Rename, delete and give feedback on conversations",
"analytics:read": "View usage analytics and logs",
"teams:read": "View teams, members and resource shares",
"chat:run": "Ask agents and search sources (answer, stream, search); used for benchmarking",
}
#: Resource families whose tokens can be narrowed to specific ids.
FILTERABLE_FAMILIES = ("agents", "sources", "prompts", "tools", "workflows")
_MAX_FILTER_IDS = 200
def auth_type_supports_pats() -> bool:
"""PATs bind to a stable user id, which simple_jwt/session_jwt don't have."""
return bool(settings.PAT_ENABLED) and settings.AUTH_TYPE in (None, "oidc")
def generate_token() -> tuple[str, str, str]:
"""Mint a token. Returns ``(plaintext, sha256_hex, display_prefix)``."""
secret = secrets.token_urlsafe(32)
token = TOKEN_PREFIX + secret
return token, hash_token(token), TOKEN_PREFIX + secret[:_DISPLAY_CHARS]
def hash_token(token: str) -> str:
return hashlib.sha256(token.encode("utf-8")).hexdigest()
def looks_like_pat(value: Optional[str]) -> bool:
return bool(value) and value.startswith(TOKEN_PREFIX)
def redact(value: Optional[str]) -> str:
"""Log-safe form of a credential: the display prefix only."""
if not value:
return ""
if looks_like_pat(value):
return value[: len(TOKEN_PREFIX) + _DISPLAY_CHARS] + "…"
return value[:4] + "…"
def expand_scopes(scopes) -> set[str]:
"""Granted scopes plus what they imply (``x:write`` includes ``x:read``)."""
granted = set(scopes or [])
for scope in list(granted):
family, _, action = scope.partition(":")
if action == "write" and f"{family}:read" in SCOPES:
granted.add(f"{family}:read")
return granted
def normalize_scopes(raw: Any) -> list[str]:
"""Validate a requested scope list. Raises ``ValueError`` with a user-facing message."""
if not isinstance(raw, list) or not raw:
raise ValueError("scopes must be a non-empty list")
unknown = sorted({s for s in raw if not isinstance(s, str) or s not in SCOPES}, key=str)
if unknown:
raise ValueError(f"Unknown scopes: {', '.join(map(str, unknown))}")
return sorted(set(raw))
def normalize_resource_filter(raw: Any, scopes: list[str]) -> dict[str, list[str]]:
"""Validate ``{"<family>": ["<uuid>", ...]}``. Raises ``ValueError`` with a user-facing message.
A family may only be restricted when the token holds a scope in it;
otherwise the restriction would be dead weight that reads as protection.
"""
if raw in (None, {}):
return {}
if not isinstance(raw, dict):
raise ValueError("resource_filter must be an object")
families = {s.partition(":")[0] for s in scopes}
# chat:run acts on agents and sources, so both may be restricted alongside it.
if "chat" in families:
families.update({"agents", "sources"})
if "tools" in raw and "chat:run" in scopes:
# Chat executes tools (an agent's own, or the user's defaults), which cannot be held to an allowlist.
raise ValueError("resource_filter.tools cannot be combined with the chat:run scope")
out: dict[str, list[str]] = {}
for family, ids in raw.items():
if family not in FILTERABLE_FAMILIES:
raise ValueError(
f"resource_filter supports only: {', '.join(FILTERABLE_FAMILIES)}"
)
if family not in families:
raise ValueError(f"resource_filter.{family} needs a {family} scope on the token")
if not isinstance(ids, list) or not ids:
raise ValueError(f"resource_filter.{family} must be a non-empty list of ids")
if len(ids) > _MAX_FILTER_IDS:
raise ValueError(f"resource_filter.{family} allows at most {_MAX_FILTER_IDS} ids")
normalized = []
for value in ids:
try:
normalized.append(str(uuid.UUID(str(value))))
except (ValueError, AttributeError, TypeError):
raise ValueError(f"resource_filter.{family} contains an invalid id: {value!r}")
out[family] = sorted(set(normalized))
return out
def resolve_expiry(expires_in_days: Any) -> Optional[datetime]:
"""Map the requested lifetime to ``expires_at``. Raises ``ValueError`` with a user-facing message.
``None`` means "use the default"; ``0`` asks for a non-expiring token,
which only an operator setting can allow.
"""
if expires_in_days is None:
days = settings.PAT_DEFAULT_LIFETIME_DAYS
elif isinstance(expires_in_days, bool) or not isinstance(expires_in_days, int):
raise ValueError("expires_in_days must be an integer")
elif expires_in_days == 0:
if not settings.PAT_ALLOW_NON_EXPIRING:
raise ValueError("Non-expiring tokens are disabled on this server")
return None
elif expires_in_days < 0:
raise ValueError("expires_in_days must be positive")
else:
days = expires_in_days
if days > settings.PAT_MAX_LIFETIME_DAYS:
raise ValueError(f"expires_in_days must not exceed {settings.PAT_MAX_LIFETIME_DAYS}")
return datetime.now(timezone.utc) + timedelta(days=days)
def _parse_moment(value: Any) -> Optional[datetime]:
if not value:
return None
try:
moment = value if isinstance(value, datetime) else datetime.fromisoformat(str(value))
except ValueError:
return None
return moment if moment.tzinfo else moment.replace(tzinfo=timezone.utc)
def renewal_lifetime_days(row: dict) -> Optional[int]:
"""The lifetime a token was last issued with, for renewing it on the same terms.
``0`` for a non-expiring token, ``None`` when it cannot be derived (the
caller then falls back to the default). The result is clamped to today's
maximum, since the policy may have tightened since the token was issued.
"""
issued = _parse_moment(row.get("regenerated_at")) or _parse_moment(row.get("created_at"))
expires = _parse_moment(row.get("expires_at"))
if expires is None:
return 0 if settings.PAT_ALLOW_NON_EXPIRING else None
if issued is None:
return None
days = round((expires - issued).total_seconds() / 86400)
return max(1, min(days, settings.PAT_MAX_LIFETIME_DAYS))
def _client_ip(request) -> Optional[str]:
# Flask exposes remote_addr; Starlette exposes client.host.
ip = getattr(request, "remote_addr", None)
if ip:
return ip
client = getattr(request, "client", None)
return getattr(client, "host", None)
_TOUCH_INTERVAL_SECONDS = 60
def _usage_is_stale(last_used_at: Any) -> bool:
"""True when ``last_used_at`` is old enough to be worth a write transaction."""
if not last_used_at:
return True
try:
seen = last_used_at if isinstance(last_used_at, datetime) else datetime.fromisoformat(str(last_used_at))
except ValueError:
return True
if seen.tzinfo is None:
seen = seen.replace(tzinfo=timezone.utc)
return (datetime.now(timezone.utc) - seen).total_seconds() >= _TOUCH_INTERVAL_SECONDS
_INVALID = {"message": "Authentication error: invalid token", "error": "invalid_token"}
def authenticate_pat(token: str, request) -> dict:
"""Resolve a PAT into the claims dict the rest of the app reads.
Fails closed: an unknown, revoked or expired token, a disabled feature or
a database error all yield the same ``invalid_token`` error.
"""
if not auth_type_supports_pats():
return dict(_INVALID)
try:
with db_readonly() as conn:
row = PersonalAccessTokensRepository(conn).find_active_by_hash(hash_token(token))
except Exception:
logger.error("PAT lookup failed for %s", redact(token), exc_info=True)
return dict(_INVALID)
if not row:
logger.warning("Rejected personal access token %s", redact(token))
return dict(_INVALID)
if _usage_is_stale(row.get("last_used_at")):
try:
with db_session() as conn:
PersonalAccessTokensRepository(conn).touch_last_used(
str(row["id"]), _client_ip(request), min_interval_seconds=_TOUCH_INTERVAL_SECONDS
)
except Exception:
# Usage telemetry must never fail a request.
logger.debug("PAT last-used update failed", exc_info=True)
return {
"sub": row["user_id"],
"auth_method": AUTH_METHOD_PAT,
"pat_id": str(row["id"]),
"pat_name": row["name"],
"scopes": sorted(expand_scopes(row.get("scopes"))),
"resource_filter": row.get("resource_filter") or {},
}
def is_pat(decoded_token: Optional[dict]) -> bool:
return bool(decoded_token) and decoded_token.get("auth_method") == AUTH_METHOD_PAT
+31
View File
@@ -36,6 +36,7 @@ from docsgpt.agents.default_tools import (
synthesized_tool_name_for_id,
)
from docsgpt.api import api
from docsgpt.api.pat.rules import allowed_ids
from docsgpt.core.model_utils import validate_model_id
from docsgpt.core.url_validation import SSRFError, validate_url
from docsgpt.security.safe_url import UnsafeUserUrlError, validate_user_base_url
@@ -1717,6 +1718,32 @@ def _read_import_payload(req):
return raw.decode("utf-8", "replace"), {}
def _restricted_token_denial(conn, user: str, doc: dict) -> Optional[str]:
"""Why a resource-restricted personal access token may not import ``doc``, if it may not.
An import resolves sources, prompts and tools by name and may create them,
so a token restricted on any of those families cannot be held to its
allowlist here. A token restricted to specific agents may update exactly
those; it can never create one.
"""
for family in ("sources", "prompts", "tools", "workflows"):
if allowed_ids(request, family) is not None:
return f"A token restricted to specific {family} cannot import agents"
allowed_agents = allowed_ids(request, "agents")
if allowed_agents is None:
return None
target = _resolve_target(conn, user, doc.get("metadata") or {})
if target["action"] != "update" or target["agent_id"] not in allowed_agents:
return "This token is restricted to specific agents and may only update those"
return None
def _token_denied_response(reason: str):
return make_response(
jsonify({"success": False, "error": "resource_not_allowed", "message": reason}), 403
)
@agents_portability_ns.route("/export_agent")
class ExportAgent(Resource):
@api.doc(params={"id": "Agent ID"}, description="Export an agent as YAML")
@@ -1763,6 +1790,8 @@ class ImportAgentPlan(Resource):
return make_response(jsonify({"success": False, "message": str(exc)}), 400)
try:
with db_readonly() as conn:
if reason := _restricted_token_denial(conn, user, doc):
return _token_denied_response(reason)
plan = plan_import(conn, user, doc)
except Exception:
current_app.logger.error("Agent import plan failed", exc_info=True)
@@ -1786,6 +1815,8 @@ class ImportAgent(Resource):
return make_response(jsonify({"success": False, "message": str(exc)}), 400)
try:
with db_session() as conn:
if reason := _restricted_token_denial(conn, user, doc):
return _token_denied_response(reason)
result = apply_import(conn, user, doc, resolution)
except AgentImportError as exc:
# Apply-time rejection of the user's document (e.g. the workflow
+13 -4
View File
@@ -9,6 +9,7 @@ from flask_restx import fields, Namespace, Resource
from pydantic import ValidationError as PydanticValidationError
from docsgpt.api import api
from docsgpt.api.pat.rules import filter_listing, mask_agent_key, may_see_agent_keys
from docsgpt.guardrails.config import AgentConfig
from docsgpt.api.user.base import (
copy_agent_image_for_user,
@@ -498,7 +499,7 @@ class GetAgents(Resource):
except Exception as err:
current_app.logger.error(f"Error retrieving agents: {err}", exc_info=True)
return make_response(jsonify({"success": False}), 400)
return make_response(jsonify(list_agents), 200)
return make_response(jsonify(filter_listing(request, "agents", list_agents)), 200)
@agents_ns.route("/create_agent")
@@ -762,7 +763,9 @@ class CreateAgent(Resource):
except Exception as err:
current_app.logger.error(f"Error creating agent: {err}", exc_info=True)
return make_response(jsonify({"success": False}), 400)
return make_response(jsonify({"id": new_id, "key": key}), 201)
# A token without agents:keys never receives the plaintext agent key.
visible_key = key if may_see_agent_keys(request) else mask_agent_key(key)
return make_response(jsonify({"id": new_id, "key": visible_key}), 201)
@agents_ns.route("/update_agent/<string:agent_id>")
@@ -1306,7 +1309,11 @@ class UpdateAgent(Resource):
"message": "Agent updated successfully",
}
if newly_generated_key:
response_data["key"] = newly_generated_key
response_data["key"] = (
newly_generated_key
if may_see_agent_keys(request)
else mask_agent_key(newly_generated_key)
)
return make_response(jsonify(response_data), 200)
@@ -1810,7 +1817,9 @@ class AdoptAgent(Resource):
)
response_agent = _format_agent_output(new_agent, include_key_masked=False)
response_agent["key"] = new_key
response_agent["key"] = (
new_key if may_see_agent_keys(request) else mask_agent_key(new_key)
)
return make_response(
jsonify({"success": True, "agent": response_agent}), 200
)
+1 -1
View File
@@ -154,7 +154,7 @@ async def download_artifact(request: Request) -> Response:
# URL. If the active backend can't mint one, that's a config error:
# surface a 500 rather than silently proxying bytes from a backend
# the operator expected to be off the hot path.
if getattr(settings, "URL_STRATEGY", "backend") == "s3":
if settings.URL_STRATEGY == "s3":
try:
url = await anyio.to_thread.run_sync(
partial(storage.generate_presigned_url, storage_path, expires_in=_PRESIGNED_URL_TTL)
+2
View File
@@ -688,6 +688,8 @@ class LiveSpeechToTextFinish(Resource):
jsonify({"success": False, "message": "Authentication required"}),
401,
)
if not STTCreator.is_enabled(settings.STT_PROVIDER):
return _feature_disabled(_STT_DISABLED_MESSAGE)
redis_client = _require_live_stt_redis()
if hasattr(redis_client, "status_code"):
+1 -1
View File
@@ -235,7 +235,7 @@ def get_vector_store(source_id):
store = VectorCreator.create_vectorstore(
settings.VECTOR_STORE,
source_id=source_id,
embeddings_key=os.getenv("EMBEDDINGS_KEY"),
embeddings_key=settings.EMBEDDINGS_KEY,
)
return store
+26 -4
View File
@@ -9,6 +9,8 @@ import threading
import uuid
from typing import Any, Callable, Optional
from celery.exceptions import Ignore, MaxRetriesExceededError
from docsgpt.storage.db.repositories.idempotency import IdempotencyRepository
from docsgpt.storage.db.session import db_readonly, db_session
@@ -81,10 +83,30 @@ def with_idempotency(
"idempotency: live lease held; deferring task=%s key=%s",
task_name, key,
)
raise self.retry(
countdown=LEASE_TTL_SECONDS,
max_retries=LEASE_RETRY_MAX,
)
try:
raise self.retry(
countdown=LEASE_TTL_SECONDS,
max_retries=LEASE_RETRY_MAX,
)
except MaxRetriesExceededError:
# The holder is simply slower than LEASE_RETRY_MAX
# deferrals: a task that outruns the broker's visibility
# timeout is redelivered while its first run is still
# going. Letting the exhaustion propagate would report a
# failure for a task that is running normally — but so
# would returning a value, only less visibly. A redelivery
# reuses the original task id (``Context`` carries
# ``task_id`` into the retry), so a return marks the very
# id the client polls SUCCESS, and ``/api/task_status``
# hands that to the UI as a finished build. ``Ignore``
# records no state at all, leaving the outcome to the run
# that actually holds the lease.
logger.info(
"idempotency: lease still held after %s deferrals; "
"leaving task=%s key=%s to its holder",
LEASE_RETRY_MAX, task_name, key,
)
raise Ignore() from None
if attempt > MAX_TASK_ATTEMPTS:
logger.error(
+47
View File
@@ -5,6 +5,9 @@ only from ``request.decoded_token`` (already populated and role-resolved by the
auth chokepoint in ``app.py``). Auth-mode-agnostic. ``email``/``name``/
``picture`` are OIDC-only and optional — they are echoed from the token and are
never present for ``simple_jwt``/``session_jwt``/no-auth modes.
``GET /api/user/quota`` returns the caller's usage against the limits an admin
set for them, without naming the policies behind those limits.
"""
from __future__ import annotations
@@ -12,6 +15,10 @@ from __future__ import annotations
from flask import jsonify, make_response, request
from flask_restx import Namespace, Resource
from docsgpt.api.pat.tokens import is_pat
from docsgpt.core.settings import settings
from docsgpt.quotas.service import REQUEST_BUCKETS, QuotaService
me_ns = Namespace("me", description="Current user identity and roles", path="/api")
@@ -31,4 +38,44 @@ class MeResource(Resource):
value = decoded_token.get(field)
if value:
body[field] = value
if is_pat(decoded_token):
# Lets a CLI or pipeline confirm what its token is allowed to do.
body["auth_method"] = "pat"
body["token"] = {
"id": decoded_token.get("pat_id"),
"name": decoded_token.get("pat_name"),
"scopes": decoded_token.get("scopes") or [],
"resource_filter": decoded_token.get("resource_filter") or {},
}
return make_response(jsonify(body), 200)
def _own_budget(budget: dict) -> dict:
return {"limit": budget["limit"], "used": budget["used"]}
@me_ns.route("/user/quota")
class MyQuotaResource(Resource):
def get(self):
"""Return the caller's limited buckets: ``{bucket, tokens, cost, resets_at}`` each."""
decoded_token = getattr(request, "decoded_token", None)
user_id = decoded_token.get("sub") if decoded_token else None
if not user_id:
return make_response(jsonify({"success": False}), 401)
statuses = QuotaService.status(user_id, ("all", *REQUEST_BUCKETS))
buckets = []
for status in statuses:
if status.limits.unlimited:
continue
data = status.to_dict()
buckets.append(
{
"bucket": data["bucket"],
"tokens": _own_budget(data["tokens"]),
"cost": _own_budget(data["cost"]),
"resets_at": data["resets_at"],
}
)
return make_response(
jsonify({"success": True, "period": settings.QUOTA_PERIOD, "buckets": buckets}), 200
)
+3 -1
View File
@@ -5,6 +5,7 @@ from flask import current_app, jsonify, make_response, request
from flask_restx import fields, Namespace, Resource
from docsgpt.api import api
from docsgpt.api.pat.rules import filter_listing
from docsgpt.api.user.team_sharing import team_access_for, visible_with_access
from docsgpt.storage.db.repositories.prompts import PromptsRepository
from docsgpt.prompts.composer import compose_preset, is_composed_preset
@@ -91,7 +92,8 @@ class GetPrompts(Resource):
except Exception as err:
current_app.logger.error(f"Error retrieving prompts: {err}", exc_info=True)
return make_response(jsonify({"success": False}), 400)
return make_response(jsonify(list_prompts), 200)
# Presets (default/creative/strict) have no row id and stay visible to a restricted token.
return make_response(jsonify(filter_listing(request, "prompts", list_prompts)), 200)
@prompts_ns.route("/get_single_prompt")
+6
View File
@@ -17,6 +17,7 @@ from sqlalchemy import text as sql_text
from docsgpt.agents.headless_runner import run_agent_headless
from docsgpt.core.settings import settings
from docsgpt.events.publisher import publish_user_event
from docsgpt.quotas.service import QuotaExceededError
from docsgpt.storage.db.base_repository import row_to_dict
from docsgpt.storage.db.engine import get_engine
from docsgpt.storage.db.repositories.conversations import (
@@ -282,6 +283,11 @@ def execute_scheduled_run_body(run_id: str, celery_task_id: Optional[str]) -> Di
outcome = {"answer": "", "tool_calls": [], "sources": [], "thought": ""}
error_type = "timeout"
error_text = "run exceeded soft time limit"
except QuotaExceededError as exc:
# The owner's usage quota is spent; the run never started.
outcome = {"answer": "", "tool_calls": [], "sources": [], "thought": ""}
error_type = "budget_exceeded"
error_text = str(exc)
except Exception as exc:
outcome = {"answer": "", "tool_calls": [], "sources": [], "thought": ""}
error_type = "agent_error"
+2 -1
View File
@@ -10,6 +10,7 @@ from pydantic import ValidationError
from docsgpt.agents.tools.path_utils import validate_tool_path
from docsgpt.api import api
from docsgpt.api.pat.rules import filter_listing
from docsgpt.api.user.tasks import (
convert_source_to_wiki,
extract_graph,
@@ -130,7 +131,7 @@ class CombinedJson(Resource):
except Exception as err:
current_app.logger.error(f"Error retrieving sources: {err}", exc_info=True)
return make_response(jsonify({"success": False}), 400)
return make_response(jsonify(data), 200)
return make_response(jsonify(filter_listing(request, "sources", data)), 200)
@sources_ns.route("/sources/paginated")
+3 -3
View File
@@ -346,9 +346,9 @@ def parse_timeout_for_size(size_bytes: Optional[int]) -> float:
"""
from docsgpt.core.settings import settings
base = float(getattr(settings, "DOCUMENT_PARSE_TIMEOUT", 120) or 120)
per_mib = float(getattr(settings, "DOCUMENT_PARSE_TIMEOUT_PER_MB", 0) or 0)
ceiling = float(getattr(settings, "DOCUMENT_PARSE_TIMEOUT_MAX", base) or base)
base = float(settings.DOCUMENT_PARSE_TIMEOUT or 120)
per_mib = float(settings.DOCUMENT_PARSE_TIMEOUT_PER_MB or 0)
ceiling = float(settings.DOCUMENT_PARSE_TIMEOUT_MAX or base)
size = float(size_bytes) if isinstance(size_bytes, (int, float)) else 0.0
scaled = base + per_mib * max(size, 0.0) / (1024 * 1024)
return min(ceiling, max(base, scaled))
+4
View File
@@ -16,6 +16,7 @@ from docsgpt.agents.default_tools import (
from docsgpt.agents.tools.spec_parser import parse_spec
from docsgpt.agents.tools.tool_manager import ToolManager
from docsgpt.api import api
from docsgpt.api.pat.rules import filter_listing
from docsgpt.api.user.artifacts.authz import Principal, authorize_artifact
from docsgpt.api.user.team_sharing import effective_write_owner, visible_with_access
from docsgpt.core.settings import settings
@@ -294,6 +295,9 @@ class GetTools(Resource):
builtin_copy.get("name") in WORKFLOW_ONLY_BUILTINS
)
user_tools.append(builtin_copy)
# A resource-restricted token sees only its allowed tools. Default
# and builtin rows have ids too, so they follow the same allowlist.
user_tools = filter_listing(request, "tools", user_tools)
except Exception as err:
current_app.logger.error(f"Error getting user tools: {err}", exc_info=True)
return make_response(jsonify({"success": False}), 400)
+9 -1
View File
@@ -258,6 +258,8 @@ def chat_completions():
try:
processor = StreamProcessor(internal_data, decoded_token)
# Set when this request took the resume claim, so a refusal can release it.
claimed_conversation_id = None
if internal_data.get("tool_actions"):
conversation_id = internal_data.get("conversation_id")
@@ -282,6 +284,7 @@ def chat_completions():
claimed_state=pending_state,
)
processor.conversation_id = conversation_id
claimed_conversation_id = conversation_id
else:
# Compatibility fallback for old/completed conversations and
# clients that resend the full transcript without resumable
@@ -338,7 +341,12 @@ def chat_completions():
)
helper = _V1AnswerHelper()
usage_error = helper.check_usage(processor.agent_config)
if claimed_conversation_id:
usage_error = helper.check_usage_on_resume(processor, claimed_conversation_id)
else:
usage_error = helper.check_usage(
processor.agent_config, processor.decoded_token, agent_id=processor.agent_id
)
if usage_error:
return usage_error
+19 -11
View File
@@ -1,5 +1,4 @@
import logging
import os
import platform
import uuid
@@ -23,8 +22,11 @@ from docsgpt.api.devices import devices_bp # noqa: E402
from docsgpt.api.internal.routes import internal # noqa: E402
from docsgpt.api.oidc import oidc_bp # noqa: E402
from docsgpt.api.oidc.denylist import is_denied as oidc_session_denied # noqa: E402
from docsgpt.api.pat.routes import pat_ns # noqa: E402
from docsgpt.api.pat.rules import authorize as authorize_pat # noqa: E402
from docsgpt.api.pat.tokens import is_pat # noqa: E402
from docsgpt.api.scim import scim_bp # noqa: E402
from docsgpt.api.user.authz import resolve_roles # noqa: E402
from docsgpt.api.user.authz import ROLE_USER, resolve_roles # noqa: E402
from docsgpt.api.user.routes import user # noqa: E402
from docsgpt.api.connector.routes import connector # noqa: E402
from docsgpt.api.v1 import v1_bp # noqa: E402
@@ -110,6 +112,8 @@ app.register_blueprint(v1_bp)
# first app and raise "add_url_rule can no longer be called".
if admin_ns not in api.namespaces:
api.add_namespace(admin_ns)
if pat_ns not in api.namespaces:
api.add_namespace(pat_ns)
app.config.update(
UPLOAD_FOLDER="inputs",
CELERY_BROKER_URL=settings.CELERY_BROKER_URL,
@@ -170,16 +174,8 @@ def enforce_document_upload_request_size_limit():
# only local development may use the atomic filesystem fallback.
settings.JWT_SECRET_KEY = resolve_jwt_secret_key(
settings.JWT_SECRET_KEY,
os.getenv("DEPLOYMENT_TYPE"),
settings.DEPLOYMENT_TYPE,
)
if settings.AUTH_TYPE == "oidc":
_missing_oidc = [
name
for name in ("OIDC_ISSUER", "OIDC_CLIENT_ID", "OIDC_FRONTEND_URL")
if not getattr(settings, name)
]
if _missing_oidc:
raise RuntimeError(f"AUTH_TYPE=oidc requires settings: {', '.join(_missing_oidc)}")
SIMPLE_JWT_TOKEN = None
if settings.AUTH_TYPE == "simple_jwt":
payload = {"sub": "local"}
@@ -326,6 +322,18 @@ def authenticate_request():
request.decoded_token = None
elif "error" in decoded_token:
return jsonify(decoded_token), 401
elif is_pat(decoded_token):
# Scopes and resource restrictions are enforced here, centrally and
# deny by default (docsgpt/api/pat/rules.py). A token never carries
# admin, whatever its owner holds, and the session denylist does not
# apply: the token lookup already excludes revoked tokens and
# deactivated users.
denied = authorize_pat(request, decoded_token)
if denied is not None:
body, status = denied
return jsonify(body), status
decoded_token["roles"] = [ROLE_USER]
request.decoded_token = decoded_token
elif settings.AUTH_TYPE == "oidc" and oidc_session_denied(decoded_token):
# Back-channel logout / SCIM deactivation revoked this session.
return (
+23
View File
@@ -4,7 +4,28 @@ from jose.exceptions import ExpiredSignatureError
from docsgpt.core.settings import settings
# Claims only the PAT verifier may set. Dropped from decoded JWTs so a session
# token can never present itself as a (differently scoped) personal access token.
_PAT_ONLY_CLAIMS = ("auth_method", "pat_id", "pat_name", "scopes", "resource_filter")
def _bearer_value(request):
header = request.headers.get("Authorization")
if not header or not isinstance(header, str):
return None
scheme, _, value = header.partition(" ")
return value.strip() if scheme.lower() == "bearer" and value else header.strip()
def handle_auth(request, data={}):
# Personal access tokens are opaque (not JWTs) and resolve against the
# database in every auth mode that supports them, including AUTH_TYPE unset.
from docsgpt.api.pat.tokens import authenticate_pat, looks_like_pat
bearer = _bearer_value(request)
if looks_like_pat(bearer):
return authenticate_pat(bearer, request)
if settings.AUTH_TYPE in ["simple_jwt", "session_jwt", "oidc"]:
jwt_token = request.headers.get("Authorization")
if not jwt_token:
@@ -26,6 +47,8 @@ def handle_auth(request, data={}):
# requirement is scoped to oidc.
options={"verify_exp": is_oidc, "require_exp": is_oidc},
)
for claim in _PAT_ONLY_CLAIMS:
decoded_token.pop(claim, None)
return decoded_token
except ExpiredSignatureError:
return {
+46
View File
@@ -13,6 +13,7 @@ from celery.signals import (
setup_logging,
task_postrun,
task_prerun,
worker_init,
worker_process_init,
worker_ready,
)
@@ -172,6 +173,51 @@ def _run_version_check(*args, **kwargs):
celery = make_celery()
celery.config_from_object("docsgpt.celeryconfig")
#: Set once this process starts as a worker; see :func:`_mark_worker_process`.
_IS_WORKER_PROCESS = False
@worker_init.connect
@worker_process_init.connect
def _mark_worker_process(*args, **kwargs):
"""Record that this process runs tasks, for :func:`in_worker`.
``worker_init`` fires in every worker's main process before its pool
starts: that is where solo, threads, eventlet and gevent run tasks, and
what prefork children fork from. ``worker_process_init`` covers prefork
children however they were started.
"""
global _IS_WORKER_PROCESS
_IS_WORKER_PROCESS = True
def in_worker() -> bool:
"""True anywhere in a Celery worker process, on any thread or greenlet.
``current_worker_task`` alone is not enough: Celery records the executing
task on the thread (or greenlet) that runs it, so one the task starts sees
none and would take the web-process branch — dispatching to the worker it
is running in and blocking on the result. Celery refuses that ``get()``
("Never call result.get() within a task!"), or, where joins are allowed,
it waits on a queue that only this busy process may be able to serve.
The worker's own startup (:func:`_mark_worker_process`) answers for every
pool. ``task_join_will_block`` — process-wide, set for every blocking pool
— and the task's own ``current_worker_task`` still count for a process
that runs tasks without having gone through that startup.
Returns:
bool: Whether this call is running inside a worker process.
"""
from celery.result import task_join_will_block
return (
_IS_WORKER_PROCESS
or task_join_will_block()
or celery.current_worker_task is not None
)
#: Task-name prefix the package carried before the rename to ``docsgpt``.
+149 -6
View File
@@ -1,4 +1,5 @@
"""The ``docsgpt`` command: run the API, the worker and the maintenance scripts.
"""The ``docsgpt`` command: run the API, the worker and the maintenance scripts,
or run and manage DocsGPT on Docker (``docsgpt up``).
Every subcommand imports what it needs when it runs, so ``docsgpt --help``
stays instant and does not touch the database.
@@ -19,10 +20,24 @@ DEFAULT_PORT = 7091
def _announce_home() -> None:
"""Say where runtime data and the env file come from; the API and the worker must agree."""
from docsgpt.core.paths import env_file, home_dir
"""Create the data home and say where data and the env file come from; the API and the worker must agree."""
from pathlib import Path
print(f"docsgpt: data home {home_dir()} (env file {env_file()})", file=sys.stderr)
from docsgpt.core import paths
home = paths.home_dir()
home.mkdir(parents=True, exist_ok=True)
env = paths.env_file()
print(f"docsgpt: data home {home} (env file {env})", file=sys.stderr)
# Up to 0.20 an installed package used the working directory as its home.
chosen = os.environ.get(paths.HOME_ENV) or os.environ.get(paths.ENV_FILE_ENV) or paths.checkout_root()
stray = Path.cwd() / ".env"
if not chosen and stray.is_file() and stray.resolve() != env.resolve():
print(
f"docsgpt: {stray} is not used; settings come from {env}. "
f"Move the file there, or set DOCSGPT_HOME={Path.cwd()} to keep using this directory.",
file=sys.stderr,
)
def _gunicorn_options(host: str, port: int, workers: int) -> dict:
@@ -74,7 +89,17 @@ def _api(args: argparse.Namespace) -> int:
if args.reload or sys.platform == "win32":
import uvicorn
uvicorn.run("docsgpt.asgi:asgi_app", host=args.host, port=args.port, reload=args.reload)
from docsgpt.core.paths import package_dir
# Watch the package, not the working directory: a checkout also holds .venv, node_modules
# and the data the app writes (indexes/, inputs/), which restarts the server mid-ingest.
uvicorn.run(
"docsgpt.asgi:asgi_app",
host=args.host,
port=args.port,
reload=args.reload,
reload_dirs=[str(package_dir())] if args.reload else None,
)
return 0
_gunicorn_application(_gunicorn_options(args.host, args.port, args.workers)).run()
@@ -139,6 +164,17 @@ def _migrate(args: argparse.Namespace) -> int:
return 0
def _deploy(name: str):
"""A subcommand handler that imports ``docsgpt.deploy.commands`` only when it runs."""
def handler(args: argparse.Namespace, context=None) -> int:
from docsgpt.deploy import commands
return getattr(commands, name)(args, context)
return handler
# Maintenance scripts keep their own argument parsers; the command hands
# everything after the script name to them untouched (argparse would try to
# interpret the options itself).
@@ -155,11 +191,107 @@ def _run_script(module: str, argv: list[str]) -> int:
return int(importlib.import_module(f"docsgpt.scripts.{module}").main(argv) or 0)
def _add_deploy_commands(commands) -> None:
"""``docsgpt up`` and the commands that manage the Docker stack it runs."""
from docsgpt.deploy.stack import EXPOSURES, PROVIDERS
def stack_command(name: str, handler: str, help_text: str) -> argparse.ArgumentParser:
parser = commands.add_parser(name, help=help_text)
parser.add_argument("--dir", help="stack directory (default: DOCSGPT_HOME, else ~/.docsgpt/server)")
parser.set_defaults(func=_deploy(handler), deploy=True)
return parser
up = stack_command("up", "up", "install or update DocsGPT on Docker and start it")
up.add_argument("--expose", choices=EXPOSURES, help="who can reach it: local (default), network or domain")
up.add_argument("--domain", help="public domain served over HTTPS by Caddy (implies --expose domain)")
up.add_argument("--port", type=int, help="host port for the UI and API (default: 7091)")
up.add_argument("--provider", choices=list(PROVIDERS), help="model provider (default: the DocsGPT public API)")
up.add_argument("--api-key", help="the provider's API key (or set DOCSGPT_API_KEY)")
up.add_argument("--model", help="model name (required for openai-compatible)")
up.add_argument("--base-url", help="base URL of an OpenAI-compatible server")
docling = up.add_mutually_exclusive_group()
docling.add_argument("--docling", dest="docling", action="store_const", const=True,
help="run the image with the docling parser engine and OCR (several GB larger)")
docling.add_argument("--no-docling", dest="docling", action="store_const", const=False,
help="go back to the default image")
up.set_defaults(docling=None)
up.add_argument("--image-tag", help="image tag to run instead of this package's version, e.g. develop")
up.add_argument("--native", action="store_true",
help="run the API and worker as services on this machine instead of on Docker")
up.add_argument("--postgres-uri", help="native mode: the PostgreSQL DocsGPT should use")
up.add_argument("--redis-url", help="native mode: the Redis for the queue and the cache (default: localhost:6379)")
up.add_argument("-y", "--yes", action="store_true", help="ask nothing: use the flags, then the defaults")
up.add_argument("--reconfigure", action="store_true", help="ask the setup questions again")
up.add_argument("--adopt", action="store_true", help="take over a DocsGPT stack started from another folder")
up.add_argument("--no-open", action="store_true", help="do not open the browser after the first install")
up.add_argument("--timeout", type=int, default=300, help="seconds to wait for the API to answer (default: 300)")
stack_command("down", "down", "stop the Docker stack (data and settings stay)")
stack_command("status", "status", "show the stack's version, address, containers and health")
logs = stack_command("logs", "logs", "show the stack's logs")
logs.add_argument("-f", "--follow", action="store_true", help="keep printing new lines")
logs.add_argument("--tail", type=int, help="only the last N lines of each service")
logs.add_argument("services", nargs="*", help="services to show, e.g. backend worker")
stack_command("token", "token", "print the access token (installs reachable beyond this computer)")
stack_command("open", "open_ui", "open DocsGPT in the browser")
upgrade = stack_command("upgrade", "upgrade", "upgrade the package and restart the stack on the new version")
upgrade.add_argument("--version", help="version to install (default: the latest release)")
uninstall = stack_command("uninstall", "uninstall", "remove the Docker stack")
uninstall.add_argument("-y", "--yes", action="store_true", help="do not ask for confirmation")
uninstall.add_argument("--purge", action="store_true", help="also delete the settings and all data")
backup = stack_command("backup", "backup", "write a backup of the database and the uploaded data")
backup.add_argument("--out", help="directory for the archive (default: <stack>/backups)")
backup.add_argument("--with-settings", action="store_true",
help="include .env in the archive; it holds this install's secrets")
restore = stack_command("restore", "restore", "restore a backup over this install")
restore.add_argument("archive", help="the .tar.gz written by `docsgpt backup`")
restore.add_argument("-y", "--yes", action="store_true", help="do not ask for confirmation")
restore.add_argument("--force", action="store_true", help="restore a backup taken with a newer DocsGPT")
restore.add_argument("--timeout", type=int, default=300,
help="seconds to wait for the API afterwards (default: 300)")
doctor = stack_command("doctor", "doctor", "check what this machine needs to run DocsGPT")
doctor.add_argument("--postgres-uri", help="check this database instead of the one in .env")
doctor.add_argument("--redis-url", help="check this Redis instead of the one in .env")
restart = stack_command("restart", "restart", "restart the services, changing nothing else")
restart.add_argument("services", nargs="*", help="services to restart, e.g. api worker")
env = stack_command("env", "env", "show, get or set the stack's settings")
env_actions = env.add_subparsers(dest="env_action", metavar="<action>")
get = env_actions.add_parser("get", help="print one setting")
get.add_argument("key")
set_ = env_actions.add_parser("set", help="set settings (KEY=VALUE ...)")
set_.add_argument("pairs", nargs="+", metavar="KEY=VALUE")
set_.add_argument("--no-restart", dest="restart", action="store_false",
help="do not restart a running native install afterwards")
dev = commands.add_parser("dev", help="run this checkout's API, worker and UI with reload")
dev.add_argument("--host", default=DEFAULT_HOST, help="interface for the API (default: localhost)")
dev.add_argument("--port", type=int, default=DEFAULT_PORT, help="port for the API (default: 7091)")
dev.add_argument("--ui", action="store_true", help="also run the frontend dev server")
dev.add_argument("--mock-llm", action="store_true",
help="run the mock LLM and point DocsGPT at it, so no API key is needed")
dev.add_argument("--no-worker", dest="worker", action="store_false", help="do not run the Celery worker")
dev.add_argument("--no-reload", dest="reload", action="store_false",
help="do not restart the API and worker when a file changes")
dev.add_argument("-l", "--loglevel", default="INFO", help="worker log level (default: INFO)")
dev.set_defaults(func=_deploy("dev"), deploy=True)
def build_parser() -> argparse.ArgumentParser:
parser = argparse.ArgumentParser(prog="docsgpt", description="DocsGPT: private AI for agents, assistants and search.")
parser.add_argument("--version", action="version", version=f"docsgpt {__version__}")
commands = parser.add_subparsers(dest="command", metavar="<command>")
_add_deploy_commands(commands)
api = commands.add_parser("api", help="serve the HTTP API")
api.add_argument("--host", default=DEFAULT_HOST, help="interface to listen on (default: localhost; 0.0.0.0 for all)")
api.add_argument("--port", type=int, default=DEFAULT_PORT)
@@ -198,7 +330,18 @@ def main(argv: Optional[Sequence[str]] = None) -> int:
if not args.command:
parser.print_help()
return 2
return args.func(args)
if not getattr(args, "deploy", False):
return args.func(args)
from docsgpt.deploy.docker import DeployError
try:
return args.func(args)
except DeployError as exc:
print(f"docsgpt: {exc}", file=sys.stderr)
return 1
except KeyboardInterrupt:
print(file=sys.stderr)
return 130
if __name__ == "__main__":
+1 -1
View File
@@ -15,7 +15,7 @@ have to know which driver a given field feeds. Each normalizer also
silently upgrades the legacy ``postgresql+psycopg2://`` prefix since
psycopg2 is no longer in the project.
This module is deliberately separate from ``docsgpt/core/settings.py``
This module is deliberately separate from ``docsgpt/core/settings``
so the Settings class stays focused on field declarations, and the
URI-rewriting logic can be unit-tested without triggering ``.env``
file loading from importing Settings.
+1 -1
View File
@@ -140,7 +140,7 @@ class ModelRegistry:
from docsgpt.llm.providers import ALL_PROVIDERS
directories = [BUILTIN_MODELS_DIR]
operator_dir = getattr(settings, "MODELS_CONFIG_DIR", None)
operator_dir = settings.MODELS_CONFIG_DIR
if operator_dir:
op_path = Path(operator_dir)
if not op_path.exists():
+9 -2
View File
@@ -32,8 +32,15 @@ class ModelCapabilities:
supports_streaming: bool = True
supported_attachment_types: List[str] = field(default_factory=list)
context_window: int = 128000
input_cost_per_token: Optional[float] = None
output_cost_per_token: Optional[float] = None
# USD per 1M tokens; consumed by ``docsgpt/pricing.py``. ``None`` means
# "not declared": the call is recorded at $0 unless
# ``QUOTA_UNPRICED_RATE_PER_MILLION`` is set.
input_cost_per_million: Optional[float] = None
output_cost_per_million: Optional[float] = None
# Rates for the prompt-cache sub-bins of the prompt total. ``None`` bills
# those tokens at ``input_cost_per_million``.
cached_input_cost_per_million: Optional[float] = None
cache_write_cost_per_million: Optional[float] = None
# OpenAI reasoning-model effort hint (none/minimal/low/medium/high/xhigh;
# the accepted subset is model-dependent). Consumed by OpenAILLM — sent
# top-level on Chat Completions and nested under ``reasoning`` on the
+29 -5
View File
@@ -18,7 +18,7 @@ from pathlib import Path
from typing import Dict, List, Optional, Sequence
import yaml
from pydantic import BaseModel, ConfigDict, Field, field_validator
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
from docsgpt.core.model_settings import (
AvailableModel,
@@ -65,11 +65,33 @@ class _CapabilityFields(BaseModel):
supports_streaming: Optional[bool] = None
attachments: Optional[List[str]] = None
context_window: Optional[int] = None
input_cost_per_token: Optional[float] = None
output_cost_per_token: Optional[float] = None
input_cost_per_million: Optional[float] = Field(default=None, ge=0)
output_cost_per_million: Optional[float] = Field(default=None, ge=0)
cached_input_cost_per_million: Optional[float] = Field(default=None, ge=0)
cache_write_cost_per_million: Optional[float] = Field(default=None, ge=0)
reasoning_effort: Optional[str] = None
api_flavor: Optional[str] = None
@model_validator(mode="before")
@classmethod
def _per_token_alias(cls, data):
"""Accept the deprecated ``*_cost_per_token`` keys, scaled to per-1M."""
if not isinstance(data, dict):
return data
data = dict(data)
for side in ("input", "output"):
old, new = f"{side}_cost_per_token", f"{side}_cost_per_million"
if old not in data:
continue
value = data.pop(old)
if new in data:
raise ValueError(f"set only one of {old} and {new}")
if isinstance(value, bool) or not isinstance(value, (int, float)):
raise ValueError(f"{old} must be a number")
logger.warning("%s is deprecated; use %s (USD per 1M tokens)", old, new)
data[new] = value * 1_000_000
return data
@field_validator("reasoning_effort")
@classmethod
def _valid_reasoning_effort(cls, v: Optional[str]) -> Optional[str]:
@@ -237,8 +259,10 @@ def _build_model(
supports_streaming=pick("supports_streaming", True),
supported_attachment_types=expanded,
context_window=pick("context_window", 128000),
input_cost_per_token=pick("input_cost_per_token", None),
output_cost_per_token=pick("output_cost_per_token", None),
input_cost_per_million=pick("input_cost_per_million", None),
output_cost_per_million=pick("output_cost_per_million", None),
cached_input_cost_per_million=pick("cached_input_cost_per_million", None),
cache_write_cost_per_million=pick("cache_write_cost_per_million", None),
reasoning_effort=pick("reasoning_effort", None),
api_flavor=pick("api_flavor", "chat_completions"),
)
+4 -2
View File
@@ -108,8 +108,10 @@ defaults: # optional, applied to every model below
supports_streaming: bool # default true
attachments: [<alias-or-mime>, ...] # default []
context_window: int # default 128000
input_cost_per_token: float # default null
output_cost_per_token: float # default null
input_cost_per_million: float # USD per 1M prompt tokens; default null (unpriced)
output_cost_per_million: float # USD per 1M generated tokens; default null
cached_input_cost_per_million: float # prompt-cache reads; default: the input rate
cache_write_cost_per_million: float # prompt-cache writes; default: the input rate
reasoning_effort: <string> # default null; none|minimal|low|medium|high|xhigh (subset is model-dependent)
api_flavor: <string> # chat_completions (default) or responses
+12
View File
@@ -10,17 +10,29 @@ models:
description: Most capable Claude model for complex reasoning and agentic coding
context_window: 1000000
supports_structured_output: true
input_cost_per_million: 5.0
output_cost_per_million: 25.0
cached_input_cost_per_million: 0.5
cache_write_cost_per_million: 6.25
- id: claude-sonnet-4-6
display_name: Claude Sonnet 4.6
description: Best balance of speed and intelligence with extended thinking
context_window: 1000000
supports_structured_output: true
input_cost_per_million: 3.0
output_cost_per_million: 15.0
cached_input_cost_per_million: 0.3
cache_write_cost_per_million: 3.75
- id: claude-haiku-4-5
display_name: Claude Haiku 4.5
description: Fastest Claude model with near-frontier intelligence
supports_structured_output: true
input_cost_per_million: 1.0
output_cost_per_million: 5.0
cached_input_cost_per_million: 0.1
cache_write_cost_per_million: 1.25
- id: claude-fable-5
display_name: Claude Fable 5
+4
View File
@@ -12,7 +12,11 @@ models:
- id: deepseek-v4-flash
display_name: DeepSeek V4 Flash
description: Cost-efficient 1M-context model with hybrid thinking / non-thinking modes, tool calling and FIM completion
input_cost_per_million: 0.14
output_cost_per_million: 0.28
- id: deepseek-v4-pro
display_name: DeepSeek V4 Pro
description: Frontier 1M-context model with hybrid thinking / non-thinking modes for advanced reasoning and agentic coding
input_cost_per_million: 0.435
output_cost_per_million: 0.87
+3
View File
@@ -7,3 +7,6 @@ models:
supports_tools: true
attachments: [image]
context_window: 1048576
input_cost_per_million: 0.15
output_cost_per_million: 0.5
cached_input_cost_per_million: 0.03
+7
View File
@@ -9,9 +9,16 @@ models:
- id: gemini-3.1-pro-preview
display_name: Gemini 3.1 Pro (preview)
description: Most capable Gemini 3 model with advanced reasoning and agentic coding (preview)
# Priced at the >200k-token tier; long prompts are common with attachments.
input_cost_per_million: 4.0
output_cost_per_million: 18.0
- id: gemini-3.5-flash
display_name: Gemini 3.5 Flash
description: Frontier-class Flash for sustained performance on agentic and coding tasks
input_cost_per_million: 1.5
output_cost_per_million: 9
- id: gemini-3.1-flash-lite
display_name: Gemini 3.1 Flash-Lite
description: Cost-efficient frontier-class multimodal model for high-throughput workloads
input_cost_per_million: 0.25
output_cost_per_million: 1.5
+7
View File
@@ -8,9 +8,16 @@ models:
display_name: GPT-OSS 120B
description: OpenAI's open-weight 120B flagship served on Groq's LPU hardware; strong general reasoning with strict structured output support
supports_structured_output: true
input_cost_per_million: 0.15
output_cost_per_million: 0.6
cached_input_cost_per_million: 0.075
- id: llama-3.3-70b-versatile
display_name: Llama 3.3 70B Versatile
description: Meta's Llama 3.3 70B for general-purpose chat with parallel tool use
input_cost_per_million: 0.59
output_cost_per_million: 0.79
- id: llama-3.1-8b-instant
display_name: Llama 3.1 8B Instant
description: Small, very low-latency Llama model (~560 tok/s) with parallel tool use
input_cost_per_million: 0.05
output_cost_per_million: 0.08
+6
View File
@@ -8,14 +8,20 @@ models:
display_name: DeepSeek V4 Pro
description: 1.6T MoE (49B active) with 1M context, hybrid CSA/HCA attention, top-tier reasoning and agentic coding
context_window: 1048576
input_cost_per_million: 1.6
output_cost_per_million: 3.2
- id: moonshotai/kimi-k2.6
display_name: Kimi K2.6
description: 1T-parameter open-weight MoE with native vision/video, multi-step tool calling, and agentic long-horizon execution
attachments: [image]
context_window: 262144
input_cost_per_million: 0.8
output_cost_per_million: 3.4
- id: zai-org/glm-5
display_name: GLM-5
description: Z.AI 754B-parameter MoE with strong general reasoning, function calling, and structured output
context_window: 202800
input_cost_per_million: 1.0
output_cost_per_million: 3.2
Loaded 100 of 300 files, more files were not shown because too many files have changed in this diff. Show more