mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
docsgpt/core/settings.py had grown to 258 fields in one 600-line class, touched by about two commits a week, with related settings scattered (GitHub ingest caps inside the embeddings block, API keys in four places, the OpenAI Responses knobs 100 lines from the other OpenAI fields). It is now a package: one module per domain (auth, llm, embeddings, retrieval, vectorstores, database, workers, ingestion, ocr, storage, connectors, server, events, agents, guardrails, scheduler, sandbox, speech), each a SettingsGroup owning its fields and validators, composed by multiple inheritance into the same flat Settings class. Every attribute name, type, default, alias and constraint is unchanged, so settings.NAME reads, .env files and test monkeypatches all keep working; the import path docsgpt.core.settings is the package. Settings.normalize_api_key is kept as a classmethod for callers that reuse it. The comment above or beside each field became its Field(description=...), so the definitions are visible to tooling; the next commit generates the docs reference from them. Pitfall recorded for future groups: pydantic collects validators by method name across the MRO, so two groups naming a validator the same would silently keep only one. Each group's validator has a unique name.
41 lines
1.6 KiB
Python
41 lines
1.6 KiB
Python
"""Retrieval strategy and GraphRAG."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from typing import Optional
|
|
|
|
from pydantic import Field
|
|
|
|
from docsgpt.core.settings._shared import SettingsGroup
|
|
|
|
|
|
class RetrievalSettings(SettingsGroup):
|
|
"""Which vector store answers searches and how retrieval fans out across sources."""
|
|
|
|
VECTOR_STORE: str = Field(
|
|
default="faiss",
|
|
description="Vector store backend: faiss, elasticsearch, mongodb, qdrant, milvus or pgvector.",
|
|
)
|
|
RETRIEVERS_ENABLED: list = Field(
|
|
default=["classic", "default"],
|
|
description=(
|
|
"Retriever keys an agent may use; must match RetrieverCreator.retrievers registry keys, NOT the "
|
|
"legacy classic_rag label which never matched the registry."
|
|
),
|
|
)
|
|
RETRIEVAL_MAX_PARALLEL_SOURCES: int = Field(
|
|
default=4,
|
|
description="Concurrent per-source searches in one retrieval; the query is embedded once and shared.",
|
|
)
|
|
PER_SOURCE_RETRIEVAL_ENABLED: bool = Field(
|
|
default=True,
|
|
description="Kill-switch for per-source retrieval dispatch; False collapses to a single retriever.",
|
|
)
|
|
GRAPHRAG_ENABLED: bool = Field(default=False, description="Gates graph-aware ingestion and retrieval.")
|
|
GRAPHRAG_EXTRACTION_MODEL: Optional[str] = Field(
|
|
default=None, description="Model for ingest-time graph extraction; unset reuses LLM_PROVIDER/LLM_NAME."
|
|
)
|
|
GRAPHRAG_MAX_CHUNKS_FOR_EXTRACTION: int = Field(
|
|
default=2000, description="Hard cap on chunks extracted per source (cost control)."
|
|
)
|