mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
fix(rename): second review pass on the package rename
- The post-task reclaim skip recognises the legacy application.* embed name,
so query embeds queued by the previous release do not pay a full collect.
- The Azure compose file mounts host data on /app/{indexes,inputs,vectors},
where the process actually reads and writes; it mounted /app/application/...
before the rename and /app/docsgpt/... after it, and nothing wrote to either.
- The offline image check triggers on docsgpt/requirements*.txt again;
dependabot's pip entry points at docsgpt/.
- install_hint() and its docstring name docsgpt/requirements-<extra>.txt;
the test asserts the full path.
- application/vectors/ stays ignored: the compose files still mount it.
- The durability QA script quiets the docsgpt logger tree.
- Upgrade note: the three renamed source-sync entries start their timers
from the upgrade.
This commit is contained in:
1 parent
adb6963523
commit
2d9568ff79
10 files changed
+27
-20
No files matched your search
@@ -6,7 +6,7 @@
|
||||
version: 2
|
||||
updates:
|
||||
- package-ecosystem: "pip" # See documentation for possible values
|
||||
directory: "/application" # Location of package manifests
|
||||
directory: "/docsgpt" # Location of package manifests
|
||||
schedule:
|
||||
interval: "daily"
|
||||
- package-ecosystem: "npm" # See documentation for possible values
|
||||
|
||||
@@ -11,7 +11,7 @@ on:
|
||||
- 'docsgpt/Dockerfile'
|
||||
- '.dockerignore'
|
||||
- 'application/**'
|
||||
- 'application/requirements*.txt'
|
||||
- 'docsgpt/requirements*.txt'
|
||||
- 'docsgpt/scripts/prefetch_models.py'
|
||||
- 'docsgpt/scripts/verify_offline.py'
|
||||
- 'docsgpt/vectorstore/model_registry.py'
|
||||
|
||||
@@ -180,6 +180,7 @@ frontend/*.sln
|
||||
frontend/*.sw?
|
||||
|
||||
docsgpt/vectors/
|
||||
application/vectors/
|
||||
|
||||
**/inputs
|
||||
|
||||
|
||||
@@ -44,12 +44,14 @@ services:
|
||||
ports:
|
||||
- "7091:7091"
|
||||
volumes:
|
||||
# Host data stays under application/ (the old package directory) for this
|
||||
# release so an upgraded checkout keeps its indexes; it moves with the
|
||||
# packaging work, together with an upgrade note.
|
||||
- ../application/indexes:/app/docsgpt/indexes
|
||||
- ../application/inputs:/app/docsgpt/inputs
|
||||
- ../application/vectors:/app/docsgpt/vectors
|
||||
# Host data lives under application/ (the old package directory) for this
|
||||
# release, like the other compose files; it moves with the packaging work,
|
||||
# together with an upgrade note. The container side is /app: the process
|
||||
# runs there and resolves indexes/inputs/vectors relative to it (earlier
|
||||
# versions of this file mounted /app/application/..., which nothing wrote to).
|
||||
- ../application/indexes:/app/indexes
|
||||
- ../application/inputs:/app/inputs
|
||||
- ../application/vectors:/app/vectors
|
||||
depends_on:
|
||||
redis:
|
||||
condition: service_started
|
||||
|
||||
@@ -106,7 +106,9 @@ alias, so nothing breaks on upgrade, but update these before the alias goes:
|
||||
- Celery task names changed with the package (`docsgpt.api.user.tasks.ingest`
|
||||
and so on). A worker on this release also accepts the old names, so tasks
|
||||
queued before the upgrade still run, and beat rewrites the periodic
|
||||
schedule in Redis on start-up. Nothing to do.
|
||||
schedule in Redis on start-up. The daily, weekly and monthly source-sync
|
||||
timers restart from the upgrade, so the first sync after it can land later
|
||||
than it would have (a monthly sync by up to a month). Nothing to do.
|
||||
- Data directories do not move: the compose files keep your indexes, inputs
|
||||
and vectors under `application/` in the checkout, where they already are.
|
||||
|
||||
|
||||
@@ -121,6 +121,7 @@ def _trim_native_heap() -> None:
|
||||
# measured at ~86 ms for the collect against ~8 ms for the embed itself on a
|
||||
# worker holding the ONNX model, i.e. a 9x slowdown of the whole round trip.
|
||||
_NO_RECLAIM_TASKS = frozenset({"docsgpt.vectorstore.embeddings_tasks.embed_texts"})
|
||||
LEGACY_TASK_PREFIX = "application."
|
||||
|
||||
|
||||
@task_postrun.connect
|
||||
@@ -132,7 +133,10 @@ def _reclaim_memory_after_task(task=None, **kwargs):
|
||||
generational collect after a task that allocated a few kilobytes just
|
||||
charges the next task for walking the whole heap.
|
||||
"""
|
||||
if getattr(task, "name", None) in _NO_RECLAIM_TASKS:
|
||||
name = getattr(task, "name", None)
|
||||
if isinstance(name, str) and name.startswith(LEGACY_TASK_PREFIX):
|
||||
name = "docsgpt." + name[len(LEGACY_TASK_PREFIX):]
|
||||
if name in _NO_RECLAIM_TASKS:
|
||||
return
|
||||
gc.collect()
|
||||
torch = sys.modules.get("torch")
|
||||
@@ -169,7 +173,6 @@ celery = make_celery()
|
||||
celery.config_from_object("docsgpt.celeryconfig")
|
||||
|
||||
#: Task-name prefix the package carried before the rename to ``docsgpt``.
|
||||
LEGACY_TASK_PREFIX = "application."
|
||||
|
||||
|
||||
def register_legacy_task_names(app: Celery) -> int:
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""Optional dependency extras and the one place their install hints come from.
|
||||
|
||||
Heavy or niche packages are not installed by default. Each extra maps to a
|
||||
pyproject extra and to an exported ``application/requirements-<extra>.txt``,
|
||||
pyproject extra and to an exported ``docsgpt/requirements-<extra>.txt``,
|
||||
so a missing module can always be explained with the exact command to run.
|
||||
"""
|
||||
|
||||
@@ -28,7 +28,7 @@ _MODULE_TO_EXTRA: Dict[str, str] = {
|
||||
def install_hint(extra: str) -> str:
|
||||
"""Install command for ``extra``, for error messages and logs."""
|
||||
return (
|
||||
f"pip install -r application/requirements-{extra}.txt "
|
||||
f"pip install -r docsgpt/requirements-{extra}.txt "
|
||||
f"(or: uv sync --extra {extra}; Docker: --build-arg EXTRAS={extra})"
|
||||
)
|
||||
|
||||
|
||||
@@ -83,12 +83,7 @@ UNIQUE = f"qa_e2e_{int(time.time())}_{uuid.uuid4().hex[:6]}"
|
||||
|
||||
|
||||
# Quiet noisy library loggers so the script's own output stays readable.
|
||||
for name in (
|
||||
"application", "docsgpt.api", "docsgpt.storage",
|
||||
"docsgpt.usage", "docsgpt.parser",
|
||||
"docsgpt.api.user.reconciliation",
|
||||
):
|
||||
logging.getLogger(name).setLevel(logging.ERROR)
|
||||
logging.getLogger("docsgpt").setLevel(logging.ERROR)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -23,7 +23,7 @@ class TestExtras:
|
||||
|
||||
def test_install_hint_names_every_install_route(self):
|
||||
hint = optional_deps.install_hint("milvus")
|
||||
assert "requirements-milvus.txt" in hint
|
||||
assert "pip install -r docsgpt/requirements-milvus.txt" in hint
|
||||
assert "--extra milvus" in hint
|
||||
assert "EXTRAS=milvus" in hint
|
||||
|
||||
|
||||
@@ -250,6 +250,10 @@ class TestReclaimIsSkippedForEmbeds:
|
||||
def test_the_embed_task_is_skipped(self):
|
||||
assert not self._collects("docsgpt.vectorstore.embeddings_tasks.embed_texts")
|
||||
|
||||
def test_the_legacy_embed_name_is_skipped_too(self):
|
||||
"""Messages from the previous release carry the application.* name."""
|
||||
assert not self._collects("application.vectorstore.embeddings_tasks.embed_texts")
|
||||
|
||||
def test_parsing_still_reclaims(self):
|
||||
assert self._collects("docsgpt.api.user.tasks.parse_document")
|
||||
|
||||
|
||||
Reference in new issue
Block a user