fix(rename): second review pass on the package rename

- The post-task reclaim skip recognises the legacy application.* embed name,
  so query embeds queued by the previous release do not pay a full collect.
- The Azure compose file mounts host data on /app/{indexes,inputs,vectors},
  where the process actually reads and writes; it mounted /app/application/...
  before the rename and /app/docsgpt/... after it, and nothing wrote to either.
- The offline image check triggers on docsgpt/requirements*.txt again;
  dependabot's pip entry points at docsgpt/.
- install_hint() and its docstring name docsgpt/requirements-<extra>.txt;
  the test asserts the full path.
- application/vectors/ stays ignored: the compose files still mount it.
- The durability QA script quiets the docsgpt logger tree.
- Upgrade note: the three renamed source-sync entries start their timers
  from the upgrade.
This commit is contained in:
Alex committed 2026-09-07 14:02:59 +01:00
1 parent adb6963523
commit 2d9568ff79
10 files changed
+27 -20

No files matched your search

+1 -1
View File
@@ -6,7 +6,7 @@
version: 2
updates:
- package-ecosystem: "pip" # See documentation for possible values
directory: "/application" # Location of package manifests
directory: "/docsgpt" # Location of package manifests
schedule:
interval: "daily"
- package-ecosystem: "npm" # See documentation for possible values
+1 -1
View File
@@ -11,7 +11,7 @@ on:
- 'docsgpt/Dockerfile'
- '.dockerignore'
- 'application/**'
- 'application/requirements*.txt'
- 'docsgpt/requirements*.txt'
- 'docsgpt/scripts/prefetch_models.py'
- 'docsgpt/scripts/verify_offline.py'
- 'docsgpt/vectorstore/model_registry.py'
+1
View File
@@ -180,6 +180,7 @@ frontend/*.sln
frontend/*.sw?
docsgpt/vectors/
application/vectors/
**/inputs
+8 -6
View File
@@ -44,12 +44,14 @@ services:
ports:
- "7091:7091"
volumes:
# Host data stays under application/ (the old package directory) for this
# release so an upgraded checkout keeps its indexes; it moves with the
# packaging work, together with an upgrade note.
- ../application/indexes:/app/docsgpt/indexes
- ../application/inputs:/app/docsgpt/inputs
- ../application/vectors:/app/docsgpt/vectors
# Host data lives under application/ (the old package directory) for this
# release, like the other compose files; it moves with the packaging work,
# together with an upgrade note. The container side is /app: the process
# runs there and resolves indexes/inputs/vectors relative to it (earlier
# versions of this file mounted /app/application/..., which nothing wrote to).
- ../application/indexes:/app/indexes
- ../application/inputs:/app/inputs
- ../application/vectors:/app/vectors
depends_on:
redis:
condition: service_started
+3 -1
View File
@@ -106,7 +106,9 @@ alias, so nothing breaks on upgrade, but update these before the alias goes:
- Celery task names changed with the package (`docsgpt.api.user.tasks.ingest`
and so on). A worker on this release also accepts the old names, so tasks
queued before the upgrade still run, and beat rewrites the periodic
schedule in Redis on start-up. Nothing to do.
schedule in Redis on start-up. The daily, weekly and monthly source-sync
timers restart from the upgrade, so the first sync after it can land later
than it would have (a monthly sync by up to a month). Nothing to do.
- Data directories do not move: the compose files keep your indexes, inputs
and vectors under `application/` in the checkout, where they already are.
+5 -2
View File
@@ -121,6 +121,7 @@ def _trim_native_heap() -> None:
# measured at ~86 ms for the collect against ~8 ms for the embed itself on a
# worker holding the ONNX model, i.e. a 9x slowdown of the whole round trip.
_NO_RECLAIM_TASKS = frozenset({"docsgpt.vectorstore.embeddings_tasks.embed_texts"})
LEGACY_TASK_PREFIX = "application."
@task_postrun.connect
@@ -132,7 +133,10 @@ def _reclaim_memory_after_task(task=None, **kwargs):
generational collect after a task that allocated a few kilobytes just
charges the next task for walking the whole heap.
"""
if getattr(task, "name", None) in _NO_RECLAIM_TASKS:
name = getattr(task, "name", None)
if isinstance(name, str) and name.startswith(LEGACY_TASK_PREFIX):
name = "docsgpt." + name[len(LEGACY_TASK_PREFIX):]
if name in _NO_RECLAIM_TASKS:
return
gc.collect()
torch = sys.modules.get("torch")
@@ -169,7 +173,6 @@ celery = make_celery()
celery.config_from_object("docsgpt.celeryconfig")
#: Task-name prefix the package carried before the rename to ``docsgpt``.
LEGACY_TASK_PREFIX = "application."
def register_legacy_task_names(app: Celery) -> int:
+2 -2
View File
@@ -1,7 +1,7 @@
"""Optional dependency extras and the one place their install hints come from.
Heavy or niche packages are not installed by default. Each extra maps to a
pyproject extra and to an exported ``application/requirements-<extra>.txt``,
pyproject extra and to an exported ``docsgpt/requirements-<extra>.txt``,
so a missing module can always be explained with the exact command to run.
"""
@@ -28,7 +28,7 @@ _MODULE_TO_EXTRA: Dict[str, str] = {
def install_hint(extra: str) -> str:
"""Install command for ``extra``, for error messages and logs."""
return (
f"pip install -r application/requirements-{extra}.txt "
f"pip install -r docsgpt/requirements-{extra}.txt "
f"(or: uv sync --extra {extra}; Docker: --build-arg EXTRAS={extra})"
)
+1 -6
View File
@@ -83,12 +83,7 @@ UNIQUE = f"qa_e2e_{int(time.time())}_{uuid.uuid4().hex[:6]}"
# Quiet noisy library loggers so the script's own output stays readable.
for name in (
"application", "docsgpt.api", "docsgpt.storage",
"docsgpt.usage", "docsgpt.parser",
"docsgpt.api.user.reconciliation",
):
logging.getLogger(name).setLevel(logging.ERROR)
logging.getLogger("docsgpt").setLevel(logging.ERROR)
# ---------------------------------------------------------------------------
+1 -1
View File
@@ -23,7 +23,7 @@ class TestExtras:
def test_install_hint_names_every_install_route(self):
hint = optional_deps.install_hint("milvus")
assert "requirements-milvus.txt" in hint
assert "pip install -r docsgpt/requirements-milvus.txt" in hint
assert "--extra milvus" in hint
assert "EXTRAS=milvus" in hint
+4
View File
@@ -250,6 +250,10 @@ class TestReclaimIsSkippedForEmbeds:
def test_the_embed_task_is_skipped(self):
assert not self._collects("docsgpt.vectorstore.embeddings_tasks.embed_texts")
def test_the_legacy_embed_name_is_skipped_too(self):
"""Messages from the previous release carry the application.* name."""
assert not self._collects("application.vectorstore.embeddings_tasks.embed_texts")
def test_parsing_still_reclaims(self):
assert self._collects("docsgpt.api.user.tasks.parse_document")