Let FastEmbed download a model when completing its cache fails

Completing a partial snapshot is best effort; a network error or rate limit
there is logged and FastEmbed still tries its own sources.
This commit is contained in:
arc53-machine committed 2026-09-29 12:36:57 +01:00
1 parent f6f9ae670c
commit 159ac03904
2 files changed
+17 -2

No files matched your search

+6 -2
View File
@@ -263,7 +263,8 @@ def _complete_model_cache(repo: str, cache_dir: Optional[str]) -> None:
holds. The chunker caches only ``tokenizer.json`` there (see
``docsgpt/parser/tokenization.py``), so FastEmbed then finds no graph and
fails to load. Fetching the files it needs first makes either order work.
Offline, nothing is fetched and FastEmbed reports what is missing.
Offline, nothing is fetched and FastEmbed reports what is missing; a
failed download is logged and left to FastEmbed's own sources too.
Args:
repo: The model's FastEmbed name.
@@ -293,7 +294,10 @@ def _complete_model_cache(repo: str, cache_dir: Optional[str]) -> None:
if os.environ.get("HF_HUB_OFFLINE", "").strip().upper() in {"1", "TRUE", "YES", "ON"}:
return
logger.warning("The cached %s has no %s; downloading the model files.", source, ", ".join(needed))
snapshot_download(repo_id=source, allow_patterns=[*_FASTEMBED_SUPPORT_FILES, *needed], cache_dir=cache)
try:
snapshot_download(repo_id=source, allow_patterns=[*_FASTEMBED_SUPPORT_FILES, *needed], cache_dir=cache)
except Exception as exc: # noqa: BLE001 -- best effort: FastEmbed still tries its own sources
logger.warning("Could not complete the cached %s (%s); leaving the download to FastEmbed.", source, exc)
def _pad_to_longest_in_batch(model: Any) -> None:
@@ -460,6 +460,17 @@ class TestIncompleteModelCache:
assert "onnx/model.onnx" in downloads[0]["allow_patterns"]
assert "tokenizer_config.json" in downloads[0]["allow_patterns"]
def test_a_failed_repair_leaves_loading_to_fastembed(self, monkeypatch, tmp_path):
"""The repair is best effort: a network error or rate limit here must
not stop FastEmbed from trying its own download."""
from docsgpt.vectorstore import embeddings_local
monkeypatch.delenv("HF_HUB_OFFLINE", raising=False)
with patch("fastembed.TextEmbedding._list_supported_models", return_value=[self._description()]), \
patch("huggingface_hub.hf_hub_download", side_effect=FileNotFoundError("missing")), \
patch("huggingface_hub.snapshot_download", side_effect=OSError("rate limited")):
embeddings_local._complete_model_cache("sentence-transformers/all-mpnet-base-v2", str(tmp_path))
def test_a_complete_snapshot_downloads_nothing(self, monkeypatch, tmp_path):
assert self._complete(monkeypatch, tmp_path, cached={"tokenizer.json", "onnx/model.onnx"}) == []