Files
DocsGPT/tests/scripts/test_verify_offline.py
T
Alex 7da46c2bea feat: air-gapped deployment guide, no implicit downloads
- Ship tiktoken's cl100k_base inside the package and build the encoding
  from it, so token counting never downloads anything.
- Default EMBEDDINGS_CACHE_DIR to <data home>/models instead of FastEmbed's
  temp dir, and read tokenizer.json and repo metadata from that cache, so
  a model downloads once and survives reboots.
- TTS_PROVIDER=none and STT_PROVIDER=none switch the speech features off:
  the endpoints return 404, audio files fail to ingest with a clear
  message, /api/config reports tts_available/stt_available, and the UI
  hides the Speak and microphone buttons.
- Drop the Google Fonts Roboto import from the web UI.
- prefetch-models fills the cache the app reads; verify-offline checks the
  packaged encoding.
- Docs: new Air-Gapped Deployment guide, settings and cache notes.
2026-09-15 17:54:24 +01:00

55 lines
2.9 KiB
Python

"""Offline verification: every check must pass, docling only when installed."""
import sys
import types
from unittest.mock import patch
from docsgpt.scripts import verify_offline
def _fake_tiktoken(monkeypatch):
"""The check must exercise the app's own loader, which reads the packaged encoding."""
encoding = types.SimpleNamespace(name="cl100k_base", encode=lambda text: [1, 2])
monkeypatch.setattr("docsgpt.utils.get_encoding", lambda: encoding)
module = types.ModuleType("tiktoken")
module.get_encoding = lambda name: (_ for _ in ()).throw(AssertionError("downloaded via tiktoken"))
monkeypatch.setitem(sys.modules, "tiktoken", module)
class TestVerify:
def test_passes_when_every_check_passes(self, monkeypatch, capsys):
_fake_tiktoken(monkeypatch)
counter = types.SimpleNamespace(name="org/model", count=lambda text: 4)
with patch("docsgpt.parser.tokenization.get_token_counter", return_value=counter), \
patch("docsgpt.vectorstore.embeddings_local.EmbeddingsWrapper") as wrapper, \
patch.object(verify_offline, "is_available", return_value=False):
wrapper.return_value.embed_query.return_value = [0.0] * 768
assert verify_offline.verify(["ibm-granite/granite-embedding-311m-multilingual-r2"]) is True
out = capsys.readouterr().out
assert "ok tiktoken cl100k_base" in out
assert "skip docling" in out
def test_fails_when_the_tokenizer_fell_back_to_cl100k(self, monkeypatch, capsys):
"""A cache miss makes chunking silently use cl100k; that is a failed check."""
_fake_tiktoken(monkeypatch)
counter = types.SimpleNamespace(name="cl100k_base", count=lambda text: 4)
with patch("docsgpt.parser.tokenization.get_token_counter", return_value=counter), \
patch("docsgpt.vectorstore.embeddings_local.EmbeddingsWrapper") as wrapper, \
patch.object(verify_offline, "is_available", return_value=False):
wrapper.return_value.embed_query.return_value = [0.0] * 768
assert verify_offline.verify(["ibm-granite/granite-embedding-311m-multilingual-r2"]) is False
assert "FAIL tokenizer" in capsys.readouterr().out
def test_runs_the_docling_check_when_installed(self, monkeypatch):
_fake_tiktoken(monkeypatch)
with patch.object(verify_offline, "is_available", return_value=True), \
patch.object(verify_offline, "_docling_check", return_value="models from /app/models/docling") as check:
assert verify_offline.verify([]) is True
check.assert_called_once()
def test_remote_models_are_skipped(self, monkeypatch, capsys):
_fake_tiktoken(monkeypatch)
with patch.object(verify_offline, "is_available", return_value=False):
assert verify_offline.verify(["openai_text-embedding-ada-002"]) is True
assert "skip openai_text-embedding-ada-002" in capsys.readouterr().out