Files
Alex 7da46c2bea feat: air-gapped deployment guide, no implicit downloads
- Ship tiktoken's cl100k_base inside the package and build the encoding
  from it, so token counting never downloads anything.
- Default EMBEDDINGS_CACHE_DIR to <data home>/models instead of FastEmbed's
  temp dir, and read tokenizer.json and repo metadata from that cache, so
  a model downloads once and survives reboots.
- TTS_PROVIDER=none and STT_PROVIDER=none switch the speech features off:
  the endpoints return 404, audio files fail to ingest with a clear
  message, /api/config reports tts_available/stt_available, and the UI
  hides the Speak and microphone buttons.
- Drop the Google Fonts Roboto import from the web UI.
- prefetch-models fills the cache the app reads; verify-offline checks the
  packaged encoding.
- Docs: new Air-Gapped Deployment guide, settings and cache notes.
2026-09-15 17:54:24 +01:00

53 lines
1.9 KiB
Python

from pathlib import Path
from typing import Dict, Union
from docsgpt.core.settings import settings
from docsgpt.parser.file.base_parser import BaseParser, DocumentParseError
from docsgpt.stt.stt_creator import STTCreator
from docsgpt.stt.upload_limits import enforce_audio_file_size_limit
class AudioParser(BaseParser):
def __init__(self, parser_config=None):
super().__init__(parser_config=parser_config)
self._transcript_metadata: Dict[str, Dict] = {}
def _init_parser(self) -> Dict:
return {}
def parse_file(self, file: Path, errors: str = "ignore") -> Union[str, list[str]]:
_ = errors
if not STTCreator.is_enabled(settings.STT_PROVIDER):
raise DocumentParseError(
f"{file.name}: audio files need speech-to-text, which is disabled (STT_PROVIDER=none)."
)
try:
enforce_audio_file_size_limit(file.stat().st_size)
except OSError:
pass
stt = STTCreator.create_stt(settings.STT_PROVIDER)
result = stt.transcribe(
file,
language=settings.STT_LANGUAGE,
timestamps=settings.STT_ENABLE_TIMESTAMPS,
diarize=settings.STT_ENABLE_DIARIZATION,
)
transcript_metadata = {
"transcript_duration_s": result.get("duration_s"),
"transcript_language": result.get("language"),
"transcript_provider": result.get("provider"),
}
if result.get("segments"):
transcript_metadata["transcript_segments"] = result["segments"]
self._transcript_metadata[str(file)] = {
key: value
for key, value in transcript_metadata.items()
if value not in (None, [], {})
}
return result.get("text", "")
def get_file_metadata(self, file: Path) -> Dict:
return self._transcript_metadata.get(str(file), {})