mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-03 11:11:58 +00:00
- Ship tiktoken's cl100k_base inside the package and build the encoding from it, so token counting never downloads anything. - Default EMBEDDINGS_CACHE_DIR to <data home>/models instead of FastEmbed's temp dir, and read tokenizer.json and repo metadata from that cache, so a model downloads once and survives reboots. - TTS_PROVIDER=none and STT_PROVIDER=none switch the speech features off: the endpoints return 404, audio files fail to ingest with a clear message, /api/config reports tts_available/stt_available, and the UI hides the Speak and microphone buttons. - Drop the Google Fonts Roboto import from the web UI. - prefetch-models fills the cache the app reads; verify-offline checks the packaged encoding. - Docs: new Air-Gapped Deployment guide, settings and cache notes.
53 lines
1.9 KiB
Python
53 lines
1.9 KiB
Python
from pathlib import Path
|
|
from typing import Dict, Union
|
|
|
|
from docsgpt.core.settings import settings
|
|
from docsgpt.parser.file.base_parser import BaseParser, DocumentParseError
|
|
from docsgpt.stt.stt_creator import STTCreator
|
|
from docsgpt.stt.upload_limits import enforce_audio_file_size_limit
|
|
|
|
|
|
class AudioParser(BaseParser):
|
|
def __init__(self, parser_config=None):
|
|
super().__init__(parser_config=parser_config)
|
|
self._transcript_metadata: Dict[str, Dict] = {}
|
|
|
|
def _init_parser(self) -> Dict:
|
|
return {}
|
|
|
|
def parse_file(self, file: Path, errors: str = "ignore") -> Union[str, list[str]]:
|
|
_ = errors
|
|
if not STTCreator.is_enabled(settings.STT_PROVIDER):
|
|
raise DocumentParseError(
|
|
f"{file.name}: audio files need speech-to-text, which is disabled (STT_PROVIDER=none)."
|
|
)
|
|
try:
|
|
enforce_audio_file_size_limit(file.stat().st_size)
|
|
except OSError:
|
|
pass
|
|
stt = STTCreator.create_stt(settings.STT_PROVIDER)
|
|
result = stt.transcribe(
|
|
file,
|
|
language=settings.STT_LANGUAGE,
|
|
timestamps=settings.STT_ENABLE_TIMESTAMPS,
|
|
diarize=settings.STT_ENABLE_DIARIZATION,
|
|
)
|
|
|
|
transcript_metadata = {
|
|
"transcript_duration_s": result.get("duration_s"),
|
|
"transcript_language": result.get("language"),
|
|
"transcript_provider": result.get("provider"),
|
|
}
|
|
if result.get("segments"):
|
|
transcript_metadata["transcript_segments"] = result["segments"]
|
|
|
|
self._transcript_metadata[str(file)] = {
|
|
key: value
|
|
for key, value in transcript_metadata.items()
|
|
if value not in (None, [], {})
|
|
}
|
|
return result.get("text", "")
|
|
|
|
def get_file_metadata(self, file: Path) -> Dict:
|
|
return self._transcript_metadata.get(str(file), {})
|