mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 16:13:23 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
32 lines
921 B
Python
32 lines
921 B
Python
"""Image parser.
|
|
|
|
Contains parser for .png, .jpg, .jpeg files.
|
|
|
|
"""
|
|
from pathlib import Path
|
|
import requests
|
|
from typing import Dict, Union
|
|
|
|
from docsgpt.parser.file.base_parser import BaseParser
|
|
from docsgpt.core.settings import settings
|
|
|
|
|
|
class ImageParser(BaseParser):
|
|
"""Image parser."""
|
|
|
|
def _init_parser(self) -> Dict:
|
|
"""Init parser."""
|
|
return {}
|
|
|
|
def parse_file(self, file: Path, errors: str = "ignore") -> Union[str, list[str]]:
|
|
if settings.PARSE_IMAGE_REMOTE:
|
|
doc2md_service = "https://llm.arc53.com/doc2md"
|
|
# alternatively you can use local vision capable LLM
|
|
with open(file, "rb") as file_loaded:
|
|
files = {'file': file_loaded}
|
|
response = requests.post(doc2md_service, files=files, timeout=100)
|
|
data = response.json()["markdown"]
|
|
else:
|
|
data = ""
|
|
return data
|