mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 14:12:58 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
90 lines
3.0 KiB
Python
90 lines
3.0 KiB
Python
"""OpenAI (and Azure OpenAI) embeddings built on the official ``openai`` SDK."""
|
|
|
|
from typing import List, Optional
|
|
|
|
from docsgpt.core.settings import settings
|
|
|
|
# openai >= 2.53 rejects a falsy api_key at construction; Azure authenticates
|
|
# through its own deployment credentials, so a placeholder keeps the client
|
|
# constructible when no key is configured.
|
|
NO_API_KEY = "sk-no-key"
|
|
|
|
DEFAULT_MODEL = "text-embedding-ada-002"
|
|
|
|
|
|
class OpenAIEmbeddings:
|
|
"""Embeddings client for OpenAI and Azure OpenAI.
|
|
|
|
Mirrors the ``embed_query``/``embed_documents`` interface the vector
|
|
stores expect, matching :class:`RemoteEmbeddings` and
|
|
:class:`EmbeddingsWrapper`.
|
|
"""
|
|
|
|
def __init__(
|
|
self,
|
|
openai_api_key: Optional[str] = None,
|
|
model: Optional[str] = None,
|
|
**kwargs,
|
|
) -> None:
|
|
"""Build the client, routing to Azure when the Azure settings are set.
|
|
|
|
Args:
|
|
openai_api_key: API key; falls back to ``EMBEDDINGS_KEY`` then
|
|
``OPENAI_API_KEY``.
|
|
model: Model name, or the Azure deployment name when running
|
|
against Azure.
|
|
"""
|
|
api_key = (
|
|
openai_api_key
|
|
or settings.EMBEDDINGS_KEY
|
|
or settings.OPENAI_API_KEY
|
|
or NO_API_KEY
|
|
)
|
|
self.model = model or DEFAULT_MODEL
|
|
self.dimension = None
|
|
|
|
is_azure = bool(
|
|
settings.OPENAI_API_BASE
|
|
and settings.OPENAI_API_VERSION
|
|
and settings.AZURE_DEPLOYMENT_NAME
|
|
)
|
|
if is_azure:
|
|
from openai import AzureOpenAI
|
|
|
|
self.client = AzureOpenAI(
|
|
api_key=api_key,
|
|
azure_endpoint=settings.OPENAI_API_BASE,
|
|
api_version=settings.OPENAI_API_VERSION,
|
|
)
|
|
else:
|
|
from openai import OpenAI
|
|
|
|
base_url = settings.OPENAI_BASE_URL or None
|
|
self.client = OpenAI(api_key=api_key, base_url=base_url)
|
|
|
|
def _embed(self, inputs: List[str]) -> List[List[float]]:
|
|
"""Embed a batch, returning vectors in request order."""
|
|
response = self.client.embeddings.create(model=self.model, input=inputs)
|
|
ordered = sorted(response.data, key=lambda item: item.index)
|
|
vectors = [item.embedding for item in ordered]
|
|
if vectors and self.dimension is None:
|
|
self.dimension = len(vectors[0])
|
|
return vectors
|
|
|
|
def embed_query(self, query: str) -> List[float]:
|
|
"""Embed a single query string."""
|
|
return self._embed([query])[0]
|
|
|
|
def embed_documents(self, documents: List[str]) -> List[List[float]]:
|
|
"""Embed a list of documents."""
|
|
if not documents:
|
|
return []
|
|
return self._embed(list(documents))
|
|
|
|
def __call__(self, text):
|
|
if isinstance(text, str):
|
|
return self.embed_query(text)
|
|
elif isinstance(text, list):
|
|
return self.embed_documents(text)
|
|
raise ValueError("Input must be a string or a list of strings")
|