diff --git a/docs/my-website/docs/proxy/deploy.md b/docs/my-website/docs/proxy/deploy.md index 10924888be..e78c128bbf 100644 --- a/docs/my-website/docs/proxy/deploy.md +++ b/docs/my-website/docs/proxy/deploy.md @@ -235,15 +235,6 @@ Your OpenAI proxy server is now running on `http://127.0.0.1:4000`. | [LiteLLM container + Redis](#litellm-container--redis) | + load balance across multiple litellm containers | | [LiteLLM Database container + PostgresDB + Redis](#litellm-database-container--postgresdb--redis) | + use Virtual Keys + Track Spend + load balance across multiple litellm containers | - - -## Machine Specifications to Deploy LiteLLM - -| Service | Spec | CPUs | Memory | Performance | Architecture | Version| -| --- | --- | --- | --- | --- | --- | --- | -| Server | `t2.small`. | `1vCPUs` | `8GB` | avg latency=`57ms`, median latency=`50ms`, Requests per second=`33` | | | -| Redis Cache | - | - | - | - | | 7.0+ Redis Engine| - ## Deploy with Database ### Docker, Kubernetes, Helm Chart @@ -485,11 +476,6 @@ docker run --name litellm-proxy \ ghcr.io/berriai/litellm-database:main-latest --config your_config.yaml ``` -## Best Practices for Deploying to Production -### 1. Switch of debug logs in production -don't use [`--detailed-debug`, `--debug`](https://docs.litellm.ai/docs/proxy/debugging#detailed-debug) or `litellm.set_verbose=True`. We found using debug logs can add 5-10% latency per LLM API call - - ## Advanced Deployment Settings ### Customization of the server root path diff --git a/docs/my-website/docs/proxy/prod.md b/docs/my-website/docs/proxy/prod.md new file mode 100644 index 0000000000..568d37a7d1 --- /dev/null +++ b/docs/my-website/docs/proxy/prod.md @@ -0,0 +1,131 @@ +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# ⚡ Best Practices for Production + +Expected Performance in Production + +1 LiteLLM Uvicorn Worker on Kubernetes + +| Description | Value | +|--------------|-------| +| Avg latency | `50ms` | +| Median latency | `51ms` | +| `/chat/completions` Requests/second | `35` | +| `/chat/completions` Requests/minute | `2100` | +| `/chat/completions` Requests/hour | `126K` | + + +## 1. Switch of Debug Logging + +Remove `set_verbose: True` from your config.yaml +```yaml +litellm_settings: + set_verbose: True +``` + +## 2. On Kubernetes - Use 1 Uvicorn worker [Suggested CMD] + +Use this Docker `CMD`. This will start the proxy with 1 Uvicorn Async Worker + +(Ensure that you're not setting `run_gunicorn` or `num_workers` in the CMD). +```shell +CMD ["--port", "4000", "--config", "./proxy_server_config.yaml"] +``` + +## 3. Switch off spend logging and resetting budgets + +Add this to your config.yaml. (Only spend per Key, User and Team will be tracked - spend per API Call will not be written to the LiteLLM Database) +```yaml +general_settings: + disable_spend_logs: true + disable_reset_budget: true +``` + +## Machine Specifications to Deploy LiteLLM + +| Service | Spec | CPUs | Memory | Architecture | Version| +| --- | --- | --- | --- | --- | --- | +| Server | `t2.small`. | `1vCPUs` | `8GB` | `x86` | +| Redis Cache | - | - | - | - | 7.0+ Redis Engine| + + +## Reference Kubernetes Deployment YAML + +Reference Kubernetes `deployment.yaml` that was load tested by us + +```yaml +apiVersion: apps/v1 +kind: Deployment +metadata: + name: litellm-deployment +spec: + replicas: 3 + selector: + matchLabels: + app: litellm + template: + metadata: + labels: + app: litellm + spec: + containers: + - name: litellm-container + image: ghcr.io/berriai/litellm:main-latest + env: + - name: AZURE_API_KEY + value: "d6******" + - name: AZURE_API_BASE + value: "https://ope******" + - name: LITELLM_MASTER_KEY + value: "sk-1234" + - name: DATABASE_URL + value: "po**********" + args: + - "--config" + - "/app/proxy_config.yaml" # Update the path to mount the config file + volumeMounts: # Define volume mount for proxy_config.yaml + - name: config-volume + mountPath: /app + readOnly: true + livenessProbe: + httpGet: + path: /health/liveliness + port: 4000 + initialDelaySeconds: 120 + periodSeconds: 15 + successThreshold: 1 + failureThreshold: 3 + timeoutSeconds: 10 + readinessProbe: + httpGet: + path: /health/readiness + port: 4000 + initialDelaySeconds: 120 + periodSeconds: 15 + successThreshold: 1 + failureThreshold: 3 + timeoutSeconds: 10 + volumes: # Define volume to mount proxy_config.yaml + - name: config-volume + configMap: + name: litellm-config + +``` + + +Reference Kubernetes `service.yaml` that was load tested by us +```yaml +apiVersion: v1 +kind: Service +metadata: + name: litellm-service +spec: + selector: + app: litellm + ports: + - protocol: TCP + port: 4000 + targetPort: 4000 + type: LoadBalancer +``` diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index fbc20224e3..6d871b4903 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -30,6 +30,7 @@ const sidebars = { items: [ "proxy/quick_start", "proxy/deploy", + "proxy/prod", "proxy/configs", { type: 'link',