diff --git a/docs/my-website/docs/load_test.md b/docs/my-website/docs/load_test.md
index f568b56961..94165fb7b2 100644
--- a/docs/my-website/docs/load_test.md
+++ b/docs/my-website/docs/load_test.md
@@ -1,5 +1,84 @@
+import Image from '@theme/IdealImage';
+
# 🔥 Load Test LiteLLM
+## Load Test LiteLLM Proxy - 1500+ req/s
+
+## 1500+ concurrent requests/s
+
+LiteLLM proxy has been load tested to handle 1500+ concurrent req/s
+
+```python
+import time, asyncio
+from openai import AsyncOpenAI, AsyncAzureOpenAI
+import uuid
+import traceback
+
+# base_url - litellm proxy endpoint
+# api_key - litellm proxy api-key, is created proxy with auth
+litellm_client = AsyncOpenAI(base_url="http://0.0.0.0:4000", api_key="sk-1234")
+
+
+async def litellm_completion():
+ # Your existing code for litellm_completion goes here
+ try:
+ response = await litellm_client.chat.completions.create(
+ model="azure-gpt-3.5",
+ messages=[{"role": "user", "content": f"This is a test: {uuid.uuid4()}"}],
+ )
+ print(response)
+ return response
+
+ except Exception as e:
+ # If there's an exception, log the error message
+ with open("error_log.txt", "a") as error_log:
+ error_log.write(f"Error during completion: {str(e)}\n")
+ pass
+
+
+async def main():
+ for i in range(1):
+ start = time.time()
+ n = 1500 # Number of concurrent tasks
+ tasks = [litellm_completion() for _ in range(n)]
+
+ chat_completions = await asyncio.gather(*tasks)
+
+ successful_completions = [c for c in chat_completions if c is not None]
+
+ # Write errors to error_log.txt
+ with open("error_log.txt", "a") as error_log:
+ for completion in chat_completions:
+ if isinstance(completion, str):
+ error_log.write(completion + "\n")
+
+ print(n, time.time() - start, len(successful_completions))
+ time.sleep(10)
+
+
+if __name__ == "__main__":
+ # Blank out contents of error_log.txt
+ open("error_log.txt", "w").close()
+
+ asyncio.run(main())
+
+```
+
+### Throughput - 30% Increase
+LiteLLM proxy + Load Balancer gives **30% increase** in throughput compared to Raw OpenAI API
+
+
+### Latency Added - 0.00325 seconds
+LiteLLM proxy adds **0.00325 seconds** latency as compared to using the Raw OpenAI API
+
+
+
+### Testing LiteLLM Proxy with Locust
+- 1 LiteLLM container can handle ~140 requests/second with 0.4 failures
+
+
+
+## Load Test LiteLLM SDK vs OpenAI
Here is a script to load test LiteLLM vs OpenAI
```python
@@ -84,4 +163,5 @@ async def loadtest_fn():
# Run the event loop to execute the async function
asyncio.run(loadtest_fn())
-```
\ No newline at end of file
+```
+
diff --git a/docs/my-website/docs/proxy/deploy.md b/docs/my-website/docs/proxy/deploy.md
index 6de8625d03..4b51f094cb 100644
--- a/docs/my-website/docs/proxy/deploy.md
+++ b/docs/my-website/docs/proxy/deploy.md
@@ -350,17 +350,3 @@ Run the command `docker-compose up` or `docker compose up` as per your docker in
Your LiteLLM container should be running now on the defined port e.g. `8000`.
-
-
-
-## LiteLLM Proxy Performance
-
-LiteLLM proxy has been load tested to handle 1500 req/s.
-
-### Throughput - 30% Increase
-LiteLLM proxy + Load Balancer gives **30% increase** in throughput compared to Raw OpenAI API
-
-
-### Latency Added - 0.00325 seconds
-LiteLLM proxy adds **0.00325 seconds** latency as compared to using the Raw OpenAI API
-
diff --git a/docs/my-website/img/locust.png b/docs/my-website/img/locust.png
new file mode 100644
index 0000000000..1bcedf1d04
Binary files /dev/null and b/docs/my-website/img/locust.png differ
diff --git a/litellm/proxy/proxy_cli.py b/litellm/proxy/proxy_cli.py
index f7eba02ecb..367bbbb700 100644
--- a/litellm/proxy/proxy_cli.py
+++ b/litellm/proxy/proxy_cli.py
@@ -16,6 +16,13 @@ from importlib import resources
import shutil
telemetry = None
+default_num_workers = 1
+try:
+ default_num_workers = os.cpu_count() or 1
+ if default_num_workers is not None and default_num_workers > 0:
+ default_num_workers -= 1
+except:
+ pass
def append_query_params(url, params):
@@ -57,7 +64,7 @@ def is_port_in_use(port):
@click.option("--port", default=8000, help="Port to bind the server to.", envvar="PORT")
@click.option(
"--num_workers",
- default=1,
+ default=default_num_workers,
help="Number of gunicorn workers to spin up",
envvar="NUM_WORKERS",
)
diff --git a/litellm/proxy/proxy_load_test/litellm_proxy_config.yaml b/litellm/proxy/proxy_load_test/litellm_proxy_config.yaml
new file mode 100644
index 0000000000..2e107d3668
--- /dev/null
+++ b/litellm/proxy/proxy_load_test/litellm_proxy_config.yaml
@@ -0,0 +1,6 @@
+model_list:
+ - model_name: gpt-3.5-turbo
+ litellm_params:
+ model: openai/my-fake-model
+ api_key: my-fake-key
+ api_base: http://0.0.0.0:8090
\ No newline at end of file
diff --git a/litellm/proxy/proxy_load_test/locustfile.py b/litellm/proxy/proxy_load_test/locustfile.py
new file mode 100644
index 0000000000..2cd2e2fcce
--- /dev/null
+++ b/litellm/proxy/proxy_load_test/locustfile.py
@@ -0,0 +1,27 @@
+from locust import HttpUser, task, between
+
+
+class MyUser(HttpUser):
+ wait_time = between(1, 5)
+
+ @task
+ def chat_completion(self):
+ headers = {
+ "Content-Type": "application/json",
+ # Include any additional headers you may need for authentication, etc.
+ }
+
+ # Customize the payload with "model" and "messages" keys
+ payload = {
+ "model": "gpt-3.5-turbo",
+ "messages": [
+ {"role": "system", "content": "You are a chat bot."},
+ {"role": "user", "content": "Hello, how are you?"},
+ ],
+ # Add more data as necessary
+ }
+
+ # Make a POST request to the "chat/completions" endpoint
+ response = self.client.post("chat/completions", json=payload, headers=headers)
+
+ # Print or log the response if needed
diff --git a/litellm/proxy/proxy_load_test/openai_endpoint.py b/litellm/proxy/proxy_load_test/openai_endpoint.py
new file mode 100644
index 0000000000..b3291ce709
--- /dev/null
+++ b/litellm/proxy/proxy_load_test/openai_endpoint.py
@@ -0,0 +1,50 @@
+# import sys, os
+# sys.path.insert(
+# 0, os.path.abspath("../")
+# ) # Adds the parent directory to the system path
+from fastapi import FastAPI, Request, status, HTTPException, Depends
+from fastapi.responses import StreamingResponse
+from fastapi.security import OAuth2PasswordBearer
+from fastapi.middleware.cors import CORSMiddleware
+
+app = FastAPI()
+
+app.add_middleware(
+ CORSMiddleware,
+ allow_origins=["*"],
+ allow_credentials=True,
+ allow_methods=["*"],
+ allow_headers=["*"],
+)
+
+
+# for completion
+@app.post("/chat/completions")
+@app.post("/v1/chat/completions")
+async def completion(request: Request):
+ return {
+ "id": "chatcmpl-123",
+ "object": "chat.completion",
+ "created": 1677652288,
+ "model": "gpt-3.5-turbo-0125",
+ "system_fingerprint": "fp_44709d6fcb",
+ "choices": [
+ {
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "\n\nHello there, how may I assist you today?",
+ },
+ "logprobs": None,
+ "finish_reason": "stop",
+ }
+ ],
+ "usage": {"prompt_tokens": 9, "completion_tokens": 12, "total_tokens": 21},
+ }
+
+
+if __name__ == "__main__":
+ import uvicorn
+
+ # run this on 8090, 8091, 8092 and 8093
+ uvicorn.run(app, host="0.0.0.0", port=8090)