Merge pull request #1461 from BerriAI/litellm_set_num_workers_as_env

[FEAT] Proxy set NUM_WORKERS as env variable
This commit is contained in:
Ishaan Jaff
2024-01-16 08:51:02 -08:00
committed by GitHub
2 changed files with 63 additions and 22 deletions
+43 -18
View File
@@ -1,31 +1,46 @@
# CLI Arguments
Cli arguments, --host, --port, --num_workers
#### --host
## --host
- **Default:** `'0.0.0.0'`
- The host for the server to listen on.
- **Usage:**
```shell
litellm --host 127.0.0.1
```
- **Usage - set Environment Variable:** `HOST`
```shell
export HOST=127.0.0.1
litellm
```
#### --port
## --port
- **Default:** `8000`
- The port to bind the server to.
- **Usage:**
```shell
litellm --port 8080
```
- **Usage - set Environment Variable:** `PORT`
```shell
export PORT=8080
litellm
```
#### --num_workers
## --num_workers
- **Default:** `1`
- The number of uvicorn workers to spin up.
- **Usage:**
```shell
litellm --num_workers 4
```
- **Usage - set Environment Variable:** `NUM_WORKERS`
```shell
export NUM_WORKERS=4
litellm
```
#### --api_base
## --api_base
- **Default:** `None`
- The API base for the model litellm should call.
- **Usage:**
@@ -33,7 +48,7 @@ Cli arguments, --host, --port, --num_workers
litellm --model huggingface/tinyllama --api_base https://k58ory32yinf1ly0.us-east-1.aws.endpoints.huggingface.cloud
```
#### --api_version
## --api_version
- **Default:** `None`
- For Azure services, specify the API version.
- **Usage:**
@@ -41,7 +56,7 @@ Cli arguments, --host, --port, --num_workers
litellm --model azure/gpt-deployment --api_version 2023-08-01 --api_base https://<your api base>"
```
#### --model or -m
## --model or -m
- **Default:** `None`
- The model name to pass to Litellm.
- **Usage:**
@@ -49,7 +64,7 @@ Cli arguments, --host, --port, --num_workers
litellm --model gpt-3.5-turbo
```
#### --test
## --test
- **Type:** `bool` (Flag)
- Proxy chat completions URL to make a test request.
- **Usage:**
@@ -57,7 +72,7 @@ Cli arguments, --host, --port, --num_workers
litellm --test
```
#### --health
## --health
- **Type:** `bool` (Flag)
- Runs a health check on all models in config.yaml
- **Usage:**
@@ -65,7 +80,7 @@ Cli arguments, --host, --port, --num_workers
litellm --health
```
#### --alias
## --alias
- **Default:** `None`
- An alias for the model, for user-friendly reference.
- **Usage:**
@@ -73,7 +88,7 @@ Cli arguments, --host, --port, --num_workers
litellm --alias my-gpt-model
```
#### --debug
## --debug
- **Default:** `False`
- **Type:** `bool` (Flag)
- Enable debugging mode for the input.
@@ -81,15 +96,25 @@ Cli arguments, --host, --port, --num_workers
```shell
litellm --debug
```
- **Usage - set Environment Variable:** `DEBUG`
```shell
export DEBUG=True
litellm
```
#### --detailed_debug
## --detailed_debug
- **Default:** `False`
- **Type:** `bool` (Flag)
- Enable debugging mode for the input.
- **Usage:**
```shell
litellm --detailed_debug
``
```
- **Usage - set Environment Variable:** `DETAILED_DEBUG`
```shell
export DETAILED_DEBUG=True
litellm
```
#### --temperature
- **Default:** `None`
@@ -100,7 +125,7 @@ Cli arguments, --host, --port, --num_workers
litellm --temperature 0.7
```
#### --max_tokens
## --max_tokens
- **Default:** `None`
- **Type:** `int`
- Set the maximum number of tokens for the model output.
@@ -109,7 +134,7 @@ Cli arguments, --host, --port, --num_workers
litellm --max_tokens 50
```
#### --request_timeout
## --request_timeout
- **Default:** `600`
- **Type:** `int`
- Set the timeout in seconds for completion calls.
@@ -118,7 +143,7 @@ Cli arguments, --host, --port, --num_workers
litellm --request_timeout 300
```
#### --drop_params
## --drop_params
- **Type:** `bool` (Flag)
- Drop any unmapped params.
- **Usage:**
@@ -126,7 +151,7 @@ Cli arguments, --host, --port, --num_workers
litellm --drop_params
```
#### --add_function_to_prompt
## --add_function_to_prompt
- **Type:** `bool` (Flag)
- If a function passed but unsupported, pass it as a part of the prompt.
- **Usage:**
@@ -134,14 +159,14 @@ Cli arguments, --host, --port, --num_workers
litellm --add_function_to_prompt
```
#### --config
## --config
- Configure Litellm by providing a configuration file path.
- **Usage:**
```shell
litellm --config path/to/config.yaml
```
#### --telemetry
## --telemetry
- **Default:** `True`
- **Type:** `bool`
- Help track usage of this feature.
+20 -4
View File
@@ -39,9 +39,16 @@ def is_port_in_use(port):
@click.command()
@click.option("--host", default="0.0.0.0", help="Host for the server to listen on.")
@click.option("--port", default=8000, help="Port to bind the server to.")
@click.option("--num_workers", default=1, help="Number of gunicorn workers to spin up")
@click.option(
"--host", default="0.0.0.0", help="Host for the server to listen on.", envvar="HOST"
)
@click.option("--port", default=8000, help="Port to bind the server to.", envvar="PORT")
@click.option(
"--num_workers",
default=1,
help="Number of gunicorn workers to spin up",
envvar="NUM_WORKERS",
)
@click.option("--api_base", default=None, help="API base URL.")
@click.option(
"--api_version",
@@ -62,7 +69,12 @@ def is_port_in_use(port):
@click.option("--headers", default=None, help="headers for the API call")
@click.option("--save", is_flag=True, type=bool, help="Save the model-specific config")
@click.option(
"--debug", default=False, is_flag=True, type=bool, help="To debug the input"
"--debug",
default=False,
is_flag=True,
type=bool,
help="To debug the input",
envvar="DEBUG",
)
@click.option(
"--detailed_debug",
@@ -70,6 +82,7 @@ def is_port_in_use(port):
is_flag=True,
type=bool,
help="To view detailed debug logs",
envvar="DETAILED_DEBUG",
)
@click.option(
"--use_queue",
@@ -472,6 +485,9 @@ def run_server(
# gunicorn app function
return self.application
print(
f"\033[1;32mLiteLLM Proxy: Starting server on {host}:{port} with {num_workers} workers\033[0m\n"
)
gunicorn_options = {
"bind": f"{host}:{port}",
"workers": num_workers, # default is 1