diff --git a/docs/my-website/docs/tutorials/prompt_caching.md b/docs/my-website/docs/tutorials/prompt_caching.md new file mode 100644 index 0000000000..bf3d5a8dda --- /dev/null +++ b/docs/my-website/docs/tutorials/prompt_caching.md @@ -0,0 +1,128 @@ +import Image from '@theme/IdealImage'; +import Tabs from '@theme/Tabs'; +import TabItem from '@theme/TabItem'; + +# Auto-Inject Prompt Caching Checkpoints + +Reduce costs by up to 90% by using LiteLLM to auto-inject prompt caching checkpoints. + + + + +## How it works + +LiteLLM can automatically inject prompt caching checkpoints into your requests to LLM providers. This allows: + +- **Cost Reduction**: Long, static parts of your prompts can be cached to avoid repeated processing +- **No need to modify your application code**: You can configure the auto-caching behavior in the LiteLLM UI or in the `litellm config.yaml` file. + +## Configuration + +You need to specify `cache_control_injection_points` in your model configuration. This tells LiteLLM: +1. Where to add the caching directive (`location`) +2. Which message to target (`role`) + +LiteLLM will then automatically add a `cache_control` directive to the specified messages in your requests: + +```json +"cache_control": { + "type": "ephemeral" +} +``` + +## Usage Example + +In this example, we'll configure caching for system messages by adding the directive to all messages with `role: system`. + + + + +```yaml showLineNumbers title="litellm config.yaml" +model_list: + - model_name: anthropic-auto-inject-cache-system-message + litellm_params: + model: anthropic/claude-3-5-sonnet-20240620 + api_key: os.environ/ANTHROPIC_API_KEY + cache_control_injection_points: + - location: message + role: system +``` + + + + +On the LiteLLM UI, you can specify the `cache_control_injection_points` in the `Advanced Settings` tab when adding a model. + + + + + + +## Detailed Example + +### 1. Original Request to LiteLLM + +In this example, we have a very long, static system message and a varying user message. It's efficient to cache the system message since it rarely changes. + +```json +{ + "messages": [ + { + "role": "system", + "content": [ + { + "type": "text", + "text": "You are a helpful assistant. This is a set of very long instructions that you will follow. Here is a legal document that you will use to answer the user's question." + } + ] + }, + { + "role": "user", + "content": [ + { + "type": "text", + "text": "What is the main topic of this legal document?" + } + ] + } + ] +} +``` + +### 2. LiteLLM's Modified Request + +LiteLLM auto-injects the caching directive into the system message based on our configuration: + +```json +{ + "messages": [ + { + "role": "system", + "content": [ + { + "type": "text", + "text": "You are a helpful assistant. This is a set of very long instructions that you will follow. Here is a legal document that you will use to answer the user's question.", + "cache_control": {"type": "ephemeral"} + } + ] + }, + { + "role": "user", + "content": [ + { + "type": "text", + "text": "What is the main topic of this legal document?" + } + ] + } + ] +} +``` + +When the model provider processes this request, it will recognize the caching directive and only process the system message once, caching it for subsequent requests. + + + + + + diff --git a/docs/my-website/img/auto_prompt_caching.png b/docs/my-website/img/auto_prompt_caching.png new file mode 100644 index 0000000000..6cd3785512 Binary files /dev/null and b/docs/my-website/img/auto_prompt_caching.png differ diff --git a/docs/my-website/img/ui_auto_prompt_caching.png b/docs/my-website/img/ui_auto_prompt_caching.png new file mode 100644 index 0000000000..e6f48e48d0 Binary files /dev/null and b/docs/my-website/img/ui_auto_prompt_caching.png differ diff --git a/docs/my-website/sidebars.js b/docs/my-website/sidebars.js index 60306dd8ca..fdf2019cc2 100644 --- a/docs/my-website/sidebars.js +++ b/docs/my-website/sidebars.js @@ -444,6 +444,7 @@ const sidebars = { items: [ "tutorials/openweb_ui", "tutorials/msft_sso", + "tutorials/prompt_caching", "tutorials/tag_management", 'tutorials/litellm_proxy_aporia', { diff --git a/litellm/proxy/proxy_config.yaml b/litellm/proxy/proxy_config.yaml index 1d32d2d71e..dedf05b389 100644 --- a/litellm/proxy/proxy_config.yaml +++ b/litellm/proxy/proxy_config.yaml @@ -1,24 +1,18 @@ model_list: - - model_name: fake-openai-endpoint + - model_name: anhropic-auto-inject-cache-user-message litellm_params: - model: openai/fake - api_key: fake-key - api_base: https://exampleopenaiendpoint-production.up.railway.app/ - - model_name: openai/gpt-4o + model: anhropic/claude-3-5-sonnet-20240620 + api_key: os.environ/ANTHROPIC_API_KEY + cache_control_injection_points: + - location: message + role: user + + - model_name: anhropic-auto-inject-cache-system-message litellm_params: - model: openai/gpt-4o - api_key: fake-key + model: anhropic/claude-3-5-sonnet-20240620 + api_key: os.environ/ANTHROPIC_API_KEY + cache_control_injection_points: + - location: message + role: user -litellm_settings: - default_team_settings: - - team_id: test_dev - success_callback: ["langfuse", "s3"] - langfuse_secret: secret-test-key - langfuse_public_key: public-test-key - - team_id: my_workflows - success_callback: ["langfuse", "s3"] - langfuse_secret: secret-workflows-key - langfuse_public_key: public-workflows-key -router_settings: - enable_tag_filtering: True # 👈 Key Change \ No newline at end of file