mirror of
https://github.com/tiennm99/litellm.git
synced 2026-08-06 20:25:29 +00:00
fix(translation): google raw guard for message name beside cache markers (verifier-integration blocker)
The cache-marker token bound is computed over the IR, but the inbound parse drops the OpenAI message 'name' field (v1's generateContent transform ignores it on the wire), so name bytes were invisible to the bound while v1's is_prompt_caching_valid_prompt token-counts them: 100 one-char marked messages with 80-char names measured v1=4603 tokens (>= 1024, context cache created) against a v2 bound of 900 (served, uncached). New providers/google_genai/guard.py falls back before parse when any message carries 'name' AND a cache_control marker is present (name bytes cannot be bounded post-IR), registered in _RAW_GUARDS for vertex_ai and gemini; vertex_anthropic needs no row (v1 context caching is gemini-only and the anthropic wire ignores name on both sides). The verifier's attack script now finds no unsound shape. Corpus: a name_beside_marker fallback row (fails with the guard unregistered, verified by stash) and a message_name_without_marker quirk row pinning that both sides drop bare 'name' identically. Differential regenerated: 297 IDENTICAL / 55 FALLBACK / 0 divergent.
This commit is contained in:
@@ -90,6 +90,10 @@ translation/
|
||||
│ │ │ # (AI Studio refuses https media + forwards
|
||||
│ │ │ # function-call ids on gemini-3+); auth/host/api-
|
||||
│ │ │ # version differences are envelope, never here
|
||||
│ │ ├── guard.py # raw guard run BEFORE parse: message `name` beside
|
||||
│ │ │ # cache markers falls back (the IR drops `name`,
|
||||
│ │ │ # so its bytes are invisible to the cache-marker
|
||||
│ │ │ # token bound while v1 token-counts them)
|
||||
│ │ ├── serialize.py # cache-marker gate (conservative token bound:
|
||||
│ │ │ # UTF-8 bytes + per-message margin < 1024 proves
|
||||
│ │ │ # v1 skips the context-cache network call; any
|
||||
@@ -312,6 +316,8 @@ bound (UTF-8 bytes + per-message margin over the whole request) reaches
|
||||
gemini's 1024-token cache minimum (context-cache create is network I/O;
|
||||
below the bound v1 provably skips the call and ignores the markers), any
|
||||
media block in a marker-bearing request (v1 token-counts images at 250),
|
||||
message `name` beside cache markers (raw guard: the IR drops `name`, so
|
||||
its bytes cannot be bounded post-parse while v1 token-counts them),
|
||||
gs:// and http:// media plus AI-Studio https media (downloads), https media
|
||||
without a recognizable extension, file/pdf parts (inbound), params outside
|
||||
the IR (n, seed, penalties, modalities, audio, web_search_options,
|
||||
|
||||
@@ -60,6 +60,9 @@ from ..providers.google_genai import (
|
||||
from ..providers.google_genai import (
|
||||
serialize_request_vertex as google_serialize_request_vertex,
|
||||
)
|
||||
from ..providers.google_genai import (
|
||||
unsupported_request_shapes as google_unsupported_request_shapes,
|
||||
)
|
||||
from ..providers.openai_compat import parse_response as openai_compat_parse_response
|
||||
from ..providers.openai_compat import (
|
||||
serialize_request as openai_compat_serialize_request,
|
||||
@@ -128,13 +131,17 @@ _RESPONSE_DIALECTS: Mapping[Provider, ResponseDialect] = MappingProxyType(
|
||||
_RawGuard = Callable[[Mapping[str, object]], TranslationError | None]
|
||||
|
||||
_RAW_GUARDS: Mapping[Provider, _RawGuard] = MappingProxyType(
|
||||
# Same-family providers run a raw-shape fidelity guard BEFORE parse: the
|
||||
# inbound parse normalizes wire forms v1 forwards verbatim, so shapes it
|
||||
# cannot round-trip losslessly fall back to v1 as typed errors.
|
||||
# Raw-shape guards run BEFORE parse. Same-family providers use them for
|
||||
# fidelity (the inbound parse normalizes wire forms v1 forwards
|
||||
# verbatim); the google routes use one because the parse DROPS the
|
||||
# message ``name`` field, whose bytes the cache-marker token bound must
|
||||
# otherwise account for (verifier-integration blocker).
|
||||
{
|
||||
"openai_compat": openai_compat_unsupported_request_shapes,
|
||||
"azure": azure_unsupported_request_shapes,
|
||||
"azure_ai": azure_ai_unsupported_request_shapes,
|
||||
"vertex_ai": google_unsupported_request_shapes,
|
||||
"gemini": google_unsupported_request_shapes,
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@ One serializer family; the two providers differ only by the drift-list
|
||||
``target`` (see serialize.py) and their envelopes (seam-owned).
|
||||
"""
|
||||
|
||||
from .guard import unsupported_request_shapes
|
||||
from .response import parse_response
|
||||
from .serialize import serialize_request_studio, serialize_request_vertex
|
||||
from .stream import parse_event
|
||||
@@ -13,4 +14,5 @@ __all__ = (
|
||||
"parse_response",
|
||||
"serialize_request_studio",
|
||||
"serialize_request_vertex",
|
||||
"unsupported_request_shapes",
|
||||
)
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
"""Raw-shape guard for the google routes (vertex_ai gemini + AI Studio).
|
||||
|
||||
The cache-marker gate (serialize.py) proves v1's context-cache network call
|
||||
unreachable by bounding v1's token count from the IR — but the inbound parse
|
||||
drops the OpenAI message ``name`` field (no IR field; v1's generateContent
|
||||
transform ignores it on the wire), so ``name`` bytes are INVISIBLE to the
|
||||
bound while v1's ``is_prompt_caching_valid_prompt`` token-counts them
|
||||
(openai_token_counter charges for ``name``). A request carrying BOTH a cache
|
||||
marker and any message ``name`` therefore falls back to v1 before parse
|
||||
(verifier-integration blocker): the bytes cannot be bounded post-IR.
|
||||
|
||||
Runs over the UNTRUSTED raw body BEFORE parse; every check is structural and
|
||||
conservative — the guard can only widen the fallback surface, never change a
|
||||
served body. vertex_anthropic needs no row here: v1's context caching is
|
||||
gemini-only (check_and_create_cache lives on the generateContent path) and
|
||||
the anthropic wire ignores ``name`` on both sides.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Mapping, Sequence
|
||||
from typing import cast
|
||||
|
||||
from litellm.constants import DEFAULT_MAX_RECURSE_DEPTH
|
||||
|
||||
from ...errors import TranslationError
|
||||
|
||||
_Raw = Mapping[str, object]
|
||||
|
||||
|
||||
def unsupported_request_shapes(raw: _Raw) -> TranslationError | None:
|
||||
messages = raw.get("messages")
|
||||
if not isinstance(messages, Sequence) or isinstance(messages, str):
|
||||
return None # parse rejects malformed messages with its own error
|
||||
entries = cast(Sequence[object], messages)
|
||||
has_name = any(isinstance(entry, Mapping) and "name" in entry for entry in entries)
|
||||
if not has_name:
|
||||
return None
|
||||
if _carries_cache_control(entries, 0):
|
||||
return TranslationError.of_unsupported(
|
||||
"message 'name' beside cache_control markers: the IR drops 'name'"
|
||||
" so its bytes are invisible to the cache-marker token bound,"
|
||||
" while v1's check_and_create_cache token-counts them"
|
||||
)
|
||||
return None
|
||||
|
||||
|
||||
def _carries_cache_control(value: object, depth: int) -> bool:
|
||||
if depth > DEFAULT_MAX_RECURSE_DEPTH:
|
||||
# exhaustion must never ADMIT a request: treat the unscannable tail
|
||||
# as if it carried the marker (fall back to v1)
|
||||
return True
|
||||
if isinstance(value, Mapping):
|
||||
mapping = cast(Mapping[str, object], value)
|
||||
if "cache_control" in mapping:
|
||||
return True
|
||||
return any(_carries_cache_control(item, depth + 1) for item in mapping.values())
|
||||
if isinstance(value, Sequence) and not isinstance(value, str):
|
||||
return any(
|
||||
_carries_cache_control(item, depth + 1)
|
||||
for item in cast(Sequence[object], value)
|
||||
)
|
||||
return False
|
||||
@@ -6,7 +6,7 @@ Bedrock and google rows additionally pin the characterization-corpus
|
||||
snapshot, so each row proves snapshot == v1-at-HEAD == v2. Regenerate with:
|
||||
`python -m tests.test_litellm.translation.generate_differential_report`
|
||||
|
||||
- commit: 96c2a14264
|
||||
- commit: ff5c320127
|
||||
|
||||
## anthropic: request bodies (v1 map_openai_params + transform_request vs v2)
|
||||
|
||||
@@ -396,6 +396,7 @@ snapshot, so each row proves snapshot == v1-at-HEAD == v2. Regenerate with:
|
||||
- IDENTICAL: quirk gemini3_default_temperature_and_level (vertex_ai/gemini-3-pro-preview)
|
||||
- IDENTICAL: quirk gemini3_studio_forwards_function_call_ids (gemini/gemini-3-pro-preview)
|
||||
- IDENTICAL: quirk image_url_format_override (vertex_ai/gemini-2.5-pro)
|
||||
- IDENTICAL: quirk message_name_without_marker (vertex_ai/gemini-2.5-pro)
|
||||
- IDENTICAL: quirk multi_system_messages_two_parts (vertex_ai/gemini-2.5-pro)
|
||||
- IDENTICAL: quirk parallel_tool_calls_never_reaches_wire (vertex_ai/gemini-2.5-pro)
|
||||
- IDENTICAL: quirk reasoning_effort_minimal_model_budget (vertex_ai/gemini-2.5-pro)
|
||||
@@ -411,6 +412,7 @@ snapshot, so each row proves snapshot == v1-at-HEAD == v2. Regenerate with:
|
||||
- IDENTICAL: quirk vertex_top_k_passthrough (vertex_ai/gemini-2.5-pro)
|
||||
- FALLBACK (v1 serves it): cache-marker token bound cjk_under_char_limit (v1's check_and_create_cache may create the context cache; the byte+margin bound fails closed)
|
||||
- FALLBACK (v1 serves it): cache-marker token bound emoji_under_char_limit (v1's check_and_create_cache may create the context cache; the byte+margin bound fails closed)
|
||||
- FALLBACK (v1 serves it): cache-marker token bound name_beside_marker (v1's check_and_create_cache may create the context cache; the byte+margin bound fails closed)
|
||||
- FALLBACK (v1 serves it): cache-marker token bound unmarked_image_beside_marker (v1's check_and_create_cache may create the context cache; the byte+margin bound fails closed)
|
||||
|
||||
## gemini: responses (snapshot == v1 transform_response == v2)
|
||||
|
||||
@@ -132,6 +132,16 @@ _WEATHER_TOOL = {
|
||||
|
||||
# (alias, case, drop_params) quirks; every row references v1 in-process.
|
||||
QUIRKS = {
|
||||
"message_name_without_marker": (
|
||||
# no marker -> the name+marker guard stays out of the way, and the
|
||||
# quirk pins that BOTH sides drop 'name' from the wire identically.
|
||||
"vertex_ai/gemini-2.5-pro",
|
||||
{
|
||||
"messages": [{"role": "user", "name": "alice", "content": "hi"}],
|
||||
"params": {"max_tokens": 64},
|
||||
},
|
||||
False,
|
||||
),
|
||||
"cache_marker_cjk_sublimit": (
|
||||
# 80 CJK chars = 240 UTF-8 bytes: under the gate's byte+margin bound,
|
||||
# so BOTH sides ignore the marker inline (v1 token-counts < 1024 and
|
||||
@@ -385,6 +395,30 @@ CACHE_GATE_FALLBACKS = {
|
||||
}
|
||||
],
|
||||
},
|
||||
"name_beside_marker": {
|
||||
# the IR drops message 'name' (v1's generateContent transform
|
||||
# ignores it on the wire), so its bytes are invisible to the
|
||||
# cache-marker token bound while v1's token_counter charges for
|
||||
# them (verifier-integration blocker: 100 x 1-char marked messages
|
||||
# with 80-char names -> v1 counts 4603 >= 1024 and creates the
|
||||
# cache); name+marker coexistence falls back via the google raw
|
||||
# guard before parse.
|
||||
"model": "gemini-2.5-pro",
|
||||
"max_tokens": 64,
|
||||
"messages": [
|
||||
{
|
||||
"role": "user",
|
||||
"name": "a" * 80,
|
||||
"content": [
|
||||
{
|
||||
"type": "text",
|
||||
"text": "x",
|
||||
"cache_control": {"type": "ephemeral"},
|
||||
}
|
||||
],
|
||||
}
|
||||
],
|
||||
},
|
||||
"unmarked_image_beside_marker": {
|
||||
# The image carries NO cache_control, but v1 token-counts it at
|
||||
# DEFAULT_IMAGE_TOKEN_COUNT (250) inside the continuous cached block.
|
||||
|
||||
Reference in New Issue
Block a user