Add REASONING_SUMMARY_TEXT_DONE and REASONING_SUMMARY_PART_DONE events in streaming

This commit is contained in:
Sameer Kankute
2026-01-21 14:32:05 +05:30
parent 828ff1c99e
commit 4127654807
2 changed files with 198 additions and 1 deletions
@@ -24,6 +24,8 @@ from litellm.types.llms.openai import (
OutputTextDeltaEvent,
OutputTextDoneEvent,
ReasoningSummaryTextDeltaEvent,
ReasoningSummaryTextDoneEvent,
ReasoningSummaryPartDoneEvent,
ResponseCompletedEvent,
ResponseCreatedEvent,
ResponseInProgressEvent,
@@ -41,7 +43,6 @@ from litellm.types.utils import (
TextCompletionResponse,
)
class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
"""
Async iterator for processing streaming responses from the Responses API.
@@ -90,8 +91,15 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
self._final_tool_events_queued: bool = False
self._sequence_number: int = 0
self._cached_reasoning_item_id: Optional[str] = None
self._sent_reasoning_summary_text_done_event: bool = False
self._sent_reasoning_summary_part_done_event: bool = False
self._reasoning_summary_text: str = ""
# -- GENERIC RESPONSE-EVENTS PENDING QUEUE as required by fix --
self._pending_response_events: List[BaseLiteLLMOpenAIResponseObject] = []
self._reasoning_active = False
self._reasoning_done_emitted = False
self._reasoning_item_id = None
def _get_or_assign_tool_output_index(self, call_id: str) -> int:
existing = self._tool_output_index_by_call_id.get(call_id)
@@ -102,6 +110,23 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
self._tool_output_index_by_call_id[call_id] = idx
return idx
def _is_reasoning_end(self, chunk):
delta = chunk.choices[0].delta
# if this indicates reasoning content, don't consider reasoning ended
if hasattr(delta, "reasoning_content") and delta.reasoning_content:
return False
if hasattr(delta, "thinking_blocks") and delta.thinking_blocks:
return False
return (
delta.content
or delta.function_call
or delta.tool_calls
or chunk.choices[0].finish_reason is not None
)
def _queue_tool_call_delta_events(self, tool_calls: object) -> None:
"""
Convert chat-completions streaming `tool_calls` deltas into Responses API streaming events.
@@ -405,6 +430,70 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
),
)
def create_reasoning_summary_text_done_event(
self,
reasoning_item_id: str,
reasoning_content: str,
sequence_number: int,
) -> ReasoningSummaryTextDoneEvent:
"""
Create response.reasoning_summary_text.done event.
Example:
{
"type": "response.reasoning_summary_text.done",
"item_id": "rs_0c5dae30e53172980069708ba2f59c8197b71ca9820edad07c",
"output_index": 0,
"sequence_number": 97,
"summary_index": 0,
"text": "**Clarifying the first humans**\n\nThe I'm addressing the user's specific interest."
}
"""
return ReasoningSummaryTextDoneEvent(
type=ResponsesAPIStreamEvents.REASONING_SUMMARY_TEXT_DONE,
item_id=reasoning_item_id,
output_index=0,
sequence_number=sequence_number,
summary_index=0,
text=reasoning_content,
)
def create_reasoning_summary_part_done_event(
self,
reasoning_item_id: str,
reasoning_content: str,
sequence_number: int,
) -> ReasoningSummaryPartDoneEvent:
"""
Create response.reasoning_summary_part.done event.
Example:
{
"type": "response.reasoning_summary_part.done",
"item_id": "rs_0c5dae30e53172980069708ba2f59c8197b71ca9820edad07c",
"output_index": 0,
"part": {
"type": "summary_text",
"text": "**Clarifying the first humans**\n\nThe earlier hominins. It feels important to ensure I'm addressing the user's specific interest."
},
"sequence_number": 98,
"summary_index": 0
}
"""
return ReasoningSummaryPartDoneEvent(
type=ResponsesAPIStreamEvents.REASONING_SUMMARY_PART_DONE,
item_id=reasoning_item_id,
output_index=0,
sequence_number=sequence_number,
summary_index=0,
part=BaseLiteLLMOpenAIResponseObject(
**{
"type": "summary_text",
"text": reasoning_content,
}
),
)
def create_output_text_done_event(
self, litellm_complete_object: ModelResponse
) -> OutputTextDoneEvent:
@@ -489,6 +578,50 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
),
)
def create_reasoning_output_item_done_event(
self,
reasoning_item_id: str,
reasoning_content: str,
sequence_number: int,
) -> OutputItemDoneEvent:
"""
Create response.output_item.done event for reasoning items.
Example:
{
"type": "response.output_item.done",
"output_index": 0,
"sequence_number": 99,
"item": {
"id": "rs_0c5dae30e53172980069708ba2f59c8197b71ca9820edad07c",
"type": "reasoning",
"summary": [
{
"type": "summary_text",
"text": "**Clarifying the first humans**..."
}
]
}
}
"""
return OutputItemDoneEvent(
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_DONE,
output_index=0,
sequence_number=sequence_number,
item=BaseLiteLLMOpenAIResponseObject(
**{
"id": reasoning_item_id,
"type": "reasoning",
"summary": [
{
"type": "summary_text",
"text": reasoning_content,
}
],
}
),
)
def return_default_done_events(
self, litellm_complete_object: ModelResponse
) -> Optional[BaseLiteLLMOpenAIResponseObject]:
@@ -570,8 +703,10 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
# Reasoning-first
if hasattr(delta, "reasoning_content") and delta.reasoning_content:
self._reasoning_active = True
if self._cached_reasoning_item_id is None:
self._cached_reasoning_item_id = f"rs_{uuid.uuid4()}"
self._reasoning_item_id = self._cached_reasoning_item_id
event = OutputItemAddedEvent(
type=ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED,
@@ -639,6 +774,46 @@ class LiteLLMCompletionStreamingIterator(ResponsesAPIStreamingIterator):
self._ensure_output_item_for_chunk(chunk)
# Proceed to transformation
self.collected_chat_completion_chunks.append(chunk)
if self._reasoning_active and not self._reasoning_done_emitted:
# get raw ModelResponse
text_reasoning = self.create_litellm_model_response()
# reasoning_content only
if self._is_reasoning_end(chunk):
reasoning_content = ""
# best effort to obtain reasoning_content from chat model response
if text_reasoning and text_reasoning.choices and hasattr(text_reasoning.choices[0].message, "reasoning_content"):
reasoning_content = getattr(text_reasoning.choices[0].message, "reasoning_content", "") or ""
# Create text.done event first with its own sequence number
self._sequence_number += 1
text_done_event = self.create_reasoning_summary_text_done_event(
reasoning_item_id=self._reasoning_item_id,
reasoning_content=reasoning_content,
sequence_number=self._sequence_number
)
# Create part.done event second with its own sequence number
self._sequence_number += 1
part_done_event = self.create_reasoning_summary_part_done_event(
reasoning_item_id=self._reasoning_item_id,
reasoning_content=reasoning_content,
sequence_number=self._sequence_number
)
self._sequence_number += 1
reasoning_output_item_done_event = self.create_reasoning_output_item_done_event(
reasoning_item_id=self._reasoning_item_id,
reasoning_content=reasoning_content,
sequence_number=self._sequence_number
)
self._pending_response_events.extend([
text_done_event,
part_done_event,
reasoning_output_item_done_event,
])
self._reasoning_done_emitted = True
self._reasoning_active = False
response_api_chunk = (
self._transform_chat_completion_chunk_to_response_api_chunk(
chunk
+22
View File
@@ -1248,6 +1248,8 @@ class ResponsesAPIStreamEvents(str, Enum):
# Reasoning summary events
RESPONSE_PART_ADDED = "response.reasoning_summary_part.added"
REASONING_SUMMARY_TEXT_DELTA = "response.reasoning_summary_text.delta"
REASONING_SUMMARY_TEXT_DONE = "response.reasoning_summary_text.done"
REASONING_SUMMARY_PART_DONE = "response.reasoning_summary_part.done"
# Output item events
OUTPUT_ITEM_ADDED = "response.output_item.added"
@@ -1337,6 +1339,24 @@ class ReasoningSummaryTextDeltaEvent(BaseLiteLLMOpenAIResponseObject):
delta: str
class ReasoningSummaryTextDoneEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.REASONING_SUMMARY_TEXT_DONE]
item_id: str
output_index: int
sequence_number: int
summary_index: int
text: str
class ReasoningSummaryPartDoneEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.REASONING_SUMMARY_PART_DONE]
item_id: str
output_index: int
sequence_number: int
summary_index: int
part: BaseLiteLLMOpenAIResponseObject
class OutputItemAddedEvent(BaseLiteLLMOpenAIResponseObject):
type: Literal[ResponsesAPIStreamEvents.OUTPUT_ITEM_ADDED]
output_index: int
@@ -1591,6 +1611,8 @@ ResponsesAPIStreamingResponse = Annotated[
ResponseIncompleteEvent,
ResponsePartAddedEvent,
ReasoningSummaryTextDeltaEvent,
ReasoningSummaryTextDoneEvent,
ReasoningSummaryPartDoneEvent,
OutputItemAddedEvent,
OutputItemDoneEvent,
ContentPartAddedEvent,