Merge pull request #22553 from dsteeley/fix/streaming-multi-tool-call-premature-finish

fix(streaming): output_item.done for function_call must not emit finish_reason
This commit is contained in:
Sameer Kankute 2026-03-05 15:05:43 +05:30 committed by GitHub
commit 9a13c76e2f
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
2 changed files with 385 additions and 12 deletions

View File

@ -1028,12 +1028,16 @@ class OpenAiResponsesToChatCompletionStreamIterator(BaseModelResponseIterator):
if provider_specific_fields:
tool_call_chunk.provider_specific_fields = provider_specific_fields # type: ignore
# Do NOT emit finish_reason here — response.completed handles the terminal
# finish_reason. Emitting "tool_calls" here would prematurely terminate
# the stream before subsequent tool calls arrive (same fix as #17246 for
# the message-type branch).
return ModelResponseStream(
choices=[
StreamingChoices(
index=0,
delta=Delta(tool_calls=[tool_call_chunk]),
finish_reason="tool_calls",
delta=Delta(),
finish_reason=None,
)
]
)

View File

@ -738,10 +738,12 @@ def test_response_completed_with_message_only_emits_stop_finish_reason():
)
def test_function_call_done_emits_is_finished():
def test_function_call_done_does_not_emit_finish_reason():
"""
Test that OUTPUT_ITEM_DONE for a function_call still emits is_finished=True.
This preserves existing behavior for tool_calls.
Test that OUTPUT_ITEM_DONE for a function_call does NOT emit finish_reason.
The response.completed event handles the terminal finish_reason correctly.
Emitting finish_reason here would prematurely terminate the stream in multi-tool
scenarios (same fix as #17246 for the message-type branch).
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
@ -761,11 +763,14 @@ def test_function_call_done_emits_is_finished():
result = iterator.chunk_parser(chunk)
# function_call completion should emit finish_reason='tool_calls'
# function_call completion should NOT emit finish_reason — response.completed handles it
assert len(result.choices) > 0, "result should have choices"
assert result.choices[0].finish_reason == "tool_calls", "function_call should emit finish_reason='tool_calls'"
assert result.choices[0].delta.tool_calls is not None and len(result.choices[0].delta.tool_calls) > 0, (
"function_call should include tool_calls"
assert result.choices[0].finish_reason is None, (
"output_item.done for function_call must not emit finish_reason; "
"response.completed is responsible for the terminal finish_reason"
)
assert not result.choices[0].delta.tool_calls, (
"output_item.done for function_call must not include a duplicate tool_calls delta"
)
@ -824,14 +829,16 @@ def test_text_plus_tool_calls_sequence():
"message done should not have finish_reason"
)
# Check function_call done (index 5) DOES have finish_reason='tool_calls'
# Check function_call done (index 5) does NOT have finish_reason set
# (response.completed is responsible for the terminal finish_reason)
function_done_result = results[5]
assert len(function_done_result.choices) > 0, "function_call done should have choices"
assert function_done_result.choices[0].finish_reason == "tool_calls", (
"function_call done should have finish_reason='tool_calls'"
assert function_done_result.choices[0].finish_reason is None, (
"output_item.done for function_call must not emit finish_reason"
)
# Check response.completed (index 6) has finish_reason='stop'
# (the mock chunk has no nested 'response' data, so has_function_calls is False → 'stop')
completed_result = results[6]
assert len(completed_result.choices) > 0, "response.completed should have choices"
assert completed_result.choices[0].finish_reason == "stop", "response.completed should have finish_reason='stop'"
@ -1320,6 +1327,155 @@ def test_transform_response_preserves_annotations():
print("✓ Annotations from Responses API are correctly preserved in Chat Completions format")
def test_multi_tool_call_stream_no_premature_finish():
"""
Regression test for multi-tool-call streaming bug.
When a response contains multiple tool calls, the stream used to be prematurely
terminated after the first output_item.done event because that handler emitted
finish_reason="tool_calls". This caused ~58% of streaming requests with multiple
tool calls to fail.
The fix: output_item.done for function_call emits delta=Delta() and finish_reason=None.
Only response.completed emits the terminal finish_reason.
Synthetic event sequence:
response.created
response.output_item.added (function_call: read_file, call_id: call_1)
response.function_call_arguments.delta (read_file args)
response.output_item.done (function_call: read_file) <- must NOT end stream
response.output_item.added (function_call: list_dir, call_id: call_2)
response.function_call_arguments.delta (list_dir args)
response.output_item.done (function_call: list_dir) <- must NOT end stream
response.completed (response with 2 function_call outputs) <- terminal
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
chunks = [
# 0: response created
{"type": "response.created", "response": {"id": "resp_001", "status": "in_progress"}},
# 1: first tool call added
{
"type": "response.output_item.added",
"item": {"type": "function_call", "name": "read_file", "call_id": "call_1"},
},
# 2: first tool call arguments delta
{"type": "response.function_call_arguments.delta", "delta": '{"path":"/etc/hostname"}'},
# 3: first tool call done ← must NOT emit finish_reason
{
"type": "response.output_item.done",
"item": {
"type": "function_call",
"name": "read_file",
"call_id": "call_1",
"arguments": '{"path":"/etc/hostname"}',
},
},
# 4: second tool call added
{
"type": "response.output_item.added",
"item": {"type": "function_call", "name": "list_dir", "call_id": "call_2"},
},
# 5: second tool call arguments delta
{"type": "response.function_call_arguments.delta", "delta": '{"path":"/tmp"}'},
# 6: second tool call done ← must NOT emit finish_reason
{
"type": "response.output_item.done",
"item": {
"type": "function_call",
"name": "list_dir",
"call_id": "call_2",
"arguments": '{"path":"/tmp"}',
},
},
# 7: response completed with both tool calls in output ← ONLY terminal chunk
{
"type": "response.completed",
"response": {
"id": "resp_001",
"status": "completed",
"output": [
{
"type": "function_call",
"name": "read_file",
"call_id": "call_1",
"arguments": '{"path":"/etc/hostname"}',
},
{
"type": "function_call",
"name": "list_dir",
"call_id": "call_2",
"arguments": '{"path":"/tmp"}',
},
],
},
},
]
results = [iterator.chunk_parser(chunk) for chunk in chunks]
# 1. output_item.done events (indices 3 and 6) must NOT emit finish_reason
for done_idx, label in [(3, "read_file done"), (6, "list_dir done")]:
r = results[done_idx]
assert r is not None, f"{label}: chunk_parser must return a result"
assert len(r.choices) > 0, f"{label}: result must have choices"
assert r.choices[0].finish_reason is None, (
f"{label}: output_item.done must not emit finish_reason (stream would terminate prematurely)"
)
assert not r.choices[0].delta.tool_calls, (
f"{label}: output_item.done must not include a duplicate tool_calls delta"
)
# 2. output_item.added events (indices 1 and 4) should carry name + call_id
for added_idx, expected_name, expected_call_id in [
(1, "read_file", "call_1"),
(4, "list_dir", "call_2"),
]:
r = results[added_idx]
if r is not None and r.choices and r.choices[0].delta.tool_calls:
tc = r.choices[0].delta.tool_calls[0]
assert tc.function.name == expected_name, (
f"output_item.added for {expected_name}: tool_call name mismatch"
)
assert tc.id == expected_call_id, (
f"output_item.added for {expected_name}: call_id mismatch"
)
# 3. argument delta events (indices 2 and 5) should carry arguments
for delta_idx, expected_args, label in [
(2, '{"path":"/etc/hostname"}', "read_file args"),
(5, '{"path":"/tmp"}', "list_dir args"),
]:
r = results[delta_idx]
if r is not None and r.choices and r.choices[0].delta.tool_calls:
tc = r.choices[0].delta.tool_calls[0]
assert tc.function.arguments == expected_args, (
f"{label}: argument delta mismatch"
)
# 4. Only response.completed (index 7) emits the terminal finish_reason
completed_result = results[7]
assert completed_result is not None, "response.completed must return a result"
assert len(completed_result.choices) > 0, "response.completed must have choices"
assert completed_result.choices[0].finish_reason == "tool_calls", (
"response.completed with function_call outputs must emit finish_reason='tool_calls'"
)
# 5. No chunk before the last one should have finish_reason set
for idx, r in enumerate(results[:-1]):
if r is not None and r.choices:
assert r.choices[0].finish_reason is None, (
f"Chunk at index {idx} (type={chunks[idx]['type']!r}) must not emit finish_reason "
f"— only response.completed should terminate the stream"
)
print("✓ Multi-tool-call stream completes without premature finish_reason termination")
# =============================================================================
# Tests for issue #21331: Parallel tool call indices in streaming
# =============================================================================
@ -1409,3 +1565,216 @@ def test_streaming_parallel_tool_calls_have_distinct_indices():
f"Event {chunk['type']}: expected tool_call.index={expected_index}, "
f"got {tc.index}"
)
# =============================================================================
# Comprehensive integration test: parallel tool calls with split argument deltas
# =============================================================================
def test_parallel_tool_calls_comprehensive_streaming_integration():
"""
Comprehensive integration test for parallel tool calls via Responses API streaming.
Regression test combining all fix invariants in a single end-to-end scenario
with split argument deltas the exact event sequence that was broken before
the fix to output_item.done.
Synthesized SSE event sequence:
response.created
response.output_item.added {output_index:0, type:function_call, call_id:call_1, name:read_file}
response.function_call_arguments.delta {output_index:0, delta:'{"path"'}
response.function_call_arguments.delta {output_index:0, delta:'":"/etc/foo"}'}
response.output_item.done {output_index:0, item:{type:function_call, call_id:call_1}}
response.output_item.added {output_index:1, type:function_call, call_id:call_2, name:list_dir}
response.function_call_arguments.delta {output_index:1, delta:'{"path"'}
response.function_call_arguments.delta {output_index:1, delta:'":"/tmp"}'}
response.output_item.done {output_index:1, item:{type:function_call, call_id:call_2}}
response.completed {response:{status:completed, output:[call_1, call_2]}}
Asserts:
1. No output_item.done chunk emits finish_reason (no premature stream termination)
2. Each call_id appears exactly once in assembled tool_call IDs (no duplicates)
3. Final assembled arguments are correct split deltas concatenate to valid JSON
4. Exactly one finish event, at the final response.completed chunk
5. Two parallel tool calls have distinct indices (output_index 0 and 1)
"""
from litellm.completion_extras.litellm_responses_transformation.transformation import (
OpenAiResponsesToChatCompletionStreamIterator,
)
chunks = [
# 0: response.created
{"type": "response.created", "response": {"id": "resp_001", "status": "in_progress"}},
# 1: call_1 (read_file) added — output_index=0
{
"type": "response.output_item.added",
"output_index": 0,
"item": {"type": "function_call", "name": "read_file", "call_id": "call_1"},
},
# 2: call_1 argument delta part 1 — split across two deltas
{
"type": "response.function_call_arguments.delta",
"output_index": 0,
"delta": '{"path":',
},
# 3: call_1 argument delta part 2
{
"type": "response.function_call_arguments.delta",
"output_index": 0,
"delta": '"/etc/foo"}',
},
# 4: call_1 done — must NOT emit finish_reason or duplicate tool_call chunk
{
"type": "response.output_item.done",
"output_index": 0,
"item": {
"type": "function_call",
"name": "read_file",
"call_id": "call_1",
"arguments": '{"path":"/etc/foo"}', # full JSON, assembled from the two deltas
},
},
# 5: call_2 (list_dir) added — output_index=1
{
"type": "response.output_item.added",
"output_index": 1,
"item": {"type": "function_call", "name": "list_dir", "call_id": "call_2"},
},
# 6: call_2 argument delta part 1
{
"type": "response.function_call_arguments.delta",
"output_index": 1,
"delta": '{"path":',
},
# 7: call_2 argument delta part 2
{
"type": "response.function_call_arguments.delta",
"output_index": 1,
"delta": '"/tmp"}',
},
# 8: call_2 done — must NOT emit finish_reason or duplicate tool_call chunk
{
"type": "response.output_item.done",
"output_index": 1,
"item": {
"type": "function_call",
"name": "list_dir",
"call_id": "call_2",
"arguments": '{"path":"/tmp"}',
},
},
# 9: response.completed — the ONLY terminal chunk
{
"type": "response.completed",
"response": {
"id": "resp_001",
"status": "completed",
"output": [
{
"type": "function_call",
"name": "read_file",
"call_id": "call_1",
"arguments": '{"path":"/etc/foo"}',
},
{
"type": "function_call",
"name": "list_dir",
"call_id": "call_2",
"arguments": '{"path":"/tmp"}',
},
],
},
},
]
iterator = OpenAiResponsesToChatCompletionStreamIterator(streaming_response=None, sync_stream=True)
results = [iterator.chunk_parser(chunk) for chunk in chunks]
# 1. output_item.done events (indices 4 and 8) must NOT emit finish_reason
for done_idx, label in [(4, "read_file done"), (8, "list_dir done")]:
r = results[done_idx]
assert r is not None, f"{label}: chunk_parser must return a result"
assert len(r.choices) > 0, f"{label}: result must have choices"
assert r.choices[0].finish_reason is None, (
f"{label}: output_item.done must not emit finish_reason "
f"(would prematurely terminate stream before subsequent tool calls arrive)"
)
assert not r.choices[0].delta.tool_calls, (
f"{label}: output_item.done must not emit a duplicate tool_calls delta"
)
# 2. Each call_id appears exactly once in assembled tool_call IDs
# Only output_item.added emits id-bearing tool_call chunks; output_item.done emits Delta()
all_tool_call_ids = [
tc.id
for r in results
if r is not None and r.choices and r.choices[0].delta.tool_calls
for tc in r.choices[0].delta.tool_calls
if tc.id
]
assert all_tool_call_ids.count("call_1") == 1, (
f"call_1 must appear exactly once in assembled tool_call IDs, "
f"got {all_tool_call_ids.count('call_1')} (duplicates indicate output_item.done still emits tool_call)"
)
assert all_tool_call_ids.count("call_2") == 1, (
f"call_2 must appear exactly once in assembled tool_call IDs, "
f"got {all_tool_call_ids.count('call_2')} (duplicates indicate output_item.done still emits tool_call)"
)
# 3. Final assembled arguments are correct when split deltas are concatenated
# output_item.added emits arguments="" (empty); the two deltas provide the content
assembled_args: dict = {}
for r in results:
if r is None or not r.choices:
continue
tool_calls = r.choices[0].delta.tool_calls
if not tool_calls:
continue
for tc in tool_calls:
if tc.function and tc.function.arguments:
idx = tc.index
assembled_args[idx] = assembled_args.get(idx, "") + tc.function.arguments
# delta 1 = '{"path":' + delta 2 = '"/etc/foo"}' → '{"path":"/etc/foo"}'
assert assembled_args.get(0) == '{"path":"/etc/foo"}', (
f"Assembled args for index 0 (read_file): "
f"expected '{{\"path\":\"/etc/foo\"}}', got '{assembled_args.get(0)}'"
)
# delta 1 = '{"path":' + delta 2 = '"/tmp"}' → '{"path":"/tmp"}'
assert assembled_args.get(1) == '{"path":"/tmp"}', (
f"Assembled args for index 1 (list_dir): "
f"expected '{{\"path\":\"/tmp\"}}', got '{assembled_args.get(1)}'"
)
# 4. Stream terminates with exactly one finish event, at the final response.completed chunk
finish_events = [
(i, r.choices[0].finish_reason)
for i, r in enumerate(results)
if r is not None and r.choices and r.choices[0].finish_reason
]
assert len(finish_events) == 1, (
f"Expected exactly 1 finish event, got {len(finish_events)}: {finish_events}"
)
assert finish_events[0][0] == len(chunks) - 1, (
f"Finish event must be at the last chunk (index {len(chunks) - 1}), "
f"but was at index {finish_events[0][0]}"
)
assert finish_events[0][1] == "tool_calls", (
f"Terminal finish_reason must be 'tool_calls', got '{finish_events[0][1]}'"
)
# 5. Parallel tool calls have distinct indices matching output_index (0 and 1)
# Collect indices from output_item.added chunks only (they carry the call id)
added_tool_call_indices = [
tc.index
for r in results
if r is not None and r.choices and r.choices[0].delta.tool_calls
for tc in r.choices[0].delta.tool_calls
if tc.id # output_item.added chunks carry the id; argument deltas do not
]
assert set(added_tool_call_indices) == {0, 1}, (
f"Parallel tool calls must have distinct indices {{0, 1}}, got: {set(added_tool_call_indices)}"
)
print("✓ Parallel tool calls with split argument deltas stream correctly end-to-end")