fix: close stream on cancellation for SambaNova and Google OpenAI services
This commit is contained in:
@@ -91,6 +91,9 @@ class GoogleLLMOpenAIBetaService(OpenAILLMService):
|
|||||||
ChatCompletionChunk
|
ChatCompletionChunk
|
||||||
] = await self._stream_chat_completions_specific_context(context)
|
] = await self._stream_chat_completions_specific_context(context)
|
||||||
|
|
||||||
|
# Use context manager to ensure stream is closed on cancellation/exception.
|
||||||
|
# Without this, CancelledError during iteration leaves the underlying socket open.
|
||||||
|
async with chunk_stream:
|
||||||
async for chunk in chunk_stream:
|
async for chunk in chunk_stream:
|
||||||
if chunk.usage:
|
if chunk.usage:
|
||||||
tokens = LLMTokenUsage(
|
tokens = LLMTokenUsage(
|
||||||
|
|||||||
@@ -131,6 +131,9 @@ class SambaNovaLLMService(OpenAILLMService): # type: ignore
|
|||||||
else self._stream_chat_completions_universal_context(context)
|
else self._stream_chat_completions_universal_context(context)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Use context manager to ensure stream is closed on cancellation/exception.
|
||||||
|
# Without this, CancelledError during iteration leaves the underlying socket open.
|
||||||
|
async with chunk_stream:
|
||||||
async for chunk in chunk_stream:
|
async for chunk in chunk_stream:
|
||||||
if chunk.usage:
|
if chunk.usage:
|
||||||
tokens = LLMTokenUsage(
|
tokens = LLMTokenUsage(
|
||||||
|
|||||||
81
tests/test_google_llm_openai.py
Normal file
81
tests/test_google_llm_openai.py
Normal file
@@ -0,0 +1,81 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024-2026, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
"""Unit tests for Google LLM OpenAI Beta service."""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
import warnings
|
||||||
|
from unittest.mock import AsyncMock, patch
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from pipecat.processors.aggregators.openai_llm_context import OpenAILLMContext
|
||||||
|
|
||||||
|
try:
|
||||||
|
from pipecat.services.google.llm_openai import GoogleLLMOpenAIBetaService
|
||||||
|
|
||||||
|
google_available = True
|
||||||
|
except Exception:
|
||||||
|
google_available = False
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
@pytest.mark.skipif(not google_available, reason="Google dependencies not installed")
|
||||||
|
async def test_google_llm_openai_stream_closed_on_cancellation():
|
||||||
|
"""Test that the stream is closed when CancelledError occurs during iteration.
|
||||||
|
|
||||||
|
This prevents socket leaks when the pipeline is interrupted (e.g., user interruption).
|
||||||
|
See issue #3639.
|
||||||
|
"""
|
||||||
|
with patch.object(GoogleLLMOpenAIBetaService, "create_client"):
|
||||||
|
with warnings.catch_warnings():
|
||||||
|
warnings.simplefilter("ignore", DeprecationWarning)
|
||||||
|
service = GoogleLLMOpenAIBetaService(api_key="test-key", model="test-model")
|
||||||
|
service._client = AsyncMock()
|
||||||
|
|
||||||
|
stream_closed = False
|
||||||
|
|
||||||
|
class MockAsyncStream:
|
||||||
|
"""Mock AsyncStream that tracks close() calls and raises CancelledError."""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.iteration_count = 0
|
||||||
|
|
||||||
|
async def __aenter__(self):
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
||||||
|
nonlocal stream_closed
|
||||||
|
stream_closed = True
|
||||||
|
return False
|
||||||
|
|
||||||
|
def __aiter__(self):
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __anext__(self):
|
||||||
|
self.iteration_count += 1
|
||||||
|
if self.iteration_count > 1:
|
||||||
|
raise asyncio.CancelledError()
|
||||||
|
mock_chunk = AsyncMock()
|
||||||
|
mock_chunk.usage = None
|
||||||
|
mock_chunk.choices = []
|
||||||
|
return mock_chunk
|
||||||
|
|
||||||
|
mock_stream = MockAsyncStream()
|
||||||
|
|
||||||
|
service._stream_chat_completions_specific_context = AsyncMock(return_value=mock_stream)
|
||||||
|
service.start_ttfb_metrics = AsyncMock()
|
||||||
|
service.stop_ttfb_metrics = AsyncMock()
|
||||||
|
service.start_llm_usage_metrics = AsyncMock()
|
||||||
|
|
||||||
|
context = OpenAILLMContext(
|
||||||
|
messages=[{"role": "user", "content": "Hello"}],
|
||||||
|
)
|
||||||
|
|
||||||
|
with pytest.raises(asyncio.CancelledError):
|
||||||
|
await service._process_context(context)
|
||||||
|
|
||||||
|
assert stream_closed, "Stream should be closed even when CancelledError occurs"
|
||||||
72
tests/test_sambanova_llm.py
Normal file
72
tests/test_sambanova_llm.py
Normal file
@@ -0,0 +1,72 @@
|
|||||||
|
#
|
||||||
|
# Copyright (c) 2024-2026, Daily
|
||||||
|
#
|
||||||
|
# SPDX-License-Identifier: BSD 2-Clause License
|
||||||
|
#
|
||||||
|
|
||||||
|
"""Unit tests for SambaNova LLM service."""
|
||||||
|
|
||||||
|
import asyncio
|
||||||
|
from unittest.mock import AsyncMock, patch
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from pipecat.processors.aggregators.llm_context import LLMContext
|
||||||
|
from pipecat.services.sambanova.llm import SambaNovaLLMService
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_sambanova_llm_stream_closed_on_cancellation():
|
||||||
|
"""Test that the stream is closed when CancelledError occurs during iteration.
|
||||||
|
|
||||||
|
This prevents socket leaks when the pipeline is interrupted (e.g., user interruption).
|
||||||
|
See issue #3639.
|
||||||
|
"""
|
||||||
|
with patch.object(SambaNovaLLMService, "create_client"):
|
||||||
|
service = SambaNovaLLMService(api_key="test-key", model="test-model")
|
||||||
|
service._client = AsyncMock()
|
||||||
|
|
||||||
|
stream_closed = False
|
||||||
|
|
||||||
|
class MockAsyncStream:
|
||||||
|
"""Mock AsyncStream that tracks close() calls and raises CancelledError."""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.iteration_count = 0
|
||||||
|
|
||||||
|
async def __aenter__(self):
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __aexit__(self, exc_type, exc_val, exc_tb):
|
||||||
|
nonlocal stream_closed
|
||||||
|
stream_closed = True
|
||||||
|
return False
|
||||||
|
|
||||||
|
def __aiter__(self):
|
||||||
|
return self
|
||||||
|
|
||||||
|
async def __anext__(self):
|
||||||
|
self.iteration_count += 1
|
||||||
|
if self.iteration_count > 1:
|
||||||
|
raise asyncio.CancelledError()
|
||||||
|
mock_chunk = AsyncMock()
|
||||||
|
mock_chunk.usage = None
|
||||||
|
mock_chunk.choices = []
|
||||||
|
return mock_chunk
|
||||||
|
|
||||||
|
mock_stream = MockAsyncStream()
|
||||||
|
|
||||||
|
service._stream_chat_completions_specific_context = AsyncMock(return_value=mock_stream)
|
||||||
|
service._stream_chat_completions_universal_context = AsyncMock(return_value=mock_stream)
|
||||||
|
service.start_ttfb_metrics = AsyncMock()
|
||||||
|
service.stop_ttfb_metrics = AsyncMock()
|
||||||
|
service.start_llm_usage_metrics = AsyncMock()
|
||||||
|
|
||||||
|
context = LLMContext(
|
||||||
|
messages=[{"role": "user", "content": "Hello"}],
|
||||||
|
)
|
||||||
|
|
||||||
|
with pytest.raises(asyncio.CancelledError):
|
||||||
|
await service._process_context(context)
|
||||||
|
|
||||||
|
assert stream_closed, "Stream should be closed even when CancelledError occurs"
|
||||||
Reference in New Issue
Block a user