diff --git a/homeassistant/components/llama_cpp/entity.py b/homeassistant/components/llama_cpp/entity.py index 2c8686803bfe..5ffedeaa00c4 100644 --- a/homeassistant/components/llama_cpp/entity.py +++ b/homeassistant/components/llama_cpp/entity.py @@ -396,15 +396,17 @@ class LlamaCppBaseLLMEntity(Entity): cast(ChatCompletion, result).choices[0].message ) - messages.extend( - [ - msg - async for content in chat_log.async_add_delta_content_stream( - self.entity_id, async_generator - ) - if (msg := convert_message(content)) - ] - ) + # Streamed responses raise API errors while being consumed + with api_error_handler(): + messages.extend( + [ + msg + async for content in chat_log.async_add_delta_content_stream( + self.entity_id, async_generator + ) + if (msg := convert_message(content)) + ] + ) if not chat_log.unresponded_tool_results: break diff --git a/homeassistant/components/llama_cpp/quality_scale.yaml b/homeassistant/components/llama_cpp/quality_scale.yaml index 672cef1fdef1..bb12c57ef55e 100644 --- a/homeassistant/components/llama_cpp/quality_scale.yaml +++ b/homeassistant/components/llama_cpp/quality_scale.yaml @@ -34,9 +34,7 @@ rules: unique-config-entry: done # Silver - action-exceptions: - status: exempt - comment: No service actions are registered by this integration. + action-exceptions: done config-entry-unloading: done docs-configuration-parameters: done docs-installation-parameters: done diff --git a/tests/components/llama_cpp/test_conversation.py b/tests/components/llama_cpp/test_conversation.py index 3132ba38c9aa..8f0269a9d98d 100644 --- a/tests/components/llama_cpp/test_conversation.py +++ b/tests/components/llama_cpp/test_conversation.py @@ -600,6 +600,73 @@ async def test_streaming_response_redundant_role( assert content[1].content == "Hello world" +@pytest.mark.parametrize(("config_entry_options"), [{CONF_STREAMING: True}]) +@pytest.mark.parametrize( + ("error", "expected_speech"), + [ + pytest.param( + openai.APIConnectionError(request=httpx.Request(method="POST", url="test")), + "Cannot connect to the server: Connection error.", + id="connection_error", + ), + pytest.param( + openai.APITimeoutError(request=httpx.Request(method="POST", url="test")), + "Connection timed out: Request timed out.", + id="timeout", + ), + pytest.param( + openai.APIError( + message="An error occurred during streaming", + request=httpx.Request(method="POST", url="test"), + body=None, + ), + "API error: An error occurred during streaming", + id="api_error", + ), + ], +) +async def test_streaming_response_error( + hass: HomeAssistant, + mock_chat_log: MockChatLog, + mock_config_entry: MockConfigEntry, + error: openai.OpenAIError, + expected_speech: str, +) -> None: + """Test an API error raised while consuming the stream.""" + + async def mock_stream() -> AsyncGenerator[ChatCompletionChunk]: + yield ChatCompletionChunk.model_construct( + id="chatcmpl-1234567890ABCDEFGHIJKLMNOPQRS", + choices=[ + ChunkChoice.model_construct( + index=0, + delta=ChoiceDelta(role="assistant", content="Hello"), + finish_reason=None, + ) + ], + created=1700000000, + model="gpt-3.5-turbo-0613", + object="chat.completion.chunk", + ) + raise error + + with patch( + "openai.resources.chat.completions.AsyncCompletions.create", + new_callable=AsyncMock, + return_value=mock_stream(), + ): + result = await conversation.async_converse( + hass, + "hello", + mock_chat_log.conversation_id, + Context(), + agent_id="conversation.llama_cpp_conversation", + ) + + assert result.response.response_type == intent.IntentResponseType.ERROR + assert result.response.speech["plain"]["speech"] == expected_speech + + @pytest.mark.parametrize( ("config_entry_options"), [{CONF_LLM_HASS_API: ["non-existing"]}] )