Handle mid-stream API errors in llama.cpp conversation (#184056)

Co-authored-by: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
epenet
2026-10-02 21:36:14 +01:00
committed by GitHub
co-authored by Claude Opus 5.5
parent 18f2c0da7d
commit 8ed5717950
3 changed files with 79 additions and 12 deletions
+11 -9
View File
@@ -396,15 +396,17 @@ class LlamaCppBaseLLMEntity(Entity):
cast(ChatCompletion, result).choices[0].message
)
messages.extend(
[
msg
async for content in chat_log.async_add_delta_content_stream(
self.entity_id, async_generator
)
if (msg := convert_message(content))
]
)
# Streamed responses raise API errors while being consumed
with api_error_handler():
messages.extend(
[
msg
async for content in chat_log.async_add_delta_content_stream(
self.entity_id, async_generator
)
if (msg := convert_message(content))
]
)
if not chat_log.unresponded_tool_results:
break
@@ -34,9 +34,7 @@ rules:
unique-config-entry: done
# Silver
action-exceptions:
status: exempt
comment: No service actions are registered by this integration.
action-exceptions: done
config-entry-unloading: done
docs-configuration-parameters: done
docs-installation-parameters: done
@@ -600,6 +600,73 @@ async def test_streaming_response_redundant_role(
assert content[1].content == "Hello world"
@pytest.mark.parametrize(("config_entry_options"), [{CONF_STREAMING: True}])
@pytest.mark.parametrize(
("error", "expected_speech"),
[
pytest.param(
openai.APIConnectionError(request=httpx.Request(method="POST", url="test")),
"Cannot connect to the server: Connection error.",
id="connection_error",
),
pytest.param(
openai.APITimeoutError(request=httpx.Request(method="POST", url="test")),
"Connection timed out: Request timed out.",
id="timeout",
),
pytest.param(
openai.APIError(
message="An error occurred during streaming",
request=httpx.Request(method="POST", url="test"),
body=None,
),
"API error: An error occurred during streaming",
id="api_error",
),
],
)
async def test_streaming_response_error(
hass: HomeAssistant,
mock_chat_log: MockChatLog,
mock_config_entry: MockConfigEntry,
error: openai.OpenAIError,
expected_speech: str,
) -> None:
"""Test an API error raised while consuming the stream."""
async def mock_stream() -> AsyncGenerator[ChatCompletionChunk]:
yield ChatCompletionChunk.model_construct(
id="chatcmpl-1234567890ABCDEFGHIJKLMNOPQRS",
choices=[
ChunkChoice.model_construct(
index=0,
delta=ChoiceDelta(role="assistant", content="Hello"),
finish_reason=None,
)
],
created=1700000000,
model="gpt-3.5-turbo-0613",
object="chat.completion.chunk",
)
raise error
with patch(
"openai.resources.chat.completions.AsyncCompletions.create",
new_callable=AsyncMock,
return_value=mock_stream(),
):
result = await conversation.async_converse(
hass,
"hello",
mock_chat_log.conversation_id,
Context(),
agent_id="conversation.llama_cpp_conversation",
)
assert result.response.response_type == intent.IntentResponseType.ERROR
assert result.response.speech["plain"]["speech"] == expected_speech
@pytest.mark.parametrize(
("config_entry_options"), [{CONF_LLM_HASS_API: ["non-existing"]}]
)