fix(stream): persist a turn that errored without an answer as failed

WorkflowEngine reports node failures by *yielding* `{"type": "error"}`
rather than raising, so complete_stream's generator returns normally and
the except handler never runs. The turn was finalized `status="complete"`
with an empty response.

Live, the client renders an error bubble with a Retry button. On reload
it does not: mapServerQueryToClient only surfaces `metadata.error` for
`failed` rows, so history showed a blank message with no error and no way
to retry. A user hitting this re-sends the same prompt into new
conversations, which is exactly what the 2026-08-01 report shows — nine
blank first messages in seven hours.

Tracks a `stream_error` flag alongside the existing `paused` machinery
and finalizes `failed` when the turn produced no answer, recording the
user-facing message in `metadata.error`. An error arriving *after* output
keeps `complete` so partial text is not discarded; structured answers
count as output too, since they live in `structured_chunks` rather than
`response_full`. The flag is recorded before the pause branches so those
paths cannot lose it.

save_conversation grew a `status` parameter (default `complete`) for the
non-WAL branch, which took no status and so landed on the column default
— the same blank-complete row on a path the WAL fix did not cover.
Title generation now also runs for failed turns: _maybe_generate_title
only regenerates while the name is still the question-prefix fallback, so
skipping it would strand a conversation whose first turn failed with the
raw prompt as its name forever.

logging.py counts a yielded error toward `activity_finished.status`.
These failures previously logged `status=ok` with `answer_length=0`,
which is why user-visible blank answers never appeared in error metrics.
This commit is contained in:
Alex committed 2026-08-05 11:25:11 +01:00
1 parent f5129a5656
commit bda11242e1
6 files changed
+261 -5

No files matched your search

+164
View File
@@ -909,6 +909,15 @@ def _patch_db_session(conn):
# uncommitted writes from this transaction.
"application.api.answer.routes.base.db_readonly",
_yield,
), patch(
# The terminal ``stream_answer`` user_logs write opens its own
# ``db_session``. Left unpatched it is a *second* connection that
# blocks on the uncommitted ``users`` row this transaction just
# inserted (via the ``ensure_user_exists`` trigger) until the
# statement timeout fires — ~30s per test, swallowed by the
# caller's except, so it only ever showed up as slowness.
"application.api.answer.routes.base.db_session",
_yield,
):
yield
@@ -972,6 +981,161 @@ class TestCompleteStreamWalAcceptance:
assert "RuntimeError" in msgs[0]["metadata"]["error"]
assert "LLM upstream failed" in msgs[0]["metadata"]["error"]
def test_workflow_node_error_persists_as_failed_not_blank_complete(
self, pg_conn, flask_app,
):
"""A workflow node failure must not land as a blank ``complete`` row.
The engine reports node failures by *yielding* ``{"type": "error"}``
rather than raising, so the generator returns normally and the turn
used to be finalized ``complete`` with an empty response. Live, the
client shows an error bubble; on reload the row mapped to an empty
answer with no error and no retry affordance — the user saw a blank
message and re-sent the prompt. Reproduces the 2026-08-01 report.
"""
from application.api.answer.routes.base import BaseAnswerResource
from application.storage.db.repositories.conversations import (
ConversationsRepository,
)
with flask_app.app_context():
resource = BaseAnswerResource()
mock_agent = MagicMock()
mock_agent.gen.return_value = iter(
[{"type": "error", "error": "No LLM class found for type foundry"}]
)
with _patch_db_session(pg_conn):
stream = list(
resource.complete_stream(
question="hello",
agent=mock_agent,
conversation_id=None,
user_api_key=None,
decoded_token={"sub": "u-wf-error"},
should_persist=True,
model_id="gpt-4",
)
)
assert len([s for s in stream if '"type": "error"' in s]) == 1
from sqlalchemy import text as sql_text
convs = pg_conn.execute(
sql_text("SELECT id FROM conversations WHERE user_id = :u"),
{"u": "u-wf-error"},
).fetchall()
assert len(convs) == 1
msgs = ConversationsRepository(pg_conn).get_messages(str(convs[0][0]))
assert len(msgs) == 1
assert msgs[0]["prompt"] == "hello"
assert msgs[0]["status"] == "failed", (
"a turn that produced no answer and emitted an error must be "
"failed, so history renders the error and a retry button"
)
assert "foundry" in msgs[0]["metadata"]["error"]
def test_error_after_partial_answer_keeps_the_answer(
self, pg_conn, flask_app,
):
"""A late error must not discard text the user already received.
Only the *blank* turn is a failure. If a workflow produced output and
then a downstream node failed, the row stays ``complete`` so the
partial answer still renders; the error was already surfaced live.
"""
from application.api.answer.routes.base import BaseAnswerResource
from application.storage.db.repositories.conversations import (
ConversationsRepository,
)
with flask_app.app_context():
resource = BaseAnswerResource()
mock_agent = MagicMock()
mock_agent.gen.return_value = iter(
[
{"answer": "partial result"},
{"type": "error", "error": "node 2 blew up"},
]
)
with _patch_db_session(pg_conn):
list(
resource.complete_stream(
question="run the workflow",
agent=mock_agent,
conversation_id=None,
user_api_key=None,
decoded_token={"sub": "u-wf-partial"},
should_persist=True,
model_id="gpt-4",
)
)
from sqlalchemy import text as sql_text
convs = pg_conn.execute(
sql_text("SELECT id FROM conversations WHERE user_id = :u"),
{"u": "u-wf-partial"},
).fetchall()
msgs = ConversationsRepository(pg_conn).get_messages(str(convs[0][0]))
assert msgs[0]["status"] == "complete"
assert msgs[0]["response"] == "partial result"
# Still recorded, so the failure is greppable in review queries.
assert "node 2 blew up" in msgs[0]["metadata"]["error"]
def test_workflow_error_fails_the_row_on_the_non_wal_path_too(
self, pg_conn, flask_app,
):
"""The same guarantee when no placeholder row was reserved.
``save_conversation`` has its own insert path (used when the WAL
reservation failed, or on a continuation carrying no
``reserved_message_id``). It took no status, so the row landed on the
column default ``complete`` — leaving exactly the blank bubble this
changeset removes on the other branch.
"""
from application.api.answer.routes.base import BaseAnswerResource
from application.storage.db.repositories.conversations import (
ConversationsRepository,
)
with flask_app.app_context():
resource = BaseAnswerResource()
mock_agent = MagicMock()
mock_agent.gen.return_value = iter(
[{"type": "error", "error": "CEL error in node 'Build reply'"}]
)
with _patch_db_session(pg_conn), patch.object(
resource.conversation_service,
"save_user_question",
side_effect=RuntimeError("WAL reservation unavailable"),
):
list(
resource.complete_stream(
question="hello",
agent=mock_agent,
conversation_id=None,
user_api_key=None,
decoded_token={"sub": "u-wf-nonwal"},
should_persist=True,
model_id="gpt-4",
)
)
from sqlalchemy import text as sql_text
convs = pg_conn.execute(
sql_text("SELECT id FROM conversations WHERE user_id = :u"),
{"u": "u-wf-nonwal"},
).fetchall()
assert len(convs) == 1
msgs = ConversationsRepository(pg_conn).get_messages(str(convs[0][0]))
assert len(msgs) == 1
assert msgs[0]["status"] == "failed"
assert "CEL error" in msgs[0]["metadata"]["error"]
def test_tool_approval_event_only_fires_when_state_saved(
self, pg_conn, flask_app,
):