mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-06 08:13:39 +00:00
fix(stream): persist a turn that errored without an answer as failed
WorkflowEngine reports node failures by *yielding* `{"type": "error"}`
rather than raising, so complete_stream's generator returns normally and
the except handler never runs. The turn was finalized `status="complete"`
with an empty response.
Live, the client renders an error bubble with a Retry button. On reload
it does not: mapServerQueryToClient only surfaces `metadata.error` for
`failed` rows, so history showed a blank message with no error and no way
to retry. A user hitting this re-sends the same prompt into new
conversations, which is exactly what the 2026-08-01 report shows — nine
blank first messages in seven hours.
Tracks a `stream_error` flag alongside the existing `paused` machinery
and finalizes `failed` when the turn produced no answer, recording the
user-facing message in `metadata.error`. An error arriving *after* output
keeps `complete` so partial text is not discarded; structured answers
count as output too, since they live in `structured_chunks` rather than
`response_full`. The flag is recorded before the pause branches so those
paths cannot lose it.
save_conversation grew a `status` parameter (default `complete`) for the
non-WAL branch, which took no status and so landed on the column default
— the same blank-complete row on a path the WAL fix did not cover.
Title generation now also runs for failed turns: _maybe_generate_title
only regenerates while the name is still the question-prefix fallback, so
skipping it would strand a conversation whose first turn failed with the
raw prompt as its name forever.
logging.py counts a yielded error toward `activity_finished.status`.
These failures previously logged `status=ok` with `answer_length=0`,
which is why user-visible blank answers never appeared in error metrics.
This commit is contained in:
1 parent
f5129a5656
commit
bda11242e1
6 files changed
+261
-5
No files matched your search
@@ -246,6 +246,14 @@ class BaseAnswerResource:
|
||||
structured_chunks = []
|
||||
query_metadata: Dict[str, Any] = {}
|
||||
paused = False
|
||||
# Set when the agent *yields* a terminal ``error`` event instead of
|
||||
# raising. Workflow node failures take that route (the engine catches
|
||||
# the node exception and reports it as an event), so the generator
|
||||
# returns normally and the ``except`` handler below never runs. Without
|
||||
# this flag the turn was finalized ``complete`` with an empty response:
|
||||
# the live client showed an error bubble, but on reload history mapped
|
||||
# the row to a blank answer with no error text and no retry.
|
||||
stream_error: Optional[str] = None
|
||||
# A ``tool_calls_pending`` event is held back and only flushed after
|
||||
# continuation state is committed (or the stateless finalize path is
|
||||
# reached): the v1 translator turns it into ``finish_reason:"tool_calls"``,
|
||||
@@ -568,6 +576,7 @@ class BaseAnswerResource:
|
||||
error_text = line.get("error", "An error occurred")
|
||||
if not line.get("user_facing"):
|
||||
error_text = sanitize_api_error(error_text)
|
||||
stream_error = error_text
|
||||
yield _emit({"type": "error", "error": error_text})
|
||||
elif line.get("type") == "notice":
|
||||
# Non-fatal, non-terminal notice (e.g. some workflow input
|
||||
@@ -595,6 +604,14 @@ class BaseAnswerResource:
|
||||
}
|
||||
)
|
||||
|
||||
# Record a yielded error before any early return so the pause /
|
||||
# stateless-tool-round paths persist it too. No producer currently
|
||||
# emits a non-terminal error and then pauses, but leaving the only
|
||||
# write below the pause blocks would make that combination lose the
|
||||
# error silently — the exact shape of the bug being fixed here.
|
||||
if stream_error:
|
||||
query_metadata.setdefault("error", stream_error)
|
||||
|
||||
# ---- Paused: save continuation state and end stream early ----
|
||||
if paused:
|
||||
continuation = getattr(agent, "_pending_continuation", None)
|
||||
@@ -848,6 +865,19 @@ class BaseAnswerResource:
|
||||
)
|
||||
llm._token_usage_source = "title"
|
||||
|
||||
# The error was recorded above so the failure stays greppable, but
|
||||
# it only *fails* the turn when nothing was produced. An error
|
||||
# arriving after partial output (e.g. a later workflow node) must
|
||||
# stay ``complete``, since the client only renders ``response`` for
|
||||
# complete rows — failing it would discard text the user already
|
||||
# saw. ``structured_chunks`` counts as output for the same reason:
|
||||
# a structured answer lives there, not in ``response_full``.
|
||||
errored_empty = (
|
||||
bool(stream_error)
|
||||
and not response_full.strip()
|
||||
and not structured_chunks
|
||||
)
|
||||
|
||||
if should_persist:
|
||||
if reserved_message_id is not None:
|
||||
self.conversation_service.finalize_message(
|
||||
@@ -858,7 +888,7 @@ class BaseAnswerResource:
|
||||
tool_calls=tool_calls,
|
||||
model_id=model_id or self.default_model_id,
|
||||
metadata=query_metadata if query_metadata else None,
|
||||
status="complete",
|
||||
status="failed" if errored_empty else "complete",
|
||||
title_inputs={
|
||||
"llm": llm,
|
||||
"question": question,
|
||||
@@ -889,6 +919,7 @@ class BaseAnswerResource:
|
||||
attachment_ids=attachment_ids,
|
||||
metadata=query_metadata if query_metadata else None,
|
||||
visibility=visibility,
|
||||
status="failed" if errored_empty else "complete",
|
||||
)
|
||||
# Persist compression metadata/summary if it exists and wasn't saved mid-execution
|
||||
compression_meta = getattr(agent, "compression_metadata", None)
|
||||
|
||||
@@ -100,9 +100,15 @@ class ConversationService:
|
||||
attachment_ids: Optional[List[str]] = None,
|
||||
metadata: Optional[Dict[str, Any]] = None,
|
||||
visibility: str = "hidden",
|
||||
status: str = "complete",
|
||||
) -> str:
|
||||
"""Save or update a conversation in Postgres.
|
||||
|
||||
``status`` lets a caller record a turn that failed without producing
|
||||
an answer. It defaults to ``complete``, matching the column default
|
||||
this path relied on before; passing ``failed`` is what stops a blank
|
||||
errored turn from rendering as an empty bubble with no retry.
|
||||
|
||||
Returns the string conversation id (PG UUID as string, or the
|
||||
caller-provided id if it was already a UUID).
|
||||
"""
|
||||
@@ -127,6 +133,7 @@ class ConversationService:
|
||||
"attachments": attachment_ids,
|
||||
"model_id": model_id,
|
||||
"timestamp": current_time,
|
||||
"status": status,
|
||||
}
|
||||
if metadata:
|
||||
message_payload["metadata"] = metadata
|
||||
@@ -410,7 +417,12 @@ class ConversationService:
|
||||
repo.confirm_executed_tool_calls(message_id)
|
||||
|
||||
# Outside the txn — title-gen is a multi-second LLM round trip.
|
||||
if title_inputs and status == "complete":
|
||||
# ``failed`` counts too: the conversation is still listed, and
|
||||
# ``_maybe_generate_title`` only regenerates while the name is still
|
||||
# the question-prefix fallback. Skipping it here would strand a
|
||||
# conversation whose first turn failed with the raw prompt as its
|
||||
# name forever, because by turn two the fallback no longer matches.
|
||||
if title_inputs and status in ("complete", "failed"):
|
||||
if async_title_generation:
|
||||
threading.Thread(
|
||||
target=self._generate_title_safely,
|
||||
|
||||
Reference in new issue
Block a user