mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 18:13:03 +00:00
The backend import package is now docsgpt, the name it will carry on PyPI; application was far too generic to install into anyone's site-packages. git mv plus a mechanical rewrite of every import, dotted string and path reference: 734 Python files, the compose files, Dockerfile, workflows, docs, setup scripts, devcontainer, k8s manifests, vscode config, pytest and coverage config, .gitignore. Behaviour is unchanged. Kept for one release: - A top-level application package whose meta-path finder resolves application.x.y to the already-imported docsgpt.x.y object, so old imports and entry points (celery -A application.app.celery, uvicorn application.asgi:asgi_app) keep working with a FutureWarning. - Celery registers every application.* task name as an alias of its docsgpt.* task on start-up, so messages queued by the previous release still run. The redbeat key prefix moves to redbeat:docsgpt:v2: so schedule entries the previous release wrote are left unread instead of firing twice. The backend image builds from the repository root (docker build -f docsgpt/Dockerfile .) so it can ship the alias package; a root .dockerignore allow-lists docsgpt/ and application/ and keeps caches, local data, .env files, the sample index files and the Dockerfile out. Compose and the image workflows point at the new context.
303 lines
12 KiB
Python
303 lines
12 KiB
Python
"""A hallucinated tool call must be correctable and must not loop.
|
|
|
|
A first-session user's model invented ``note_view`` (a real tool in the repo,
|
|
but not one they had enabled) and called it 22 times in five and a half
|
|
minutes. Two defects turned one hallucination into 22 paid model calls: the
|
|
parse-failure branch returned no list of valid tools — unlike the sibling
|
|
tool-not-found branch, which does — and nothing noticed that the identical call
|
|
had already failed. The only bound was ``MAX_TOOL_ITERATIONS = 25`` per turn.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
from unittest.mock import Mock
|
|
|
|
import pytest
|
|
|
|
from docsgpt.agents.tool_executor import ToolExecutor
|
|
|
|
|
|
def _action(name):
|
|
return {"name": name, "description": "D", "active": True, "parameters": {"properties": {}}}
|
|
|
|
|
|
def _tools_dict():
|
|
# Rows carry actions, as every production row does: the model calls the
|
|
# ACTION name, so that is what an error may advertise.
|
|
return {
|
|
"t1": {"name": "memory", "actions": [_action("memory_view")], "config": {}},
|
|
"t2": {"name": "read_webpage", "actions": [_action("read_webpage")], "config": {}},
|
|
}
|
|
|
|
|
|
def _call(name, arguments="{}", call_id="c1"):
|
|
call = Mock()
|
|
call.name = name
|
|
call.arguments = arguments
|
|
call.id = call_id
|
|
return call
|
|
|
|
|
|
def _drain(executor, call, tools=None):
|
|
"""Run ``execute`` to completion and return the result string it produced.
|
|
|
|
The yielded status events are not asserted on anywhere in this module —
|
|
``executor.tool_calls`` records the same outcome — so they are dropped
|
|
rather than accumulated into a binding every call site would discard.
|
|
"""
|
|
gen = executor.execute(tools if tools is not None else _tools_dict(), call, "OpenAILLM")
|
|
while True:
|
|
try:
|
|
next(gen)
|
|
except StopIteration as stop:
|
|
result, _call_id = stop.value
|
|
return result
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestHallucinatedToolCalls:
|
|
def test_a_registered_name_with_bad_arguments_is_not_blamed_on_the_name(self):
|
|
"""Only the half that actually failed may be reported."""
|
|
executor = ToolExecutor()
|
|
tools = _tools_dict()
|
|
executor._name_to_tool = {"memory_view": ("t1", "memory_view")}
|
|
result = _drain(
|
|
executor, _call("memory_view", arguments="{not json"), tools=tools
|
|
)
|
|
assert "arguments were not a valid JSON object" in result, result
|
|
assert "the tool name could not be resolved" not in result, result
|
|
|
|
def test_parse_failure_tells_the_model_which_tools_exist(self):
|
|
executor = ToolExecutor()
|
|
# Unresolvable name AND unusable arguments: the branch under test.
|
|
result = _drain(executor, _call("bash", arguments="not json"))
|
|
|
|
assert executor.tool_calls[0]["status"] == "error"
|
|
reported = executor.tool_calls[0]["result"]
|
|
assert "memory" in reported and "read_webpage" in reported, reported
|
|
assert "memory" in result and "read_webpage" in result, result
|
|
|
|
def test_tool_not_found_still_lists_available_tools(self):
|
|
executor = ToolExecutor()
|
|
result = _drain(executor, _call("note_view"))
|
|
assert "memory" in result
|
|
|
|
def test_repeated_identical_failure_is_cut_short(self):
|
|
"""The third identical failing call must be refused without re-running."""
|
|
executor = ToolExecutor()
|
|
for index in range(3):
|
|
_drain(executor, _call("note_view", call_id=f"c{index}"))
|
|
|
|
assert len(executor.tool_calls) == 3
|
|
last = executor.tool_calls[-1]["result"]
|
|
assert "has already failed" in last, last
|
|
assert "Stop calling it" in last, last
|
|
|
|
def test_a_different_failing_call_is_not_suppressed(self):
|
|
executor = ToolExecutor()
|
|
for index in range(3):
|
|
_drain(executor, _call("note_view", call_id=f"c{index}"))
|
|
result = _drain(executor, _call("todo_view", call_id="other"))
|
|
assert "has already failed" not in result, result
|
|
assert "no such tool" in result, result
|
|
|
|
def test_the_guard_does_not_fire_on_the_first_two_attempts(self):
|
|
executor = ToolExecutor()
|
|
for index in range(2):
|
|
result = _drain(executor, _call("note_view", call_id=f"c{index}"))
|
|
assert "has already failed" not in result, result
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestErrorNamesWhatTheModelCanCall:
|
|
def test_prefers_llm_visible_action_names_over_tool_names(self):
|
|
"""The model calls action names, so those are what the error must list."""
|
|
executor = ToolExecutor()
|
|
tools_dict = {
|
|
"t1": {
|
|
"name": "artifact_generator",
|
|
"actions": [
|
|
{
|
|
"name": "create_artifact",
|
|
"description": "D",
|
|
"active": True,
|
|
"parameters": {"properties": {}},
|
|
}
|
|
],
|
|
}
|
|
}
|
|
executor.prepare_tools_for_llm(tools_dict)
|
|
result = _drain(executor, _call("make_a_pdf"), tools=tools_dict)
|
|
assert "create_artifact" in result
|
|
|
|
def test_the_fallback_advertises_action_names_not_tool_names(self):
|
|
"""With no name mapping built, the fallback must still name callables.
|
|
|
|
``_tool_to_name`` is empty whenever a turn produced zero LLM schemas —
|
|
every action toggled off, an unsynced MCP row, an ``api_tool`` with no
|
|
``config.actions``. Advertising ``code_executor`` there invites a call
|
|
named ``code_executor``, which cannot resolve: the error feeds the very
|
|
loop it exists to break.
|
|
"""
|
|
executor = ToolExecutor()
|
|
tools_dict = {
|
|
"t1": {"name": "code_executor", "actions": [_action("run_code")]},
|
|
"t2": {
|
|
"name": "api_tool",
|
|
"config": {"actions": {"a": _action("fetch_invoice")}},
|
|
},
|
|
}
|
|
assert executor._tool_to_name == {}
|
|
result = _drain(executor, _call("bash"), tools=tools_dict)
|
|
|
|
assert "run_code" in result and "fetch_invoice" in result, result
|
|
assert "code_executor" not in result, result
|
|
assert "api_tool" not in result, result
|
|
|
|
def test_the_fallback_skips_inactive_actions(self):
|
|
"""An action the user switched off is not callable, so it is not offered."""
|
|
executor = ToolExecutor()
|
|
off = _action("run_code")
|
|
off["active"] = False
|
|
tools_dict = {"t1": {"name": "code_executor", "actions": [off]}}
|
|
result = _drain(executor, _call("bash"), tools=tools_dict)
|
|
assert "(none available)" in result, result
|
|
|
|
def test_only_advertises_tools_in_scope_for_this_call(self):
|
|
"""A narrowed ``tools_dict`` must not be told about out-of-scope tools.
|
|
|
|
The error string is a tool RESULT handed straight back to the model, so
|
|
naming a tool it cannot call this round just buys another failed round.
|
|
"""
|
|
executor = ToolExecutor()
|
|
wide = {
|
|
f"t{n}": {
|
|
"name": f"server_{n}",
|
|
"actions": [
|
|
{
|
|
"name": f"action_{n}",
|
|
"description": "D",
|
|
"active": True,
|
|
"parameters": {"properties": {}},
|
|
}
|
|
],
|
|
}
|
|
for n in range(4)
|
|
}
|
|
executor.prepare_tools_for_llm(wide)
|
|
narrowed = {"t1": wide["t1"]}
|
|
result = _drain(
|
|
executor, _call("make_a_pdf"), tools=narrowed
|
|
)
|
|
assert "action_1" in result, result
|
|
for out_of_scope in ("action_0", "action_2", "action_3"):
|
|
assert out_of_scope not in result, (out_of_scope, result)
|
|
|
|
def test_the_advertised_list_is_capped(self):
|
|
"""This string joins the message history and is re-sent every round.
|
|
|
|
Uncapped, a large MCP fleet turns a single failed call into kilobytes
|
|
of prose duplicating the tool schema the provider already has.
|
|
"""
|
|
executor = ToolExecutor()
|
|
many = {
|
|
f"t{n}": {
|
|
"name": f"server_{n}",
|
|
"actions": [
|
|
{
|
|
"name": f"action_{n:03d}",
|
|
"description": "D",
|
|
"active": True,
|
|
"parameters": {"properties": {}},
|
|
}
|
|
],
|
|
}
|
|
for n in range(150)
|
|
}
|
|
executor.prepare_tools_for_llm(many)
|
|
result = _drain(executor, _call("make_a_pdf"), tools=many)
|
|
assert "and 120 more" in result, result
|
|
assert len(result) < 1000, len(result)
|
|
|
|
|
|
@pytest.mark.unit
|
|
class TestThrottleScope:
|
|
"""The throttle must fire on invented names only, and per distinct payload."""
|
|
|
|
def test_registered_tool_with_bad_arguments_is_never_refused(self):
|
|
"""Three malformed bodies for a real tool must not strand it for the turn.
|
|
|
|
Truncated ``code``/``spec`` payloads are the common shape here, and they
|
|
differ every time — collapsing them into one signature refused a working
|
|
tool and named it as its own alternative.
|
|
"""
|
|
executor = ToolExecutor()
|
|
tools = _tools_dict()
|
|
executor._name_to_tool = {"memory_view": ("t1", "memory_view")}
|
|
bodies = ['{"a": 1', '{"b": 2', '{"c": 3', '{"d": 4']
|
|
|
|
for index, body in enumerate(bodies):
|
|
result = _drain(
|
|
executor, _call("memory_view", arguments=body, call_id=f"c{index}"), tools=tools
|
|
)
|
|
assert "has already failed" not in result, (index, result)
|
|
assert "arguments were not a valid JSON object" in result, (index, result)
|
|
assert executor._unresolvable_calls == {}
|
|
|
|
def test_invented_name_is_refused_on_the_third_attempt(self):
|
|
executor = ToolExecutor()
|
|
for index in range(2):
|
|
result = _drain(
|
|
executor, _call("note_view", call_id=f"c{index}")
|
|
)
|
|
assert "has already failed" not in result, (index, result)
|
|
|
|
result = _drain(executor, _call("note_view", call_id="c2"))
|
|
assert "has already failed 2 times" in result, result
|
|
|
|
def test_the_refusal_does_not_suggest_the_tool_it_refuses(self):
|
|
"""``memory`` is a real tool; refusing it must not offer it as the way out."""
|
|
executor = ToolExecutor()
|
|
tools = _tools_dict()
|
|
executor._tool_to_name = {("t1", "memory_view"): "memory_view"}
|
|
for index in range(3):
|
|
result = _drain(
|
|
executor, _call("memory_view", call_id=f"c{index}"), tools=tools
|
|
)
|
|
assert "has already failed" in result, result
|
|
assert "(none available)" in result, result
|
|
|
|
def test_varying_payloads_for_an_invented_name_still_trip_the_guard(self):
|
|
"""Varying the arguments must not reset the count for an unknown name.
|
|
|
|
Told a call failed, a model's natural next move is to adjust its
|
|
arguments. Keying the counter on name+payload minted a fresh signature
|
|
every round, so the guard never fired and the turn ran to
|
|
``MAX_TOOL_ITERATIONS``. Arguments cannot rescue an unknown name:
|
|
``ToolActionParser`` resolves from ``call.name`` alone.
|
|
"""
|
|
executor = ToolExecutor()
|
|
results = []
|
|
for index, body in enumerate(['{"a": 1}', '{"b": 2}', '{"c": 3}']):
|
|
result = _drain(
|
|
executor, _call("note_view", arguments=body, call_id=f"c{index}")
|
|
)
|
|
results.append(result)
|
|
assert "has already failed" not in results[0], results[0]
|
|
assert "has already failed" not in results[1], results[1]
|
|
assert "has already failed 2 times" in results[2], results[2]
|
|
# One counter for the name, not one per payload.
|
|
assert list(executor._unresolvable_calls) == ["note_view"]
|
|
|
|
def test_the_failure_count_keeps_escalating(self):
|
|
"""A count frozen at the limit makes the message and the ops log useless."""
|
|
executor = ToolExecutor()
|
|
results = []
|
|
for index in range(5):
|
|
result = _drain(
|
|
executor, _call("note_view", call_id=f"c{index}")
|
|
)
|
|
results.append(result)
|
|
assert "has already failed 2 times" in results[2], results[2]
|
|
assert "has already failed 4 times" in results[4], results[4]
|