mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 08:13:02 +00:00
* fix: route remote-device tool through Redis so scheduled runs reach the device
The remote-device tool worked interactively but timed out on every scheduled
run. DeviceBroker was an in-process, in-memory singleton, but scheduled runs
execute in the Celery worker — a different process from the gunicorn web tier
that holds the device's SSE session — so a worker-side dispatch never reached
the device and the tool always hit its deadline.
Make the broker Redis-backed so every hop crosses the process boundary:
- queued commands -> Redis list dev:cmd:{device_id}
- output chunks -> Redis stream dev:out:{invocation_id}
- invocation metadata -> Redis hash dev:inv:{invocation_id}
- SSE upgrade tickets -> Redis key dev🎫{device_id}
Per-connection SSE session state stays in the web process. Reuses the existing
get_redis_instance()/CACHE_REDIS_URL; no new infrastructure. Also makes the web
tier safe to scale past one worker.
Concurrency hardening (from adversarial review + real-Redis e2e):
- XADD the output/control chunk before flipping completed=1, and have
drain_output do a final non-blocking flush after observing completion, so a
reader can't see completion and stop before the control chunk lands (this had
reintroduced the false "device did not respond (timed out)" under a race).
- _collect_result builds the result from drained chunks, checks the deadline
only after capturing a chunk, and falls back to the authoritative snapshot
(before cleanup) when no control chunk was observed.
- Audit outcome is written from locally-known fields so it survives the worker
racing to delete the invocation; a denied command now records a terminal
"denied" outcome instead of staying "dispatched".
- cmd-queue TTL raised to 900s (>= max drain deadline); dispatch-failure and
reaped-invocation cleanup; UTF-8 byte counts.
Tests: new tests/devices/{conftest (FakeRedis double), test_broker_cross_process,
test_broker_race, test_submit_output_audit}; drain/cleanup/ticket tests rewritten
for the Redis contract. The race tests fail against the pre-fix code. ruff clean;
device + tool-executor suites green.
* fix: log instead of silently passing on failed-dispatch cleanup
Addresses the code-quality lint on the best-effort hash delete in
dispatch_invocation's failure path: replace the bare `except: pass` with a
logger.debug carrying the invocation_id. No behavior change — cleanup stays
best-effort and still returns a failed Invocation.
84 lines
3.2 KiB
Python
84 lines
3.2 KiB
Python
"""Regression: a command dispatched in one process reaches a session in another.
|
|
|
|
This reproduces the scheduled-run failure. The agent runs inside a Celery
|
|
worker (``worker_broker``) while the device's SSE session lives in the web
|
|
process (``web_broker``). With the old in-process broker these two never
|
|
shared state, so every scheduled invocation timed out. Backed by one Redis
|
|
(here a shared ``FakeRedis``), dispatch and drain cross the process line.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import time
|
|
|
|
from application.devices.broker import DeviceBroker
|
|
|
|
|
|
def test_dispatch_in_worker_reaches_session_in_web(monkeypatch, fake_redis):
|
|
monkeypatch.setattr(
|
|
"application.devices.broker.get_redis_instance", lambda: fake_redis
|
|
)
|
|
worker_broker = DeviceBroker() # e.g. Celery scheduled run
|
|
web_broker = DeviceBroker() # e.g. gunicorn web tier holding the SSE socket
|
|
|
|
# 1) Agent (worker process) dispatches a command.
|
|
envelope = {
|
|
"invocation_id": "inv_xproc",
|
|
"action": "run_command",
|
|
"params": {"command": "echo hi"},
|
|
}
|
|
worker_broker.dispatch_invocation("dev_1", "user_1", envelope)
|
|
|
|
# 2) Device polls/upgrades on the web process and receives the command.
|
|
assert web_broker.claim_ticket("dev_1", 30) is not None
|
|
sess = web_broker.register_session("dev_1", "user_1")
|
|
delivered = web_broker.next_command(sess, timeout=0.1)
|
|
assert delivered is not None
|
|
assert delivered["invocation_id"] == "inv_xproc"
|
|
assert delivered["params"]["command"] == "echo hi"
|
|
|
|
# 3) Device streams output back through the web process.
|
|
web_broker.submit_output_chunk("inv_xproc", {"stream": "stdout", "chunk": "hi\n"})
|
|
web_broker.submit_output_chunk(
|
|
"inv_xproc",
|
|
{"stream": "control", "exit_code": 0, "duration_ms": 7},
|
|
)
|
|
|
|
# 4) The worker's drain (what the tool does) sees the full result.
|
|
deadline = time.time() + 2.0
|
|
chunks = list(
|
|
worker_broker.drain_output("inv_xproc", timeout=0.05, deadline=deadline)
|
|
)
|
|
stdout = "".join(c.get("chunk", "") for c in chunks if c.get("stream") == "stdout")
|
|
control = [c for c in chunks if c.get("stream") == "control"]
|
|
assert stdout == "hi\n"
|
|
assert control and control[0]["exit_code"] == 0
|
|
|
|
# And the metadata snapshot reflects completion across instances.
|
|
final = worker_broker.get_invocation("inv_xproc")
|
|
assert final is not None
|
|
assert final.completed is True
|
|
assert final.exit_code == 0
|
|
assert final.stdout_bytes == len("hi\n")
|
|
|
|
|
|
def test_denied_ack_unblocks_drain(monkeypatch, fake_redis):
|
|
# A denial on the web side must promptly stop a worker-side drain.
|
|
monkeypatch.setattr(
|
|
"application.devices.broker.get_redis_instance", lambda: fake_redis
|
|
)
|
|
worker_broker = DeviceBroker()
|
|
web_broker = DeviceBroker()
|
|
|
|
worker_broker.dispatch_invocation(
|
|
"dev_2", "user_2", {"invocation_id": "inv_deny", "action": "run_command"}
|
|
)
|
|
web_broker.submit_ack("inv_deny", "denied", reason="user rejected")
|
|
|
|
deadline = time.time() + 2.0
|
|
chunks = list(
|
|
worker_broker.drain_output("inv_deny", timeout=0.05, deadline=deadline)
|
|
)
|
|
assert chunks and chunks[-1]["stream"] == "control"
|
|
assert chunks[-1]["error"] == "denied"
|