From da1cff5008460ed9e954279db0d629a8d774716d Mon Sep 17 00:00:00 2001 From: Alex Date: Mon, 29 Jun 2026 13:11:51 +0200 Subject: [PATCH] feat: better code exec tool calling --- application/agents/tools/code_executor.py | 42 +++++++-- application/prompts/agentic/creative.txt | 2 +- application/prompts/agentic/default.txt | 2 +- application/prompts/agentic/strict.txt | 2 +- application/prompts/chat_combine_creative.txt | 2 +- application/prompts/chat_combine_default.txt | 2 +- application/prompts/chat_combine_strict.txt | 2 +- application/sandbox/artifacts_capture.py | 43 ++++++++- tests/sandbox/test_artifacts_capture.py | 93 +++++++++++++++++++ tests/test_code_executor_tool.py | 10 ++ 10 files changed, 184 insertions(+), 16 deletions(-) create mode 100644 tests/sandbox/test_artifacts_capture.py diff --git a/application/agents/tools/code_executor.py b/application/agents/tools/code_executor.py index 3520230e..aeebf51c 100644 --- a/application/agents/tools/code_executor.py +++ b/application/agents/tools/code_executor.py @@ -78,8 +78,9 @@ class CodeExecutorTool(Tool): "name": "run_code", "description": ( "Execute Python in a sandboxed, stateful session bound to this conversation. " - "Files written by the code are captured as downloadable artifacts; only a " - "compact summary (output tail + artifact references) is returned, never raw bytes. " + "Files written by the code are saved as downloadable artifacts (write throwaway " + "files under `tmp/`, or pass `outputs` to save only specific files); only a compact " + "summary (output tail + artifact references) is returned, never raw bytes. " "Each call is capped at ~60s of wall-clock; for longer work, start it in the " "background and poll with additional run_code calls (use persist=true to keep state)." ), @@ -100,6 +101,13 @@ class CodeExecutorTool(Tool): "ref like `A1` returned by a previous artifact action, a full artifact id, or " "the name/id of a file the user attached to this conversation.", }, + "outputs": { + "type": "array", + "items": {"type": "string"}, + "description": "Filenames or globs (e.g. `report.pdf`, `*.csv`) to save as " + "downloadable artifacts. When set, only matching files are saved; when omitted, " + "every produced file is saved except scratch paths under `tmp/`.", + }, "ttl": { "type": "integer", "description": "Keep-alive lifetime (seconds) for the session; clamped by SANDBOX_MAX_TTL.", @@ -114,7 +122,9 @@ class CodeExecutorTool(Tool): }, "capture_artifacts": { "type": "boolean", - "description": "Capture newly written workspace files as artifacts (default: true).", + "description": "Save produced workspace files as downloadable artifacts " + "(default: true). Set false for setup or install-only steps that write nothing " + "worth keeping.", }, }, "required": ["code"], @@ -161,6 +171,7 @@ class CodeExecutorTool(Tool): return {"status": "error", "error": "code is required."} should_capture = kwargs.get("capture_artifacts", True) + outputs = self._normalize_outputs(kwargs.get("outputs")) ttl = self._coerce_int(kwargs.get("ttl")) timeout = self._exec_timeout() inputs = kwargs.get("inputs") or [] @@ -192,7 +203,7 @@ class CodeExecutorTool(Tool): artifacts: List[Dict[str, Any]] = [] if should_capture: try: - artifacts = self._capture_artifacts(manager, session_id, pre_signatures) + artifacts = self._capture_artifacts(manager, session_id, pre_signatures, outputs) except Exception: logger.exception("code_executor: artifact capture failed") @@ -299,10 +310,28 @@ class CodeExecutorTool(Tool): """Map each non-input workspace file to a (size, sha256) signature for change detection.""" return snapshot_signatures(manager, session_id) + @staticmethod + def _normalize_outputs(raw: Any) -> Optional[List[str]]: + """Coerce the ``outputs`` arg to a list of non-empty glob strings, or None. + + Tolerates a bare string (some models pass one instead of an array); an empty + or non-list value means "no allow-list" (auto-capture). + """ + if isinstance(raw, str): + raw = [raw] + if not isinstance(raw, list): + return None + patterns = [str(p).strip() for p in raw if isinstance(p, str) and str(p).strip()] + return patterns or None + def _capture_artifacts( - self, manager: Any, session_id: str, pre_signatures: Dict[str, Tuple[int, Optional[str]]] + self, + manager: Any, + session_id: str, + pre_signatures: Dict[str, Tuple[int, Optional[str]]], + outputs: Optional[List[str]] = None, ) -> List[Dict[str, Any]]: - """Persist each non-input workspace file that is new or whose content changed.""" + """Persist produced workspace files (only ``outputs`` globs when given).""" captured = capture_artifacts( manager, session_id, @@ -315,6 +344,7 @@ class CodeExecutorTool(Tool): "action": "run_code", "session_id": session_id, }, + outputs=outputs, ) if captured: self._last_artifact_id = captured[0]["artifact_id"] diff --git a/application/prompts/agentic/creative.txt b/application/prompts/agentic/creative.txt index f9b33978..e88cec0f 100644 --- a/application/prompts/agentic/creative.txt +++ b/application/prompts/agentic/creative.txt @@ -12,7 +12,7 @@ You are DocsGPT, an AI assistant that answers questions using the user's documen ## Producing documents and running code - When the user wants a document, slide deck, spreadsheet, PDF, or other file and a document or artifact tool is available, create it as an artifact instead of pasting the full file into the chat. The user gets a downloadable, versioned file they can reopen and edit. - For follow-up changes to a file you already produced, edit that artifact with a targeted change rather than regenerating it from scratch, so its version history stays clean. -- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes come back as downloadable artifacts; do not paste their raw contents. Each run is time-limited, so start long work in the background and check on it with another run. +- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes are saved as downloadable artifacts, so write scratch or intermediate files under `tmp/` (not saved), pass `outputs` to save only specific files, or set `capture_artifacts` to false for setup-only steps; never paste raw file contents. Each run is time-limited, so start long work in the background and check on it with another run. {% endif %} ## Formatting diff --git a/application/prompts/agentic/default.txt b/application/prompts/agentic/default.txt index 21000df6..a25aafed 100644 --- a/application/prompts/agentic/default.txt +++ b/application/prompts/agentic/default.txt @@ -11,7 +11,7 @@ You are DocsGPT, an AI assistant that answers questions using the user's documen ## Producing documents and running code - When the user wants a document, slide deck, spreadsheet, PDF, or other file and a document or artifact tool is available, create it as an artifact instead of pasting the full file into the chat. The user gets a downloadable, versioned file they can reopen and edit. - For follow-up changes to a file you already produced, edit that artifact with a targeted change rather than regenerating it from scratch, so its version history stays clean. -- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes come back as downloadable artifacts; do not paste their raw contents. Each run is time-limited, so start long work in the background and check on it with another run. +- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes are saved as downloadable artifacts, so write scratch or intermediate files under `tmp/` (not saved), pass `outputs` to save only specific files, or set `capture_artifacts` to false for setup-only steps; never paste raw file contents. Each run is time-limited, so start long work in the background and check on it with another run. {% endif %} ## Formatting diff --git a/application/prompts/agentic/strict.txt b/application/prompts/agentic/strict.txt index 8636803c..c50a2718 100644 --- a/application/prompts/agentic/strict.txt +++ b/application/prompts/agentic/strict.txt @@ -11,7 +11,7 @@ You are DocsGPT, an AI assistant that answers questions using the user's documen ## Producing documents and running code - When the user wants a document, slide deck, spreadsheet, PDF, or other file and a document or artifact tool is available, create it as an artifact instead of pasting the full file into the chat. The user gets a downloadable, versioned file they can reopen and edit. - For follow-up changes to a file you already produced, edit that artifact with a targeted change rather than regenerating it from scratch, so its version history stays clean. -- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes come back as downloadable artifacts; do not paste their raw contents. Each run is time-limited, so start long work in the background and check on it with another run. +- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes are saved as downloadable artifacts, so write scratch or intermediate files under `tmp/` (not saved), pass `outputs` to save only specific files, or set `capture_artifacts` to false for setup-only steps; never paste raw file contents. Each run is time-limited, so start long work in the background and check on it with another run. {% endif %} ## Formatting diff --git a/application/prompts/chat_combine_creative.txt b/application/prompts/chat_combine_creative.txt index 95c57935..0f46fd27 100644 --- a/application/prompts/chat_combine_creative.txt +++ b/application/prompts/chat_combine_creative.txt @@ -11,7 +11,7 @@ You are DocsGPT, an AI assistant that answers questions using the user's documen ## Producing documents and running code - When the user wants a document, slide deck, spreadsheet, PDF, or other file and a document or artifact tool is available, create it as an artifact instead of pasting the full file into the chat. The user gets a downloadable, versioned file they can reopen and edit. - For follow-up changes to a file you already produced, edit that artifact with a targeted change rather than regenerating it from scratch, so its version history stays clean. -- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes come back as downloadable artifacts; do not paste their raw contents. Each run is time-limited, so start long work in the background and check on it with another run. +- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes are saved as downloadable artifacts, so write scratch or intermediate files under `tmp/` (not saved), pass `outputs` to save only specific files, or set `capture_artifacts` to false for setup-only steps; never paste raw file contents. Each run is time-limited, so start long work in the background and check on it with another run. {% endif %} ## Formatting diff --git a/application/prompts/chat_combine_default.txt b/application/prompts/chat_combine_default.txt index a10a9ba5..b7e5c7ba 100644 --- a/application/prompts/chat_combine_default.txt +++ b/application/prompts/chat_combine_default.txt @@ -10,7 +10,7 @@ You are DocsGPT, an AI assistant that answers questions using the user's documen ## Producing documents and running code - When the user wants a document, slide deck, spreadsheet, PDF, or other file and a document or artifact tool is available, create it as an artifact instead of pasting the full file into the chat. The user gets a downloadable, versioned file they can reopen and edit. - For follow-up changes to a file you already produced, edit that artifact with a targeted change rather than regenerating it from scratch, so its version history stays clean. -- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes come back as downloadable artifacts; do not paste their raw contents. Each run is time-limited, so start long work in the background and check on it with another run. +- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes are saved as downloadable artifacts, so write scratch or intermediate files under `tmp/` (not saved), pass `outputs` to save only specific files, or set `capture_artifacts` to false for setup-only steps; never paste raw file contents. Each run is time-limited, so start long work in the background and check on it with another run. {% endif %} ## Formatting diff --git a/application/prompts/chat_combine_strict.txt b/application/prompts/chat_combine_strict.txt index 750cd7a0..74d730f5 100644 --- a/application/prompts/chat_combine_strict.txt +++ b/application/prompts/chat_combine_strict.txt @@ -10,7 +10,7 @@ You are DocsGPT, an AI assistant that answers questions using the user's documen ## Producing documents and running code - When the user wants a document, slide deck, spreadsheet, PDF, or other file and a document or artifact tool is available, create it as an artifact instead of pasting the full file into the chat. The user gets a downloadable, versioned file they can reopen and edit. - For follow-up changes to a file you already produced, edit that artifact with a targeted change rather than regenerating it from scratch, so its version history stays clean. -- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes come back as downloadable artifacts; do not paste their raw contents. Each run is time-limited, so start long work in the background and check on it with another run. +- When a code-execution tool is available, run code for real computation, data processing, file parsing or conversion, and charts instead of estimating or writing results by hand. Files the code writes are saved as downloadable artifacts, so write scratch or intermediate files under `tmp/` (not saved), pass `outputs` to save only specific files, or set `capture_artifacts` to false for setup-only steps; never paste raw file contents. Each run is time-limited, so start long work in the background and check on it with another run. {% endif %} ## Formatting diff --git a/application/sandbox/artifacts_capture.py b/application/sandbox/artifacts_capture.py index 107b0ee1..e1c78b5a 100644 --- a/application/sandbox/artifacts_capture.py +++ b/application/sandbox/artifacts_capture.py @@ -8,6 +8,7 @@ here; binary bytes live in ``BaseStorage`` and never enter LLM context or workfl from __future__ import annotations +import fnmatch import hashlib import io import logging @@ -33,6 +34,13 @@ class QuotaExceeded(Exception): # one exec into an unbounded read+persist sweep. MAX_CAPTURED_FILES = 64 +# Auto-capture skips scratch/intermediate workspace paths so install steps, extracted +# archives, and temp files don't each become a downloadable artifact. Agents write +# throwaway files under ``tmp/``; an explicit ``outputs`` list bypasses this skip (the +# agent is then naming exactly what to keep). +_SCRATCH_PREFIXES = ("tmp/",) +_SCRATCH_SUFFIXES = (".tmp", ".lock", ".pyc", ".pyo") + _DEFAULT_KIND = "file" # Coarse mime -> artifact kind mapping for the UI rail; defaults to "file". @@ -64,8 +72,27 @@ def kind_for_mime(mime: str) -> str: return _DEFAULT_KIND +def _is_scratch(rel_path: str) -> bool: + """True for scratch/junk workspace paths excluded from auto-capture.""" + if rel_path == "tmp" or rel_path.startswith(_SCRATCH_PREFIXES): + return True + if any(part == "__pycache__" or part.startswith(".") for part in rel_path.split("/")): + return True + return rel_path.endswith(_SCRATCH_SUFFIXES) + + +def _matches_outputs(rel_path: str, outputs: List[str]) -> bool: + """True when a workspace path matches any caller-supplied ``outputs`` glob. + + Each pattern is matched against the full workspace-relative path and the bare + filename, so ``report.pdf``, ``*.csv``, and ``out/*.json`` all work. + """ + name = rel_path.rsplit("/", 1)[-1] + return any(fnmatch.fnmatch(rel_path, pat) or fnmatch.fnmatch(name, pat) for pat in outputs) + + def snapshot_signatures(manager: Any, session_id: str) -> Dict[str, Tuple[int, Optional[str]]]: - """Map each non-input workspace file to a (size, sha256) signature for change detection.""" + """Map each non-input, non-scratch workspace file to a (size, sha256) signature.""" signatures: Dict[str, Tuple[int, Optional[str]]] = {} try: files = manager.list_files(session_id) @@ -73,7 +100,7 @@ def snapshot_signatures(manager: Any, session_id: str) -> Dict[str, Tuple[int, O logger.exception("artifacts_capture: pre-exec listing failed") return signatures for rel_path in files: - if rel_path.startswith("inputs/"): + if rel_path.startswith("inputs/") or _is_scratch(rel_path): continue try: data = manager.get_file(session_id, rel_path) @@ -93,9 +120,13 @@ def capture_artifacts( conversation_id: Optional[str] = None, workflow_run_id: Optional[str] = None, produced_by: Optional[Dict[str, Any]] = None, + outputs: Optional[List[str]] = None, ) -> List[Dict[str, Any]]: - """Persist each non-input workspace file that is new or whose content changed. + """Persist workspace files that are new or whose content changed. + When ``outputs`` is given, capture only files matching those globs (the caller is + naming exactly what to keep). Otherwise auto-capture every produced file except + scratch/intermediate paths (``tmp/`` and obvious junk; see ``_is_scratch``). Returns one artifact reference per captured file: ``{artifact_id, version, filename, mime_type, size}`` (JSON primitives only; never bytes). """ @@ -105,7 +136,11 @@ def capture_artifacts( logger.exception("artifacts_capture: post-exec listing failed") return [] - candidates = sorted(f for f in post_files if not f.startswith("inputs/")) + produced = (f for f in post_files if not f.startswith("inputs/")) + if outputs: + candidates = sorted(f for f in produced if _matches_outputs(f, outputs)) + else: + candidates = sorted(f for f in produced if not _is_scratch(f)) captured: List[Dict[str, Any]] = [] for rel_path in candidates: if len(captured) >= MAX_CAPTURED_FILES: diff --git a/tests/sandbox/test_artifacts_capture.py b/tests/sandbox/test_artifacts_capture.py new file mode 100644 index 00000000..ef65f487 --- /dev/null +++ b/tests/sandbox/test_artifacts_capture.py @@ -0,0 +1,93 @@ +"""Unit tests for artifacts_capture scratch/outputs filtering.""" + +from __future__ import annotations + +import hashlib + +import pytest + +from application.sandbox import artifacts_capture as ac +from application.sandbox.artifacts_capture import _is_scratch, _matches_outputs + + +@pytest.mark.unit +class TestIsScratch: + @pytest.mark.parametrize( + "path", + ["tmp/x.csv", "tmp/sub/y.json", "__pycache__/m.pyc", "pkg/__pycache__/m.pyc", + ".cache/blob", ".ipynb_checkpoints/nb", "a.tmp", "b.lock", "c.pyc"], + ) + def test_scratch_paths_excluded(self, path): + assert _is_scratch(path) is True + + @pytest.mark.parametrize( + "path", ["report.pdf", "out/data.csv", "deck.pptx", "notes.txt", "tmpfile.txt"], + ) + def test_real_outputs_kept(self, path): + assert _is_scratch(path) is False + + +@pytest.mark.unit +class TestMatchesOutputs: + def test_basename_and_path(self): + assert _matches_outputs("report.pdf", ["report.pdf"]) + assert _matches_outputs("out/report.pdf", ["report.pdf"]) # basename also matches + + def test_globs(self): + assert _matches_outputs("a/b.csv", ["*.csv"]) + assert _matches_outputs("out/x.json", ["out/*.json"]) + + def test_no_match(self): + assert not _matches_outputs("report.pdf", ["*.csv"]) + + +class _FakeMgr: + """Serves a fixed {rel_path: bytes} workspace listing.""" + + def __init__(self, files): + self._files = files + + def list_files(self, _sid): + return list(self._files) + + def get_file(self, _sid, path): + return self._files[path] + + +@pytest.mark.unit +class TestCaptureFiltering: + @staticmethod + def _captured(monkeypatch, files, pre=None, outputs=None): + seen = [] + + def fake_persist(rel_path, data, **_kw): + seen.append(rel_path) + return {"artifact_id": rel_path, "version": 1, + "filename": rel_path.rsplit("/", 1)[-1], "mime_type": "x", "size": len(data)} + + monkeypatch.setattr(ac, "persist_artifact", fake_persist) + ac.capture_artifacts(_FakeMgr(files), "sid", pre or {}, user_id="u", outputs=outputs) + return seen + + def test_auto_skips_scratch(self, monkeypatch): + files = {"report.pdf": b"x", "tmp/scratch.csv": b"y", "__pycache__/m.pyc": b"z"} + assert self._captured(monkeypatch, files) == ["report.pdf"] + + def test_inputs_never_captured(self, monkeypatch): + files = {"report.pdf": b"x", "inputs/source.csv": b"y"} + assert self._captured(monkeypatch, files) == ["report.pdf"] + + def test_outputs_allow_list_only(self, monkeypatch): + files = {"report.pdf": b"x", "data.csv": b"y", "notes.txt": b"z"} + assert self._captured(monkeypatch, files, outputs=["report.pdf"]) == ["report.pdf"] + + def test_outputs_bypass_scratch(self, monkeypatch): + # An explicit pattern wins over the scratch skip. + files = {"tmp/keep.csv": b"x", "skip.txt": b"y"} + assert self._captured(monkeypatch, files, outputs=["*.csv"]) == ["tmp/keep.csv"] + + def test_unchanged_file_skipped(self, monkeypatch): + pre = {"report.pdf": (1, hashlib.sha256(b"x").hexdigest())} + assert self._captured(monkeypatch, {"report.pdf": b"x"}, pre=pre) == [] + # Content change is captured. + assert self._captured(monkeypatch, {"report.pdf": b"xy"}, pre=pre) == ["report.pdf"] diff --git a/tests/test_code_executor_tool.py b/tests/test_code_executor_tool.py index 7aabc475..7e7e71e0 100644 --- a/tests/test_code_executor_tool.py +++ b/tests/test_code_executor_tool.py @@ -190,6 +190,16 @@ def test_coerce_int_and_keep_alive(): assert CodeExecutorTool._keep_alive(False, None) is False +def test_normalize_outputs(): + n = CodeExecutorTool._normalize_outputs + assert n(["a.csv", "b.pdf"]) == ["a.csv", "b.pdf"] + assert n("report.pdf") == ["report.pdf"] # a bare string is tolerated + assert n([" x ", "", 5, "y"]) == ["x", "y"] # stripped; empties/non-strings dropped + assert n([]) is None + assert n(None) is None + assert n(" ") is None + + # --------------------------------------------------------------------------- # Action metadata / approval surface # ---------------------------------------------------------------------------