mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-04 10:13:06 +00:00
Correctness - /api/remote never recorded source.created, so URL, GitHub and connector sources had a source.deleted with no matching creation. All three creation paths now go through one _audit_source_created helper. - The prompt-cache rate divided cached tokens by a whole bucket's prompt tokens. A bucket is a day and mixes calls whose provider reports a cache breakdown with calls whose provider does not, so filtering buckets in the client could not separate them and the rate was understated by however much traffic ran on a non-reporting provider. The denominator is now computed in SQL over the reporting rows. - The outcome pill matched values nothing writes. Guardrails emit triggered / not_evaluated and the device feed emits dispatched; the map had blocked / denied / allowed, so a guardrail that fired rendered neutral grey -- the one signal the merged feed exists to surface. Fixtures were seeding the fictional values, so the tests passed on it too. - Stream duration_ms timed the consumer. stream_token_usage is a generator, so start-to-exhaustion includes the agent loop's tool handling and the SSE client's pace; a slow browser recorded ~30s for a sub-second call. It now accumulates only the time spent inside next(). Safety - Activity filters failed open: an unknown facet or unparseable timestamp was dropped, and no filter means every row, so a typo widened an audit view and on the export streamed the full history. Both are now a 400. - The search term was interpolated into an ILIKE pattern, so "100%" matched everything and "q1_report" matched more than it should. Escaped. - 0034 set actor_id NOT NULL with no default. A previous-release process inserting mid-rollout would raise, and in admin/routes.py that insert shares the request transaction, so a role grant beside it would roll back too. Noise and dead code - The per-user panel is a security panel: data-plane events file under the actor, so an active account's routine deletes pushed a denied login out of the 20-row window. It now excludes them; the Activity tab shows everything. - device_audit_log had no created_at-leading index, so the merged feed sequentially scanned that branch every page (migration 0036). - conversation.deleted_all no longer records when nothing was deleted, and agent.updated no longer records an empty field list. - Dropped by_model from /admin/usage (no consumer; an extra aggregate per page load), the duplicate filter surface on AuthEventsRepository that nothing called, and the unreachable FLOW_LABELS.schedule entry. - Type hints on record_event's conn and the remaining unannotated helpers.
284 lines
9.9 KiB
Python
284 lines
9.9 KiB
Python
"""Admin activity feed: one timeline over all three audit journals.
|
|
|
|
``GET /api/admin/audit`` remains the ``auth_events``-only feed. These endpoints
|
|
add the merged view an incident review actually wants — identity, access,
|
|
configuration, data-plane, remote-device and guardrail activity in one ordered
|
|
list — plus the filters and the export that make a review possible without
|
|
shelling into the database.
|
|
|
|
Every resource here is ``@admin_required``, like the rest of the namespace.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import csv
|
|
import io
|
|
import json
|
|
import logging
|
|
from datetime import datetime, timezone
|
|
from typing import Iterator, Optional
|
|
|
|
from flask import Response, jsonify, make_response, request, stream_with_context
|
|
from flask_restx import Resource
|
|
|
|
from docsgpt.api.admin.routes import _int_arg, _page, admin_ns
|
|
from docsgpt.api.user.authz import admin_required
|
|
from docsgpt.audit_events import ACTIVITY_CATEGORIES
|
|
from docsgpt.storage.db.repositories.activity import (
|
|
ACTIVITY_COLUMNS,
|
|
ActivityRepository,
|
|
)
|
|
from docsgpt.storage.db.session import db_readonly
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
_FEEDS = ("auth", "device", "guardrail")
|
|
_EXPORT_FORMATS = ("csv", "ndjson")
|
|
|
|
# An export runs against a live instance, so it is bounded. Operators needing
|
|
# the full history take a database dump, not an HTTP request.
|
|
_EXPORT_MAX_ROWS = 100_000
|
|
_EXPORT_CHUNK = 1_000
|
|
|
|
_MAX_SEARCH_LENGTH = 200
|
|
|
|
|
|
class _InvalidFilter(ValueError):
|
|
"""A filter argument the caller got wrong. Surfaces as a 400."""
|
|
|
|
|
|
def _facet_list(name: str, allowed: tuple[str, ...]) -> Optional[list[str]]:
|
|
"""Parse a repeated/comma-separated query arg into known facet values.
|
|
|
|
Rejects unknown values rather than dropping them. Dropping would leave the
|
|
request with no filter on that facet, and "no filter" means *everything* --
|
|
so a typo or a stale chip would silently widen an audit view instead of
|
|
narrowing it, and on the export would stream the whole history.
|
|
|
|
Args:
|
|
name: Query parameter name.
|
|
allowed: The permitted values.
|
|
|
|
Returns:
|
|
The distinct values in the order given, or None when absent.
|
|
|
|
Raises:
|
|
_InvalidFilter: If any supplied value is not in ``allowed``.
|
|
"""
|
|
raw: list[str] = []
|
|
for value in request.args.getlist(name):
|
|
raw.extend(part.strip() for part in value.split(",") if part.strip())
|
|
seen: list[str] = []
|
|
for value in raw:
|
|
if value not in allowed:
|
|
raise _InvalidFilter(
|
|
f"Unknown {name}: {value!r}. Expected one of {', '.join(allowed)}."
|
|
)
|
|
if value not in seen:
|
|
seen.append(value)
|
|
return seen or None
|
|
|
|
|
|
def _event_list() -> Optional[list[str]]:
|
|
"""Event names to filter on. Unconstrained — the catalogue is data, not an enum."""
|
|
names: list[str] = []
|
|
for value in request.args.getlist("event"):
|
|
for part in value.split(","):
|
|
part = part.strip()
|
|
if part and part not in names:
|
|
names.append(part)
|
|
return names or None
|
|
|
|
|
|
def _time_arg(name: str) -> Optional[datetime]:
|
|
"""Parse an ISO-8601 timestamp query arg, or None when absent.
|
|
|
|
Raises:
|
|
_InvalidFilter: If present but unparseable. Silently ignoring it would
|
|
drop the time bound and return the full history.
|
|
"""
|
|
raw = (request.args.get(name) or "").strip()
|
|
if not raw:
|
|
return None
|
|
try:
|
|
parsed = datetime.fromisoformat(raw.replace("Z", "+00:00"))
|
|
except ValueError as exc:
|
|
raise _InvalidFilter(
|
|
f"{name} must be an ISO-8601 timestamp, got {raw!r}."
|
|
) from exc
|
|
# A naive bound is read as UTC, matching how the feed stores timestamps.
|
|
return parsed if parsed.tzinfo else parsed.replace(tzinfo=timezone.utc)
|
|
|
|
|
|
def _search_arg() -> Optional[str]:
|
|
value = (request.args.get("search") or "").strip()
|
|
return value[:_MAX_SEARCH_LENGTH] or None
|
|
|
|
|
|
def _filters() -> dict:
|
|
"""The filter set shared by the feed and the export."""
|
|
return {
|
|
"feeds": _facet_list("feed", _FEEDS),
|
|
"categories": _facet_list("category", ACTIVITY_CATEGORIES),
|
|
"events": _event_list(),
|
|
"actor_id": request.args.get("actor_id") or None,
|
|
"user_id": request.args.get("user_id") or None,
|
|
"since": _time_arg("since"),
|
|
"until": _time_arg("until"),
|
|
"search": _search_arg(),
|
|
}
|
|
|
|
|
|
def _serialize(row: dict) -> dict:
|
|
"""Make one feed row JSON-safe (timestamps to ISO-8601)."""
|
|
created = row.get("created_at")
|
|
return {
|
|
**row,
|
|
"created_at": created.isoformat() if hasattr(created, "isoformat") else created,
|
|
}
|
|
|
|
|
|
@admin_ns.route("/admin/activity")
|
|
class AdminActivityResource(Resource):
|
|
@admin_required
|
|
def get(self):
|
|
"""Merged audit feed, newest first, with filters and a total."""
|
|
page, page_size, offset = _page()
|
|
try:
|
|
filters = _filters()
|
|
except _InvalidFilter as exc:
|
|
return make_response(jsonify({"success": False, "message": str(exc)}), 400)
|
|
with db_readonly() as conn:
|
|
repo = ActivityRepository(conn)
|
|
total = repo.count(**filters)
|
|
rows = repo.list(limit=page_size, offset=offset, **filters)
|
|
return make_response(
|
|
jsonify(
|
|
{
|
|
"success": True,
|
|
"activity": [_serialize(row) for row in rows],
|
|
"page": page,
|
|
"page_size": page_size,
|
|
"total": total,
|
|
"has_more": offset + len(rows) < total,
|
|
}
|
|
),
|
|
200,
|
|
)
|
|
|
|
|
|
@admin_ns.route("/admin/activity/events")
|
|
class AdminActivityCatalogueResource(Resource):
|
|
@admin_required
|
|
def get(self):
|
|
"""Distinct ``(event, category)`` pairs this instance has recorded.
|
|
|
|
The filter UI offers these instead of a free-text box, so an operator
|
|
never has to guess that denied logins are ``oidc_login_denied``.
|
|
"""
|
|
with db_readonly() as conn:
|
|
catalogue = ActivityRepository(conn).event_names()
|
|
return make_response(
|
|
jsonify(
|
|
{
|
|
"success": True,
|
|
"events": catalogue,
|
|
"categories": list(ACTIVITY_CATEGORIES),
|
|
"feeds": list(_FEEDS),
|
|
}
|
|
),
|
|
200,
|
|
)
|
|
|
|
|
|
@admin_ns.route("/admin/activity/export")
|
|
class AdminActivityExportResource(Resource):
|
|
@admin_required
|
|
def get(self):
|
|
"""Stream the filtered feed as CSV or NDJSON.
|
|
|
|
Streamed, not buffered: a compliance export of a busy instance should
|
|
not be built in memory first. Capped at ``_EXPORT_MAX_ROWS``, and the
|
|
``X-Export-Max-Rows`` header reports the cap that was in effect so a
|
|
truncated export is not mistaken for a complete one.
|
|
"""
|
|
fmt = (request.args.get("format") or "csv").lower()
|
|
if fmt not in _EXPORT_FORMATS:
|
|
return make_response(
|
|
jsonify({"success": False, "message": "Invalid format"}), 400
|
|
)
|
|
limit = max(1, min(_EXPORT_MAX_ROWS, _int_arg("max_rows", _EXPORT_MAX_ROWS)))
|
|
try:
|
|
filters = _filters()
|
|
except _InvalidFilter as exc:
|
|
return make_response(jsonify({"success": False, "message": str(exc)}), 400)
|
|
stamp = datetime.now(timezone.utc).strftime("%Y%m%d-%H%M%S")
|
|
generator = _csv_rows if fmt == "csv" else _ndjson_rows
|
|
return Response(
|
|
stream_with_context(generator(filters, limit)),
|
|
mimetype="text/csv" if fmt == "csv" else "application/x-ndjson",
|
|
headers={
|
|
"Content-Disposition": (
|
|
f'attachment; filename="docsgpt-activity-{stamp}.{fmt}"'
|
|
),
|
|
# The cap in effect. Without it a truncated export reads as
|
|
# a complete one.
|
|
"X-Export-Max-Rows": str(limit),
|
|
"Cache-Control": "no-store",
|
|
},
|
|
)
|
|
|
|
|
|
def _export_rows(filters: dict, limit: int) -> Iterator[dict]:
|
|
"""Walk the filtered feed, holding the connection open for the whole stream."""
|
|
with db_readonly() as conn:
|
|
yield from ActivityRepository(conn).iter_all(
|
|
chunk_size=_EXPORT_CHUNK, max_rows=limit, **filters
|
|
)
|
|
|
|
|
|
# Characters a spreadsheet reads as the start of a formula.
|
|
_FORMULA_TRIGGERS = ("=", "+", "-", "@", "\t", "\r")
|
|
|
|
|
|
def _csv_safe(value: object) -> object:
|
|
"""Neutralize a leading formula trigger before the value becomes a CSV cell.
|
|
|
|
The feed carries attacker-controlled text: ``user_agent`` is the raw header
|
|
of whoever made the request, and a denied login records one without ever
|
|
authenticating. An export is opened in a spreadsheet by an admin, so a cell
|
|
beginning ``=`` would be evaluated there. Prefixing an apostrophe keeps the
|
|
value readable and inert.
|
|
"""
|
|
if isinstance(value, str) and value.startswith(_FORMULA_TRIGGERS):
|
|
return f"'{value}"
|
|
return value
|
|
|
|
|
|
def _csv_rows(filters: dict, limit: int) -> Iterator[str]:
|
|
buffer = io.StringIO()
|
|
writer = csv.DictWriter(buffer, fieldnames=list(ACTIVITY_COLUMNS))
|
|
writer.writeheader()
|
|
yield _drain(buffer)
|
|
for row in _export_rows(filters, limit):
|
|
serialized = _serialize(row)
|
|
# ``detail`` is a JSON object; a CSV cell holds its compact encoding.
|
|
serialized["detail"] = json.dumps(
|
|
serialized.get("detail") or {}, default=str
|
|
)
|
|
writer.writerow({key: _csv_safe(value) for key, value in serialized.items()})
|
|
yield _drain(buffer)
|
|
|
|
|
|
def _ndjson_rows(filters: dict, limit: int) -> Iterator[str]:
|
|
for row in _export_rows(filters, limit):
|
|
yield json.dumps(_serialize(row), default=str) + "\n"
|
|
|
|
|
|
def _drain(buffer: io.StringIO) -> str:
|
|
"""Return and clear the writer's buffer, so each row streams as written."""
|
|
value = buffer.getvalue()
|
|
buffer.seek(0)
|
|
buffer.truncate(0)
|
|
return value
|