Files
DocsGPT/docsgpt/api/user/analytics/routes.py
T
2026-09-29 10:42:39 +04:00

1361 lines
58 KiB
Python

"""Analytics and reporting routes."""
import datetime
from typing import Optional, Tuple
from flask import current_app, jsonify, make_response, request
from flask_restx import fields, Namespace, Resource
from sqlalchemy import Connection, text as _sql_text
from docsgpt.api import api
from docsgpt.api.user.resource_access import AccessDenied, resolve
from docsgpt.api.user.base import (
generate_date_range,
generate_hourly_range,
generate_minute_range,
)
from docsgpt.storage.db.redaction import redact_secrets
from docsgpt.storage.db.repositories.agents import AgentsRepository
from docsgpt.storage.db.repositories.request_traces import (
REF_FIELDS as TRACE_REF_FIELDS,
RequestTracesRepository,
)
from docsgpt.storage.db.repositories.token_usage import TokenUsageRepository
from docsgpt.storage.db.session import db_readonly
analytics_ns = Namespace(
"analytics", description="Analytics and reporting operations", path="/api"
)
_FILTER_BUCKETS = {
"last_hour": ("minute", "%Y-%m-%d %H:%M:00", "YYYY-MM-DD HH24:MI:00"),
"last_24_hour": ("hour", "%Y-%m-%d %H:00", "YYYY-MM-DD HH24:00"),
"last_7_days": ("day", "%Y-%m-%d", "YYYY-MM-DD"),
"last_15_days": ("day", "%Y-%m-%d", "YYYY-MM-DD"),
"last_30_days": ("day", "%Y-%m-%d", "YYYY-MM-DD"),
}
def _range_for_filter(filter_option: str):
"""Return ``(start_date, end_date, bucket_unit, pg_fmt)`` for the filter.
Returns ``None`` on invalid filter.
"""
if filter_option not in _FILTER_BUCKETS:
return None
end_date = datetime.datetime.now(datetime.timezone.utc)
bucket_unit, _py_fmt, pg_fmt = _FILTER_BUCKETS[filter_option]
if filter_option == "last_hour":
start_date = end_date - datetime.timedelta(hours=1)
elif filter_option == "last_24_hour":
start_date = end_date - datetime.timedelta(hours=24)
else:
days = {
"last_7_days": 6,
"last_15_days": 14,
"last_30_days": 29,
}[filter_option]
start_date = end_date - datetime.timedelta(days=days)
start_date = start_date.replace(hour=0, minute=0, second=0, microsecond=0)
end_date = end_date.replace(
hour=23, minute=59, second=59, microsecond=999999
)
return start_date, end_date, bucket_unit, pg_fmt
def _intervals_for_filter(filter_option, start_date, end_date):
if filter_option == "last_hour":
return generate_minute_range(start_date, end_date)
if filter_option == "last_24_hour":
return generate_hourly_range(start_date, end_date)
return generate_date_range(start_date, end_date)
def _resolve_agent(conn, api_key_id, user_id):
"""Access-checked agent lookup for analytics filters.
Returns ``(agent, api_key, agent_pg_id)``. ``agent`` is ``None`` when
the id doesn't resolve to an agent the caller can see — callers must
short-circuit with an empty result, not fall back to sentinel filter
values. A visible agent needs ``view_logs`` (owner and editors; viewers
when the owner turns on ``viewers_can_see_logs``), and the caller then
sees exactly the owner's view of it. ``api_key`` is ``None`` (never
``""``) for key-less agents: draft agents store ``key = ''``, and an
``''`` filter would match the ``''`` that writers like ``stack_logs``
stamp on every key-less request — leaking rows across users. NULL
matches nothing. Accepts UUID or legacy Mongo ObjectId ids.
Raises:
AccessDenied: 403 when the agent is visible but ``view_logs`` isn't allowed.
"""
ra = resolve(conn, "agent", api_key_id, user_id) if api_key_id else None
if ra is None:
return None, None, None
if not ra.can("view_logs"):
raise AccessDenied(403, "Your access to this agent doesn't include its logs")
agent = AgentsRepository(conn).get_by_id(ra.resource_id)
api_key = (agent or {}).get("key") or None
agent_pg_id = str(agent["id"]) if agent else None
return agent, api_key, agent_pg_id
def _denied(err: AccessDenied):
"""The JSON error response for an :class:`AccessDenied`."""
return make_response(jsonify({"success": False, "message": err.message}), err.status)
def _trace_branch(name: str, sources_sql: str, scope: str) -> dict:
"""A ``get_user_logs`` branch listing stored traces of the given sources."""
return {
"name": name,
"level": "CASE WHEN t.status = 'error' THEN 'error' ELSE 'info' END",
# A search is listed by its query (copied into the small ``summary``
# at flush so this never detoasts ``spans``); graph builds are named
# after their source.
"summary": "COALESCE(t.summary->>'query', t.name, t.source)",
"where": [f"t.source IN {sources_sql}", scope],
"sql": f"""
SELECT '{name}' AS event_type,
CAST(t.id AS text) AS id,
t.user_id AS user_id,
t.started_at AS timestamp,
{{level}} AS level,
t.source AS action,
{{summary}} AS summary,
jsonb_build_object(
'status', t.status,
'source', t.source,
'duration_ms', t.duration_ms
) AS payload
FROM request_traces t
WHERE {{where}}
""",
}
def _trace_ref(item: dict) -> Optional[Tuple[str, str]]:
"""The ``(field, value)`` that finds a Logs row's trace, if it has one."""
event_type = item["event_type"]
row_id = item["id"].split("-", 1)[1]
if event_type == "chat":
return ("request_id", item.get("request_id")) if item.get("request_id") else None
if event_type in ("system", "webhook"):
return ("activity_id", item.get("activity_id")) if item.get("activity_id") else None
if event_type == "workflow":
return ("workflow_run_id", row_id)
if event_type == "schedule":
# The scheduler records the run under its run id.
return ("request_id", row_id)
if event_type in ("search", "graph"):
return ("id", row_id)
return None
def _merge_trace_summaries(traces: list) -> dict:
"""One Logs-row summary over every trace for it (a turn plus its resumes)."""
totals: dict = {}
for trace in traces:
for key, value in (trace.get("summary") or {}).items():
if isinstance(value, (int, float)) and not isinstance(value, bool):
totals[key] = totals.get(key, 0) + value
if "retrieval_ms" in totals:
totals["retrieval_ms"] = round(totals["retrieval_ms"], 1)
return {
"count": len(traces),
"duration_ms": sum(int(t.get("duration_ms") or 0) for t in traces),
"status": traces[-1].get("status"),
"started_at": traces[0].get("started_at"),
"summary": totals,
}
def _attach_trace_summaries(
conn: Connection, items: list, *, user_id: Optional[str], agent_id: Optional[str]
) -> None:
"""Add a ``trace`` summary to each Logs row that has a stored trace.
One batched lookup per link field for the whole page, rather than a join
inside the UNION, so the timeline query is unchanged. Rows without a
trace (older than the feature, or tracing disabled) get no key.
"""
refs: dict = {}
for item in items:
ref = _trace_ref(item)
if ref:
refs.setdefault(ref[0], set()).add(str(ref[1]))
if not refs:
return
found = RequestTracesRepository(conn).summaries_for_refs(
refs, user_id=user_id, agent_id=agent_id
)
for item in items:
ref = _trace_ref(item)
if not ref:
continue
traces = found.get(ref[0], {}).get(str(ref[1]))
if traces:
item["trace"] = {
"ref": {"field": ref[0], "value": str(ref[1])},
**_merge_trace_summaries(traces),
}
@analytics_ns.route("/traces")
class GetTraces(Resource):
@api.doc(
description=(
"Stored execution traces (span timelines) for one Logs row. Pass exactly "
"one of message_id, request_id, activity_id, workflow_run_id or id; "
"api_key_id scopes to an agent you own."
),
params={
"message_id": "Assistant message id",
"request_id": "Request id (chat turn, scheduled run)",
"activity_id": "Agent activity id (webhook and system rows)",
"workflow_run_id": "Workflow run id",
"id": "Trace id",
"api_key_id": "Agent id to scope to",
},
)
def get(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
refs = [
(field, request.args.get(field))
for field in TRACE_REF_FIELDS
if request.args.get(field)
]
if len(refs) != 1:
return make_response(
jsonify(
{
"success": False,
"message": "Pass exactly one of: " + ", ".join(TRACE_REF_FIELDS),
}
),
400,
)
field, value = refs[0]
api_key_id = request.args.get("api_key_id")
try:
with db_readonly() as conn:
agent, _api_key, agent_pg_id = _resolve_agent(conn, api_key_id, user)
if api_key_id and agent is None:
return make_response(jsonify({"success": True, "traces": []}), 200)
traces = RequestTracesRepository(conn).list_by_ref(
field, value, user_id=user, agent_id=agent_pg_id
)
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(f"Error getting traces: {err}", exc_info=True)
return make_response(jsonify({"success": False}), 400)
for trace in traces:
trace.pop("_id", None)
return make_response(jsonify({"success": True, "traces": traces}), 200)
@analytics_ns.route("/get_message_analytics")
class GetMessageAnalytics(Resource):
get_message_analytics_model = api.model(
"GetMessageAnalyticsModel",
{
"api_key_id": fields.String(required=False, description="API Key ID"),
"filter_option": fields.String(
required=False,
description="Filter option for analytics",
default="last_30_days",
enum=list(_FILTER_BUCKETS.keys()),
),
},
)
@api.expect(get_message_analytics_model)
@api.doc(description="Get message analytics based on filter option")
def post(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
data = request.get_json() or {}
api_key_id = data.get("api_key_id")
filter_option = data.get("filter_option", "last_30_days")
window = _range_for_filter(filter_option)
if window is None:
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
start_date, end_date, _bucket_unit, pg_fmt = window
try:
with db_readonly() as conn:
agent, api_key, agent_pg_id = _resolve_agent(
conn, api_key_id, user
)
if api_key_id and agent is None:
# Unknown / not-owned agent: empty result, not a
# sentinel filter (see _resolve_agent).
intervals = _intervals_for_filter(
filter_option, start_date, end_date
)
return make_response(
jsonify(
{
"success": True,
"messages": {i: 0 for i in intervals},
}
),
200,
)
# Count messages per bucket. When filtering by agent the
# owner-scoped lookup above already gates access, so the
# user clause is dropped (matching tokens / tools / logs):
# a shared agent's conversations carry the caller's
# user_id, and the owner should see that traffic on their
# own agent's dashboard. Agent matching covers both
# shapes: external traffic stamps ``api_key``, owner /
# shared chats stamp ``agent_id``.
clauses = [
"m.timestamp >= :start",
"m.timestamp <= :end",
]
params: dict = {
"start": start_date,
"end": end_date,
"fmt": pg_fmt,
}
if api_key_id:
clauses.append(
"(c.api_key = :api_key"
" OR c.agent_id = CAST(:agent_pg_id AS uuid))"
)
params["api_key"] = api_key
params["agent_pg_id"] = agent_pg_id
else:
clauses.append("c.user_id = :user_id")
params["user_id"] = user
where = " AND ".join(clauses)
sql = (
"SELECT to_char(m.timestamp AT TIME ZONE 'UTC', :fmt) AS bucket, "
"COUNT(*) AS count "
"FROM conversation_messages m "
"JOIN conversations c ON c.id = m.conversation_id "
f"WHERE {where} "
"GROUP BY bucket ORDER BY bucket ASC"
)
rows = conn.execute(_sql_text(sql), params).fetchall()
intervals = _intervals_for_filter(filter_option, start_date, end_date)
daily_messages = {interval: 0 for interval in intervals}
for row in rows:
daily_messages[row._mapping["bucket"]] = int(row._mapping["count"])
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(
f"Error getting message analytics: {err}", exc_info=True
)
return make_response(jsonify({"success": False}), 400)
return make_response(
jsonify({"success": True, "messages": daily_messages}), 200
)
@analytics_ns.route("/get_token_analytics")
class GetTokenAnalytics(Resource):
get_token_analytics_model = api.model(
"GetTokenAnalyticsModel",
{
"api_key_id": fields.String(required=False, description="API Key ID"),
"filter_option": fields.String(
required=False,
description="Filter option for analytics",
default="last_30_days",
enum=list(_FILTER_BUCKETS.keys()),
),
"group_by": fields.String(
required=False,
description="Second grouping dimension for the series",
default="none",
enum=["none", "model", "agent", "source"],
),
"include_side_channel": fields.Boolean(
required=False,
description=(
"Include non-user-initiated token usage (title "
"generation, compression, RAG condensing, fallback)"
),
default=True,
),
},
)
@api.expect(get_token_analytics_model)
@api.doc(description="Get token analytics data")
def post(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
data = request.get_json() or {}
api_key_id = data.get("api_key_id")
filter_option = data.get("filter_option", "last_30_days")
group_by = data.get("group_by") or "none"
# ``@api.expect`` documents but never validates/coerces — a JSON
# string like "false" must not truthy-coerce to True.
raw_side = data.get("include_side_channel", True)
if isinstance(raw_side, str):
include_side_channel = raw_side.strip().lower() not in (
"false",
"0",
"no",
)
else:
include_side_channel = bool(raw_side)
window = _range_for_filter(filter_option)
if window is None or group_by not in ("none", "model", "agent", "source"):
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
start_date, end_date, bucket_unit, _pg_fmt = window
try:
with db_readonly() as conn:
agent, api_key, agent_pg_id = _resolve_agent(
conn, api_key_id, user
)
if api_key_id and agent is None:
# Unknown / not-owned agent: empty result, not a
# sentinel filter (see _resolve_agent).
rows = []
else:
# The owner-scoped lookup gates access, so the
# user_id filter is dropped when agent-filtering
# (shared-agent rows carry the caller's user_id).
# The agent match is key-OR-id: chat stamps the
# key, headless runs stamp agent_id.
rows = TokenUsageRepository(conn).bucketed_totals(
bucket_unit=bucket_unit,
user_id=None if api_key_id else user,
api_key=api_key,
agent_id=agent_pg_id,
timestamp_gte=start_date,
timestamp_lt=end_date,
group_by=None if group_by == "none" else group_by,
include_side_channel=include_side_channel,
)
intervals = _intervals_for_filter(filter_option, start_date, end_date)
daily_token_usage = {interval: 0 for interval in intervals}
# ``series`` is the multi-dataset shape the dashboard renders
# as stacked bars: {series_key: {bucket: tokens}}. Without
# grouping the two series are the prompt/generated split.
series: dict = {}
if group_by == "none":
series = {
"prompt": {interval: 0 for interval in intervals},
"generated": {interval: 0 for interval in intervals},
}
for entry in rows:
bucket = entry["bucket"]
daily_token_usage[bucket] = int(
entry["prompt_tokens"] + entry["generated_tokens"]
)
series["prompt"][bucket] = int(entry["prompt_tokens"])
series["generated"][bucket] = int(entry["generated_tokens"])
else:
for entry in rows:
bucket = entry["bucket"]
total = int(entry["prompt_tokens"] + entry["generated_tokens"])
daily_token_usage[bucket] = (
daily_token_usage.get(bucket, 0) + total
)
key = entry["group_key"]
if key not in series:
series[key] = {interval: 0 for interval in intervals}
series[key][bucket] = series[key].get(bucket, 0) + total
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(
f"Error getting token analytics: {err}", exc_info=True
)
return make_response(jsonify({"success": False}), 400)
return make_response(
jsonify(
{
"success": True,
"token_usage": daily_token_usage,
"group_by": group_by,
"series": series,
}
),
200,
)
@analytics_ns.route("/get_feedback_analytics")
class GetFeedbackAnalytics(Resource):
get_feedback_analytics_model = api.model(
"GetFeedbackAnalyticsModel",
{
"api_key_id": fields.String(required=False, description="API Key ID"),
"filter_option": fields.String(
required=False,
description="Filter option for analytics",
default="last_30_days",
enum=list(_FILTER_BUCKETS.keys()),
),
},
)
@api.expect(get_feedback_analytics_model)
@api.doc(description="Get feedback analytics data")
def post(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
data = request.get_json() or {}
api_key_id = data.get("api_key_id")
filter_option = data.get("filter_option", "last_30_days")
window = _range_for_filter(filter_option)
if window is None:
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
start_date, end_date, _bucket_unit, pg_fmt = window
try:
with db_readonly() as conn:
agent, api_key, agent_pg_id = _resolve_agent(
conn, api_key_id, user
)
if api_key_id and agent is None:
intervals = _intervals_for_filter(
filter_option, start_date, end_date
)
return make_response(
jsonify(
{
"success": True,
"feedback": {
i: {"positive": 0, "negative": 0}
for i in intervals
},
}
),
200,
)
# Feedback lives inside the ``conversation_messages.feedback``
# JSONB as ``{"text": "like"|"dislike", "timestamp": "..."}``.
# There is no scalar ``feedback_timestamp`` column — extract
# the timestamp from the JSONB and cast it to timestamptz for
# the range filter + bucket grouping.
clauses = [
"m.feedback IS NOT NULL",
"(m.feedback->>'timestamp')::timestamptz >= :start",
"(m.feedback->>'timestamp')::timestamptz <= :end",
]
params: dict = {
"start": start_date,
"end": end_date,
"fmt": pg_fmt,
}
if api_key_id:
# Owner-gated agent match (see GetMessageAnalytics):
# drop the user clause so shared-agent feedback is
# visible to the owner, consistent with the other
# per-agent charts.
clauses.append(
"(c.api_key = :api_key"
" OR c.agent_id = CAST(:agent_pg_id AS uuid))"
)
params["api_key"] = api_key
params["agent_pg_id"] = agent_pg_id
else:
clauses.append("c.user_id = :user_id")
params["user_id"] = user
where = " AND ".join(clauses)
sql = (
"SELECT to_char("
"(m.feedback->>'timestamp')::timestamptz AT TIME ZONE 'UTC', :fmt"
") AS bucket, "
"SUM(CASE WHEN m.feedback->>'text' = 'like' THEN 1 ELSE 0 END) AS positive, "
"SUM(CASE WHEN m.feedback->>'text' = 'dislike' THEN 1 ELSE 0 END) AS negative "
"FROM conversation_messages m "
"JOIN conversations c ON c.id = m.conversation_id "
f"WHERE {where} "
"GROUP BY bucket ORDER BY bucket ASC"
)
rows = conn.execute(_sql_text(sql), params).fetchall()
intervals = _intervals_for_filter(filter_option, start_date, end_date)
daily_feedback = {
interval: {"positive": 0, "negative": 0} for interval in intervals
}
for row in rows:
bucket = row._mapping["bucket"]
daily_feedback[bucket] = {
"positive": int(row._mapping["positive"] or 0),
"negative": int(row._mapping["negative"] or 0),
}
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(
f"Error getting feedback analytics: {err}", exc_info=True
)
return make_response(jsonify({"success": False}), 400)
return make_response(
jsonify({"success": True, "feedback": daily_feedback}), 200
)
@analytics_ns.route("/get_tool_analytics")
class GetToolAnalytics(Resource):
get_tool_analytics_model = api.model(
"GetToolAnalyticsModel",
{
"api_key_id": fields.String(required=False, description="API Key ID"),
"filter_option": fields.String(
required=False,
description="Filter option for analytics",
default="last_30_days",
enum=list(_FILTER_BUCKETS.keys()),
),
},
)
@api.expect(get_tool_analytics_model)
@api.doc(description="Get tool call analytics from the tool execution journal")
def post(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
data = request.get_json() or {}
api_key_id = data.get("api_key_id")
filter_option = data.get("filter_option", "last_30_days")
window = _range_for_filter(filter_option)
if window is None:
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
start_date, end_date, _bucket_unit, _pg_fmt = window
try:
with db_readonly() as conn:
agent, api_key, agent_pg_id = _resolve_agent(
conn, api_key_id, user
)
if api_key_id and agent is None:
return make_response(
jsonify({"success": True, "tools": []}), 200
)
# Terminal rows only. ``proposed`` (pending) and
# ``executed`` (ran, not yet finalized) are non-terminal:
# counting them inflates ``calls`` and — since the client
# computes successful = calls - failures — renders them as
# phantom successes that later flip to failures when the
# reconciler escalates a stuck row. ``confirmed`` is the
# only success state; ``failed`` the only failure.
clauses = [
"t.status IN ('confirmed', 'failed')",
"t.attempted_at >= :start",
"t.attempted_at <= :end",
]
params: dict = {
"start": start_date,
"end": end_date,
}
join = (
"LEFT JOIN conversation_messages m ON m.id = t.message_id "
"LEFT JOIN conversations c ON c.id = m.conversation_id "
)
if api_key_id:
# Match by direct agent stamp (headless), the
# conversation's api_key (external chat), or the
# conversation's agent_id (owner chats / pre-0018
# rows). The owner-scoped lookup gates access, so
# no user clause — the owner also sees shared-agent
# traffic logged under callers' user_ids.
clauses.append(
"(t.agent_id = CAST(:agent_pg_id AS uuid)"
" OR c.api_key = :api_key"
" OR c.agent_id = CAST(:agent_pg_id AS uuid))"
)
params["agent_pg_id"] = agent_pg_id
params["api_key"] = api_key
else:
# ``t.user_id`` is stamped at propose time (0018);
# pre-migration rows fall back to the parent
# message's user (LEFT join — headless runs have no
# message). OR rather than COALESCE keeps the first
# arm index-sargable.
clauses.append(
"(t.user_id = :user_id OR m.user_id = :user_id)"
)
params["user_id"] = user
where = " AND ".join(clauses)
sql = (
"SELECT t.tool_name, "
"COUNT(*) AS calls, "
"COUNT(*) FILTER (WHERE t.status = 'failed') AS failures "
"FROM tool_call_attempts t "
f"{join}"
f"WHERE {where} "
"GROUP BY t.tool_name "
"ORDER BY calls DESC"
)
rows = conn.execute(_sql_text(sql), params).fetchall()
tools = [
{
"tool_name": row._mapping["tool_name"],
"calls": int(row._mapping["calls"]),
"failures": int(row._mapping["failures"]),
}
for row in rows
]
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(
f"Error getting tool analytics: {err}", exc_info=True
)
return make_response(jsonify({"success": False}), 400)
return make_response(jsonify({"success": True, "tools": tools}), 200)
@analytics_ns.route("/get_schedule_analytics")
class GetScheduleAnalytics(Resource):
get_schedule_analytics_model = api.model(
"GetScheduleAnalyticsModel",
{
"api_key_id": fields.String(required=False, description="API Key ID"),
"filter_option": fields.String(
required=False,
description="Filter option for analytics",
default="last_30_days",
enum=list(_FILTER_BUCKETS.keys()),
),
},
)
@api.expect(get_schedule_analytics_model)
@api.doc(description="Get scheduled agent run outcomes over time")
def post(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
data = request.get_json() or {}
api_key_id = data.get("api_key_id")
filter_option = data.get("filter_option", "last_30_days")
window = _range_for_filter(filter_option)
if window is None:
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
start_date, end_date, _bucket_unit, pg_fmt = window
try:
with db_readonly() as conn:
agent, _api_key, agent_pg_id = _resolve_agent(
conn, api_key_id, user
)
if api_key_id and agent is None:
intervals = _intervals_for_filter(
filter_option, start_date, end_date
)
return make_response(
jsonify(
{
"success": True,
"runs": {
i: {
"completed": 0,
"failed": 0,
"skipped": 0,
}
for i in intervals
},
}
),
200,
)
# A run's effective time is when it finished (fell back to
# started/scheduled for runs that never got that far).
ts = "COALESCE(r.finished_at, r.started_at, r.scheduled_for)"
clauses = [
f"{ts} >= :start",
f"{ts} <= :end",
]
params: dict = {
"start": start_date,
"end": end_date,
"fmt": pg_fmt,
}
if api_key_id:
# Owner-gated agent match: drop the user clause so a
# shared agent's runs (created by callers under their
# own user_id via the scheduler tool) are visible to
# the owner, consistent with the per-agent timeline.
clauses.append("r.agent_id = CAST(:agent_id AS uuid)")
params["agent_id"] = agent_pg_id
else:
clauses.append("r.user_id = :user_id")
params["user_id"] = user
where = " AND ".join(clauses)
# The worker writes ``success`` / ``failed`` / ``timeout`` /
# ``skipped`` (scheduler_worker.py); ``completed`` is kept in
# the success bucket defensively. Timeouts count as failures.
sql = (
f"SELECT to_char({ts} AT TIME ZONE 'UTC', :fmt) AS bucket, "
"COUNT(*) FILTER (WHERE r.status IN ('success', 'completed')) AS completed, "
"COUNT(*) FILTER (WHERE r.status IN ('failed', 'timeout')) AS failed, "
"COUNT(*) FILTER (WHERE r.status = 'skipped') AS skipped "
"FROM schedule_runs r "
f"WHERE {where} "
"GROUP BY bucket ORDER BY bucket ASC"
)
rows = conn.execute(_sql_text(sql), params).fetchall()
intervals = _intervals_for_filter(filter_option, start_date, end_date)
runs = {
interval: {"completed": 0, "failed": 0, "skipped": 0}
for interval in intervals
}
for row in rows:
runs[row._mapping["bucket"]] = {
"completed": int(row._mapping["completed"] or 0),
"failed": int(row._mapping["failed"] or 0),
"skipped": int(row._mapping["skipped"] or 0),
}
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(
f"Error getting schedule analytics: {err}", exc_info=True
)
return make_response(jsonify({"success": False}), 400)
return make_response(jsonify({"success": True, "runs": runs}), 200)
@analytics_ns.route("/get_user_logs")
class GetUserLogs(Resource):
get_user_logs_model = api.model(
"GetUserLogsModel",
{
"page": fields.Integer(
required=False,
description="Page number for pagination",
default=1,
),
"api_key_id": fields.String(required=False, description="API Key ID"),
"page_size": fields.Integer(
required=False,
description="Number of logs per page",
default=10,
),
"level": fields.String(
required=False,
description="Filter by log level",
enum=["info", "error", "warning"],
),
"event_type": fields.String(
required=False,
description="Filter by event source",
enum=[
"chat", "schedule", "webhook", "workflow", "system", "search", "graph",
],
),
"search": fields.String(
required=False, description="Substring filter on the summary"
),
},
)
@api.expect(get_user_logs_model)
@api.doc(
description=(
"Get user activity logs with pagination. Merges chat answers "
"(user_logs), request errors (stack_logs) and scheduled agent "
"runs (schedule_runs) into one timeline."
)
)
def post(self):
decoded_token = request.decoded_token
if not decoded_token:
return make_response(jsonify({"success": False}), 401)
user = decoded_token.get("sub")
data = request.get_json() or {}
try:
page = max(1, int(data.get("page") or 1))
page_size = max(1, min(100, int(data.get("page_size") or 10)))
except (TypeError, ValueError):
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
api_key_id = data.get("api_key_id")
level = data.get("level")
event_type = data.get("event_type")
search = data.get("search")
if level not in (None, "info", "error", "warning") or event_type not in (
None,
"chat",
"schedule",
"webhook",
"workflow",
"system",
"search",
"graph",
):
return make_response(
jsonify({"success": False, "message": "Invalid option"}), 400
)
try:
with db_readonly() as conn:
agent, api_key, agent_pg_id = _resolve_agent(
conn, api_key_id, user
)
if api_key_id and agent is None:
# Unknown / not-owned agent: empty page, not a
# sentinel filter (see _resolve_agent).
return make_response(
jsonify(
{
"success": True,
"logs": [],
"page": page,
"page_size": page_size,
"has_more": False,
}
),
200,
)
params: dict = {
# Agent-scoped logs are the owner's view of the agent.
"user_id": agent["user_id"] if agent else user,
"limit": page_size + 1,
"offset": (page - 1) * page_size,
}
# ``schedule`` / ``webhook`` errors are first-class events in
# their own branches; keep them out of ``system`` so a failed
# run doesn't appear twice. A failed webhook activity writes
# both an error row and an info row for the same activity_id
# (logging.py:_consume_and_log); NOT-EXISTS drops the info twin.
webhook_dedupe = (
"NOT (s.level = 'info' AND EXISTS ("
"SELECT 1 FROM stack_logs e "
"WHERE e.activity_id = s.activity_id "
"AND e.level = 'error'))"
)
# A failed chat turn now writes its own chat row (level
# ``error``) linked to its trace; the agent's error row for
# the same activity would list the failure twice. Rows from
# before execution traces, or with tracing off, have no such
# trace and still show here.
chat_failure_dedupe = (
"NOT EXISTS (SELECT 1 FROM request_traces t "
"WHERE t.activity_id = s.activity_id "
"AND t.source IN ('stream', 'answer', 'v1'))"
)
if api_key_id:
# The owner-scoped lookup gates access, so the
# chat/webhook/system branches match on the agent
# key/id alone — the owner also sees shared-agent
# traffic logged under callers' user_ids. The
# agent_id arm covers key-less (draft) agents,
# whose owner chats log a null api_key.
params["api_key"] = api_key
params["agent_pg_id"] = agent_pg_id
params["agent_workflow_id"] = (
str(agent["workflow_id"])
if agent and agent.get("workflow_id")
else None
)
chat_where = [
"(l.data->>'api_key' = :api_key"
" OR l.data->>'agent_id' = :agent_pg_id)"
]
# ``stack_logs.agent_id`` (0026) is the stable join key;
# ``api_key`` is mutable (agents rotate keys), so match the
# id first and keep the key arm for rows predating the
# backfill / any null agent_id.
stack_agent_match = (
"(s.agent_id = CAST(:agent_pg_id AS uuid)"
" OR s.api_key = :api_key)"
)
webhook_where = [
"COALESCE(s.endpoint, '') = 'webhook'",
stack_agent_match,
webhook_dedupe,
]
system_where = [
"s.level = 'error'",
"COALESCE(s.endpoint, '') NOT IN ('webhook', 'schedule')",
stack_agent_match,
chat_failure_dedupe,
]
# Owner-gated agent match: drop the user clause so a
# shared agent's runs (stamped with the caller's
# user_id by the scheduler tool) appear on the owner's
# per-agent timeline, consistent with the chat branch.
schedule_where = [
"r.status IN ('success', 'completed', 'failed', 'timeout', 'skipped')",
"r.agent_id = CAST(:agent_pg_id AS uuid)",
]
workflow_where = [
"wr.user_id = :user_id",
"wr.workflow_id = CAST(:agent_workflow_id AS uuid)",
]
trace_scope = "t.agent_id = CAST(:agent_pg_id AS uuid)"
else:
chat_where = ["l.user_id = :user_id"]
webhook_where = [
"s.user_id = :user_id",
"COALESCE(s.endpoint, '') = 'webhook'",
webhook_dedupe,
]
system_where = [
"s.user_id = :user_id",
"s.level = 'error'",
"COALESCE(s.endpoint, '') NOT IN ('webhook', 'schedule')",
chat_failure_dedupe,
]
# Terminal statuses only (worker writes ``success`` /
# ``failed`` / ``timeout`` / ``skipped``; ``completed``
# kept defensively). Pending/running runs aren't log
# entries yet.
schedule_where = [
"r.user_id = :user_id",
"r.status IN ('success', 'completed', 'failed', 'timeout', 'skipped')",
]
workflow_where = ["wr.user_id = :user_id"]
trace_scope = "t.user_id = :user_id"
# One normalized timeline over five event sources.
# ``payload`` carries the per-type detail; the outer query
# paginates the merged, time-ordered result. level /
# event_type / search are pushed into each branch so a
# filtered request only scans the branches it can match.
branches = [
{
"name": "chat",
"level": "COALESCE(l.data->>'level', 'info')",
"summary": "l.data->>'question'",
"where": chat_where,
"sql": """
SELECT 'chat' AS event_type,
CAST(l.id AS text) AS id,
l.user_id AS user_id,
l.timestamp AS timestamp,
{level} AS level,
COALESCE(l.data->>'action', 'stream_answer') AS action,
{summary} AS summary,
l.data AS payload
FROM user_logs l
WHERE {where}
""",
},
{
"name": "system",
"level": "'error'",
"summary": "s.query",
"where": system_where,
"sql": """
SELECT 'system' AS event_type,
CAST(s.id AS text) AS id,
s.user_id AS user_id,
s.timestamp AS timestamp,
{level} AS level,
COALESCE(s.endpoint, 'request') AS action,
{summary} AS summary,
jsonb_build_object(
'endpoint', s.endpoint,
'stacks', s.stacks,
'activity_id', s.activity_id
) AS payload
FROM stack_logs s
WHERE {where}
""",
},
{
"name": "webhook",
"level": "COALESCE(s.level, 'info')",
"summary": "s.query",
"where": webhook_where,
"sql": """
SELECT 'webhook' AS event_type,
CAST(s.id AS text) AS id,
s.user_id AS user_id,
s.timestamp AS timestamp,
{level} AS level,
'webhook_run' AS action,
{summary} AS summary,
jsonb_build_object(
'endpoint', s.endpoint,
'stacks', s.stacks,
'activity_id', s.activity_id
) AS payload
FROM stack_logs s
WHERE {where}
""",
},
{
"name": "workflow",
"level": (
"CASE WHEN wr.status = 'failed' "
"THEN 'error' ELSE 'info' END"
),
"summary": (
"COALESCE(wr.inputs->>'query', w.name, 'Workflow run')"
),
"where": workflow_where,
"sql": """
SELECT 'workflow' AS event_type,
CAST(wr.id AS text) AS id,
wr.user_id AS user_id,
COALESCE(wr.ended_at, wr.started_at) AS timestamp,
{level} AS level,
'workflow_run' AS action,
{summary} AS summary,
jsonb_build_object(
'status', wr.status,
'workflow_name', w.name,
'result', wr.result,
'steps', wr.steps,
'started_at', wr.started_at,
'finished_at', wr.ended_at
) AS payload
FROM workflow_runs wr
LEFT JOIN workflows w ON w.id = wr.workflow_id
WHERE {where}
""",
},
{
"name": "schedule",
"level": (
"CASE WHEN r.status IN ('failed', 'timeout') "
"THEN 'error' WHEN r.status = 'skipped' "
"THEN 'warning' ELSE 'info' END"
),
"summary": (
"COALESCE(sc.name, sc.instruction, 'Scheduled run')"
),
"where": schedule_where,
"sql": """
SELECT 'schedule' AS event_type,
CAST(r.id AS text) AS id,
r.user_id AS user_id,
COALESCE(r.finished_at, r.started_at, r.scheduled_for) AS timestamp,
{level} AS level,
'scheduled_run' AS action,
{summary} AS summary,
jsonb_build_object(
'status', r.status,
'trigger_source', r.trigger_source,
'schedule_name', sc.name,
'instruction', sc.instruction,
'output', r.output,
'error', r.error,
'error_type', r.error_type,
'prompt_tokens', r.prompt_tokens,
'generated_tokens', r.generated_tokens,
'conversation_id', r.conversation_id,
'scheduled_for', r.scheduled_for,
'started_at', r.started_at,
'finished_at', r.finished_at
) AS payload
FROM schedule_runs r
LEFT JOIN schedules sc ON sc.id = r.schedule_id
WHERE {where}
""",
},
# Runs with no log row of their own are listed from their
# stored trace: searches (/api/search and MCP) and graph
# builds.
_trace_branch("search", "('search', 'mcp')", trace_scope),
_trace_branch("graph", "('graph_extraction')", trace_scope),
]
if level:
params["level"] = level
if search:
escaped = (
search.replace("\\", "\\\\")
.replace("%", "\\%")
.replace("_", "\\_")
)
params["search"] = f"%{escaped}%"
branch_sqls = []
for branch in branches:
if event_type and branch["name"] != event_type:
continue
where = list(branch["where"])
if level:
where.append(f"{branch['level']} = :level")
if search:
where.append(
f"{branch['summary']} ILIKE :search ESCAPE '\\'"
)
branch_sqls.append(
branch["sql"].format(
level=branch["level"],
summary=branch["summary"],
where=" AND ".join(where),
)
)
# ``ev.id`` is a unique per-branch tiebreaker: equal
# timestamps (transaction-stable ``now()`` makes ties
# routine) would otherwise sort non-deterministically
# across page queries, duplicating a row on one page and
# dropping its sibling from the next under OFFSET paging.
sql = (
"SELECT * FROM ("
+ " UNION ALL ".join(branch_sqls)
+ ") ev ORDER BY ev.timestamp DESC, ev.id DESC"
" LIMIT :limit OFFSET :offset"
)
rows = conn.execute(_sql_text(sql), params).fetchall()
has_more = len(rows) > page_size
results = []
for row in rows[:page_size]:
m = row._mapping
payload = m["payload"] or {}
item = {
# Prefix with the source so ids stay unique across the
# merged tables (each has its own id sequence).
"id": f"{m['event_type']}-{m['id']}",
"event_type": m["event_type"],
"action": m["action"],
"level": m["level"],
"user": m["user_id"],
"question": m["summary"],
"timestamp": (
m["timestamp"].isoformat()
if hasattr(m["timestamp"], "isoformat")
else m["timestamp"]
),
}
if m["event_type"] == "chat":
item.update(
{
"response": payload.get("response"),
"sources": payload.get("sources"),
"tool_calls": payload.get("tool_calls"),
"agent_id": payload.get("agent_id"),
"attachments": payload.get("attachments"),
"request_id": payload.get("request_id"),
"message_id": payload.get("message_id"),
"error": payload.get("error"),
}
)
elif m["event_type"] in ("system", "webhook"):
item.update(
{
"endpoint": payload.get("endpoint"),
# Redact on the way out too: rows written
# before write-time redaction still carry the
# reflected provider/user secrets in ``stacks``.
"stacks": redact_secrets(payload.get("stacks")),
"activity_id": payload.get("activity_id"),
}
)
elif m["event_type"] in ("search", "graph"):
item.update(
{
"status": payload.get("status"),
"source": payload.get("source"),
"duration_ms": payload.get("duration_ms"),
}
)
elif m["event_type"] == "workflow":
item.update(
{
"status": payload.get("status"),
"workflow_name": payload.get("workflow_name"),
"result": payload.get("result"),
"steps": payload.get("steps"),
"started_at": payload.get("started_at"),
"finished_at": payload.get("finished_at"),
}
)
else: # schedule
item.update(
{
"status": payload.get("status"),
"trigger_source": payload.get("trigger_source"),
"schedule_name": payload.get("schedule_name"),
"instruction": payload.get("instruction"),
"output": payload.get("output"),
"error": payload.get("error"),
"error_type": payload.get("error_type"),
"prompt_tokens": payload.get("prompt_tokens"),
"generated_tokens": payload.get("generated_tokens"),
"conversation_id": payload.get("conversation_id"),
"scheduled_for": payload.get("scheduled_for"),
"started_at": payload.get("started_at"),
"finished_at": payload.get("finished_at"),
}
)
results.append(item)
if results:
# Trace chips are an extra: a failed lookup must leave the
# page intact, just without them.
try:
with db_readonly() as conn:
_attach_trace_summaries(
conn,
results,
user_id=user,
agent_id=agent_pg_id if api_key_id else None,
)
except Exception:
current_app.logger.warning(
"Could not attach trace summaries to the logs page",
exc_info=True,
)
except AccessDenied as denied:
return _denied(denied)
except Exception as err:
current_app.logger.error(
f"Error getting user logs: {err}", exc_info=True
)
return make_response(jsonify({"success": False}), 400)
return make_response(
jsonify(
{
"success": True,
"logs": results,
"page": page,
"page_size": page_size,
"has_more": has_more,
}
),
200,
)