From 6db9014dfb146b37f65dcaafcedba0ce74ed8ce5 Mon Sep 17 00:00:00 2001 From: Alex Date: Mon, 21 Sep 2026 12:14:14 +0100 Subject: [PATCH] fix(quotas): list unpriced models by recorded cost; integer token limits in status The unpriced-model notice asked the live registry whether a model has a price, so a priced model whose provider was later disabled showed up as unpriced. It now lists models whose calls this period were all recorded at $0. Token limits are serialized as integers. --- docsgpt/api/admin/quotas.py | 8 +++++--- docsgpt/quotas/service.py | 13 +++++++------ tests/api/test_quota_endpoints.py | 13 ++++++++----- tests/quotas/test_service.py | 2 +- 4 files changed, 21 insertions(+), 15 deletions(-) diff --git a/docsgpt/api/admin/quotas.py b/docsgpt/api/admin/quotas.py index 8210ba8d..58b49387 100644 --- a/docsgpt/api/admin/quotas.py +++ b/docsgpt/api/admin/quotas.py @@ -151,13 +151,15 @@ def _delete_policy(scope: str, subject_id: Optional[str]): def _unpriced_models(conn) -> list[dict]: - """Catalog models used this period that no cost limit can see.""" + """Models used this period whose calls were all recorded at $0 for want of a price.""" start, _ = window_bounds(settings.QUOTA_PERIOD) return [ row for row in TokenUsageRepository(conn).tokens_by_model(start=start) - # BYOM ids are UUIDs; those calls are $0 by design, not by omission. - if not looks_like_uuid(row["model_id"]) and not is_priced(row["model_id"]) + # Judged by what was recorded, so a priced model whose provider has since + # been disabled is not listed. BYOM ids are UUIDs and $0 by design; a + # model explicitly priced at $0 is free, not unpriced. + if row["cost"] == 0 and not looks_like_uuid(row["model_id"]) and not is_priced(row["model_id"]) ] diff --git a/docsgpt/quotas/service.py b/docsgpt/quotas/service.py index 7659fe1d..fff7178a 100644 --- a/docsgpt/quotas/service.py +++ b/docsgpt/quotas/service.py @@ -43,18 +43,19 @@ class BucketStatus: def to_dict(self) -> dict: """Return the JSON shape shared by the admin and user quota endpoints.""" - def budget(limit: ResolvedLimit, used: float) -> dict: + def budget(limit: Optional[float], resolved: ResolvedLimit, used: float) -> dict: return { - "limit": limit.limit, + "limit": limit, "used": used, - "source": limit.source, - "source_id": limit.source_id, + "source": resolved.source, + "source_id": resolved.source_id, } + tokens, cost = self.limits.tokens, self.limits.cost return { "bucket": self.bucket, - "tokens": budget(self.limits.tokens, self.tokens_used), - "cost": budget(self.limits.cost, round(self.cost_used, 6)), + "tokens": budget(None if tokens.unlimited else int(tokens.limit), tokens, self.tokens_used), + "cost": budget(cost.limit, cost, round(self.cost_used, 6)), "resets_at": self.resets_at.isoformat(), } diff --git a/tests/api/test_quota_endpoints.py b/tests/api/test_quota_endpoints.py index 44c424db..c7337d1b 100644 --- a/tests/api/test_quota_endpoints.py +++ b/tests/api/test_quota_endpoints.py @@ -222,7 +222,8 @@ class TestUserPolicy: body = _body(client.get("/api/admin/quotas/users/u1")) overall = body["effective"][0] assert overall["bucket"] == "all" - assert overall["tokens"] == {"limit": 900.0, "used": 40, "source": "team", "source_id": big} + assert overall["tokens"] == {"limit": 900, "used": 40, "source": "team", "source_id": big} + assert isinstance(overall["tokens"]["limit"], int) assert overall["cost"] == {"limit": 1.0, "used": 0.25, "source": "instance", "source_id": None} assert body["policies"] == [] @@ -243,12 +244,14 @@ class TestUserPolicy: class TestUnpricedModels: - def test_lists_used_catalog_models_without_a_price(self, client, db): + def test_lists_models_recorded_at_zero_for_want_of_a_price(self, client, db): usage = TokenUsageRepository(db) usage.insert(user_id="u1", prompt_tokens=10, model_id="local-llama") - usage.insert(user_id="u1", prompt_tokens=5, model_id="claude-haiku-4-5", cost=0.1) + # Priced when called; its provider may be disabled by now. + usage.insert(user_id="u1", prompt_tokens=5, model_id="retired-priced-model", cost=0.1) + usage.insert(user_id="u1", prompt_tokens=3, model_id="free-model") usage.insert(user_id="u1", prompt_tokens=7, model_id="7d0c1a52-2f5e-4c53-9a0e-111111111111") - with _admin(), patch("docsgpt.api.admin.quotas.is_priced", lambda m: m == "claude-haiku-4-5"): + with _admin(), patch("docsgpt.api.admin.quotas.is_priced", lambda m: m == "free-model"): unpriced = _body(client.get("/api/admin/quotas"))["unpriced_models"] assert unpriced == [{"model_id": "local-llama", "tokens": 10, "cost": 0.0}] @@ -266,7 +269,7 @@ class TestMyQuota: body = _body(client.get("/api/user/quota")) (bucket,) = body["buckets"] assert bucket["bucket"] == "all" - assert bucket["tokens"] == {"limit": 100.0, "used": 30} + assert bucket["tokens"] == {"limit": 100, "used": 30} assert bucket["cost"] == {"limit": None, "used": 0.0} assert "source" not in json.dumps(body) and "secret" not in json.dumps(body) diff --git a/tests/quotas/test_service.py b/tests/quotas/test_service.py index 0a365799..13975bb3 100644 --- a/tests/quotas/test_service.py +++ b/tests/quotas/test_service.py @@ -210,7 +210,7 @@ class TestStatusAndPayload: (status,) = QuotaService.status("u1", now=NOW) assert status.to_dict() == { "bucket": "all", - "tokens": {"limit": 100.0, "used": 40, "source": "instance", "source_id": None}, + "tokens": {"limit": 100, "used": 40, "source": "instance", "source_id": None}, "cost": {"limit": 2.0, "used": 0.5, "source": "instance", "source_id": None}, "resets_at": "2026-10-01T00:00:00+00:00", }