From 1e14605ee7c18636ba46e5f7bd158b1ef0d91398 Mon Sep 17 00:00:00 2001 From: Alex Date: Mon, 21 Sep 2026 15:26:29 +0100 Subject: [PATCH] fix(quotas): keyless agent chat bucket, keep disabled policies disabled, cached rates - check_usage treats a request through a keyless (draft) agent as agent traffic, matching how its usage rows are bucketed and the headless rule - dashboard edits carry the stored enabled flag instead of re-enabling the policy; disabled policies are labelled in the Quotas tab and the editor - quota 429s send x-should-retry: false so OpenAI SDK clients do not retry a refusal that cannot succeed before the reset - cached-input and cache-write rates for Anthropic, OpenRouter and Groq gpt-oss-120b; refresh OpenRouter deepseek-v3.2 list prices - UsageQuota reuses usagePercent; docs note that a user override needs an existing user --- docs/content/Deploying/Usage-Quotas.mdx | 2 +- docsgpt/api/answer/routes/answer.py | 4 ++- docsgpt/api/answer/routes/base.py | 13 +++++++--- docsgpt/api/answer/routes/stream.py | 4 ++- docsgpt/api/v1/routes.py | 4 ++- docsgpt/core/models/anthropic.yaml | 6 +++++ docsgpt/core/models/groq.yaml | 1 + docsgpt/core/models/openrouter.yaml | 9 ++++--- docsgpt/quotas/http.py | 2 ++ frontend/src/admin/QuotaEditor.tsx | 7 +++++- frontend/src/admin/Quotas.tsx | 10 +++++--- frontend/src/admin/quotaUtils.test.ts | 8 ++++++ frontend/src/admin/quotaUtils.ts | 8 +++++- .../src/settings/components/UsageQuota.tsx | 6 ++--- tests/quotas/test_enforcement.py | 25 ++++++++++++++++--- 15 files changed, 87 insertions(+), 22 deletions(-) diff --git a/docs/content/Deploying/Usage-Quotas.mdx b/docs/content/Deploying/Usage-Quotas.mdx index c7391625..e4ffbaf1 100644 --- a/docs/content/Deploying/Usage-Quotas.mdx +++ b/docs/content/Deploying/Usage-Quotas.mdx @@ -97,7 +97,7 @@ Every admin endpoint requires the admin role, and every change is written to the | `GET` | `/api/admin/quotas` | All policies by layer, the current window, and used models without a price. | | `PUT` `DELETE` | `/api/admin/quotas/instance` | The instance default. | | `GET` `PUT` `DELETE` | `/api/admin/quotas/teams/` | A team's per-member allowance. | -| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/` | A user's override. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. | +| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/` | A user's override; the user must already exist (SCIM-provisioned, or signed in once), otherwise `404`. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. | | `GET` | `/api/user/quota` | The caller's own limits, usage and reset time. | A `PUT` body sets, per budget, a limit or the unlimited flag; leave both out to defer to the next layer: diff --git a/docsgpt/api/answer/routes/answer.py b/docsgpt/api/answer/routes/answer.py index b5819e3e..7cdcd722 100644 --- a/docsgpt/api/answer/routes/answer.py +++ b/docsgpt/api/answer/routes/answer.py @@ -132,7 +132,9 @@ class AnswerResource(Resource, BaseAnswerResource): return make_response({"error": "Unauthorized"}, 401) if error := self.check_usage( - processor.agent_config, processor.decoded_token + processor.agent_config, + processor.decoded_token, + agent_id=processor.agent_id, ): return error diff --git a/docsgpt/api/answer/routes/base.py b/docsgpt/api/answer/routes/base.py index c2b3f929..e0cfe4cd 100644 --- a/docsgpt/api/answer/routes/base.py +++ b/docsgpt/api/answer/routes/base.py @@ -115,7 +115,10 @@ class BaseAnswerResource: return prepared def check_usage( - self, agent_config: Dict, decoded_token: Optional[Dict] = None + self, + agent_config: Dict, + decoded_token: Optional[Dict] = None, + agent_id: Optional[str] = None, ) -> Optional[Response]: """Refuse the request when a usage limit is exhausted. @@ -127,6 +130,8 @@ class BaseAnswerResource: agent_config: The config dict of agent instance decoded_token: The request's resolved identity; its ``sub`` is the billable user. + agent_id: The agent the request runs through. A draft agent has no + key, but its usage rows carry the agent id, so it is agent traffic. Returns: None or Response if either of limits exceeded. @@ -134,7 +139,7 @@ class BaseAnswerResource: """ api_key = agent_config.get("user_api_key") user_id = (decoded_token or {}).get("sub") or agent_config.get("user_id") - exceeded = QuotaService.check(user_id, "agent" if api_key else "direct") + exceeded = QuotaService.check(user_id, "agent" if api_key or agent_id else "direct") if exceeded is not None: return quota_exceeded_response(exceeded) if not api_key: @@ -227,7 +232,9 @@ class BaseAnswerResource: Returns: None, or the refusal Response. """ - error = self.check_usage(processor.agent_config, processor.decoded_token) + error = self.check_usage( + processor.agent_config, processor.decoded_token, agent_id=processor.agent_id + ) if error is None or not conversation_id: return error user = processor.initial_user_id or (processor.decoded_token or {}).get("sub") diff --git a/docsgpt/api/answer/routes/stream.py b/docsgpt/api/answer/routes/stream.py index 54b62ec8..cdee274b 100644 --- a/docsgpt/api/answer/routes/stream.py +++ b/docsgpt/api/answer/routes/stream.py @@ -154,7 +154,9 @@ class StreamResource(Resource, BaseAnswerResource): ) if error := self.check_usage( - processor.agent_config, processor.decoded_token + processor.agent_config, + processor.decoded_token, + agent_id=processor.agent_id, ): return error should_persist, visibility = resolve_persistence( diff --git a/docsgpt/api/v1/routes.py b/docsgpt/api/v1/routes.py index a9902a4a..fa915d24 100644 --- a/docsgpt/api/v1/routes.py +++ b/docsgpt/api/v1/routes.py @@ -344,7 +344,9 @@ def chat_completions(): if claimed_conversation_id: usage_error = helper.check_usage_on_resume(processor, claimed_conversation_id) else: - usage_error = helper.check_usage(processor.agent_config, processor.decoded_token) + usage_error = helper.check_usage( + processor.agent_config, processor.decoded_token, agent_id=processor.agent_id + ) if usage_error: return usage_error diff --git a/docsgpt/core/models/anthropic.yaml b/docsgpt/core/models/anthropic.yaml index 784ab8e3..34a2cedc 100644 --- a/docsgpt/core/models/anthropic.yaml +++ b/docsgpt/core/models/anthropic.yaml @@ -12,6 +12,8 @@ models: supports_structured_output: true input_cost_per_million: 5.0 output_cost_per_million: 25.0 + cached_input_cost_per_million: 0.5 + cache_write_cost_per_million: 6.25 - id: claude-sonnet-4-6 display_name: Claude Sonnet 4.6 @@ -20,6 +22,8 @@ models: supports_structured_output: true input_cost_per_million: 3.0 output_cost_per_million: 15.0 + cached_input_cost_per_million: 0.3 + cache_write_cost_per_million: 3.75 - id: claude-haiku-4-5 display_name: Claude Haiku 4.5 @@ -27,3 +31,5 @@ models: supports_structured_output: true input_cost_per_million: 1.0 output_cost_per_million: 5.0 + cached_input_cost_per_million: 0.1 + cache_write_cost_per_million: 1.25 diff --git a/docsgpt/core/models/groq.yaml b/docsgpt/core/models/groq.yaml index c6e28d7a..a4d7edfd 100644 --- a/docsgpt/core/models/groq.yaml +++ b/docsgpt/core/models/groq.yaml @@ -10,6 +10,7 @@ models: supports_structured_output: true input_cost_per_million: 0.15 output_cost_per_million: 0.6 + cached_input_cost_per_million: 0.075 - id: llama-3.3-70b-versatile display_name: Llama 3.3 70B Versatile description: Meta's Llama 3.3 70B for general-purpose chat with parallel tool use diff --git a/docsgpt/core/models/openrouter.yaml b/docsgpt/core/models/openrouter.yaml index f0b2cffd..2957fa98 100644 --- a/docsgpt/core/models/openrouter.yaml +++ b/docsgpt/core/models/openrouter.yaml @@ -15,12 +15,13 @@ models: - id: deepseek/deepseek-v3.2 display_name: DeepSeek V3.2 - description: Open-weights reasoning model, very low cost (~$0.23 in / $0.34 out per 1M) + description: Open-weights reasoning model, very low cost (~$0.27 in / $0.40 out per 1M) context_window: 131072 attachments: [] supports_structured_output: true - input_cost_per_million: 0.23 - output_cost_per_million: 0.34 + input_cost_per_million: 0.269 + output_cost_per_million: 0.4 + cached_input_cost_per_million: 0.1345 - id: anthropic/claude-sonnet-4.6 display_name: Claude Sonnet 4.6 (via OpenRouter) @@ -29,3 +30,5 @@ models: supports_structured_output: true input_cost_per_million: 3.0 output_cost_per_million: 15.0 + cached_input_cost_per_million: 0.3 + cache_write_cost_per_million: 3.75 diff --git a/docsgpt/quotas/http.py b/docsgpt/quotas/http.py index 494b87cf..c9a1a180 100644 --- a/docsgpt/quotas/http.py +++ b/docsgpt/quotas/http.py @@ -11,4 +11,6 @@ def quota_exceeded_response(exceeded: QuotaExceeded) -> Response: """Return the 429 for an exhausted quota, with ``Retry-After`` set to the reset.""" response = make_response(jsonify(exceeded.to_payload()), 429) response.headers["Retry-After"] = str(exceeded.retry_after_seconds) + # The reset can be weeks away; the OpenAI SDKs would otherwise retry with backoff. + response.headers["x-should-retry"] = "false" return response diff --git a/frontend/src/admin/QuotaEditor.tsx b/frontend/src/admin/QuotaEditor.tsx index e962ed46..7b9de108 100644 --- a/frontend/src/admin/QuotaEditor.tsx +++ b/frontend/src/admin/QuotaEditor.tsx @@ -185,7 +185,7 @@ export default function QuotaEditor({ if (policy) remove(); return; } - const result = formToPolicy(form); + const result = formToPolicy(form, policy); if (!result.ok) { setError(result.error); return; @@ -196,6 +196,11 @@ export default function QuotaEditor({ return (

{inheritHint}

+ {policy && !policy.enabled ? ( +

+ This policy is disabled and is not enforced. Saving keeps it disabled. +

+ ) : null} ; if (!data?.success) return ; - const bucketPill = (policy: QuotaPolicy) => - isAll(policy) ? null : {policy.bucket} traffic; + const bucketPill = (policy: QuotaPolicy) => ( + <> + {isAll(policy) ? null : {policy.bucket} traffic} + {policy.enabled ? null : Disabled} + + ); return (
@@ -163,7 +167,7 @@ export default function Quotas() {

{instancePolicy - ? `Tokens: ${describeBudget(instancePolicy.token_limit, instancePolicy.token_unlimited, 'tokens')} · Cost: ${describeBudget(instancePolicy.cost_limit_usd, instancePolicy.cost_unlimited, 'cost')}` + ? `${instancePolicy.enabled ? '' : 'Disabled · '}Tokens: ${describeBudget(instancePolicy.token_limit, instancePolicy.token_unlimited, 'tokens')} · Cost: ${describeBudget(instancePolicy.cost_limit_usd, instancePolicy.cost_unlimited, 'cost')}` : 'No default: users without a team allowance or override are unlimited.'}

diff --git a/frontend/src/admin/quotaUtils.test.ts b/frontend/src/admin/quotaUtils.test.ts index 60e4f468..2afad959 100644 --- a/frontend/src/admin/quotaUtils.test.ts +++ b/frontend/src/admin/quotaUtils.test.ts @@ -54,6 +54,7 @@ describe('formToPolicy', () => { ok: true, policy: { bucket: 'all', + enabled: true, token_limit: 5000, token_unlimited: false, cost_limit_usd: 2.5, @@ -63,6 +64,13 @@ describe('formToPolicy', () => { }); }); + it('keeps a disabled policy disabled', () => { + const form = { ...base, tokenMode: 'limit' as const, tokenLimit: '10' }; + const stored = policy({ enabled: false }); + const result = formToPolicy(form, stored); + expect(result.ok && result.policy.enabled).toBe(false); + }); + it('sends unlimited without a limit', () => { const result = formToPolicy({ ...base, diff --git a/frontend/src/admin/quotaUtils.ts b/frontend/src/admin/quotaUtils.ts index 10038574..02bc2f6b 100644 --- a/frontend/src/admin/quotaUtils.ts +++ b/frontend/src/admin/quotaUtils.ts @@ -66,9 +66,15 @@ export function isEmptyForm(form: QuotaForm): boolean { return form.tokenMode === 'inherit' && form.costMode === 'inherit'; } -export function formToPolicy(form: QuotaForm): FormResult { +// ``existing`` carries the stored ``enabled`` flag through an edit: the form has +// no control for it, and a body without it would switch the policy back on. +export function formToPolicy( + form: QuotaForm, + existing?: QuotaPolicy | null, +): FormResult { const policy: Record = { bucket: 'all', + enabled: existing?.enabled ?? true, token_limit: null, token_unlimited: form.tokenMode === 'unlimited', cost_limit_usd: null, diff --git a/frontend/src/settings/components/UsageQuota.tsx b/frontend/src/settings/components/UsageQuota.tsx index 30511081..a2147beb 100644 --- a/frontend/src/settings/components/UsageQuota.tsx +++ b/frontend/src/settings/components/UsageQuota.tsx @@ -2,6 +2,7 @@ import { useEffect, useState } from 'react'; import { useTranslation } from 'react-i18next'; import { useSelector } from 'react-redux'; +import { usagePercent } from '../../admin/quotaUtils'; import userService from '../../api/services/userService'; import { selectToken } from '../../preferences/preferenceSlice'; @@ -24,10 +25,7 @@ function Meter({ }) { const { t } = useTranslation(); if (budget.limit === null) return null; - const percent = - budget.limit <= 0 - ? 100 - : Math.min(100, Math.max(0, (budget.used / budget.limit) * 100)); + const percent = usagePercent(budget.used, budget.limit); const tone = percent >= 100 ? 'bg-red-500' diff --git a/tests/quotas/test_enforcement.py b/tests/quotas/test_enforcement.py index c6300b22..0a54974c 100644 --- a/tests/quotas/test_enforcement.py +++ b/tests/quotas/test_enforcement.py @@ -30,11 +30,11 @@ def _spend(conn, user_id, tokens, api_key=None): TokenUsageRepository(conn).insert(user_id=user_id, api_key=api_key, prompt_tokens=tokens) -def _check(flask_app, agent_config, decoded_token=None): +def _check(flask_app, agent_config, decoded_token=None, agent_id=None): from docsgpt.api.answer.routes.base import BaseAnswerResource with flask_app.app_context(): - return BaseAnswerResource().check_usage(agent_config, decoded_token) + return BaseAnswerResource().check_usage(agent_config, decoded_token, agent_id=agent_id) class TestCheckUsage: @@ -90,6 +90,23 @@ class TestCheckUsage: _spend(db, "u1", 500) assert _check(flask_app, {}, {"sub": "u1"}) is None + def test_a_keyless_agent_chat_is_agent_traffic(self, db, flask_app): + agent_id = str(AgentsRepository(db).create("u1", "draft", "draft")["id"]) + QuotaPoliciesRepository(db).upsert(scope="user", subject_id="u1", bucket="agent", token_limit=10) + TokenUsageRepository(db).insert(user_id="u1", agent_id=agent_id, prompt_tokens=10) + + response = _check(flask_app, {"user_api_key": None}, {"sub": "u1"}, agent_id=agent_id) + + assert response.status_code == 429 + assert json.loads(response.data)["bucket"] == "agent" + # The same spend leaves chat without an agent alone. + assert _check(flask_app, {}, {"sub": "u1"}) is None + + def test_a_refusal_tells_sdk_clients_not_to_retry(self, db, flask_app): + QuotaPoliciesRepository(db).upsert(scope="user", subject_id="u1", token_limit=0) + response = _check(flask_app, {}, {"sub": "u1"}) + assert response.headers["x-should-retry"] == "false" + class TestHeadless: def test_exhausted_owner_is_refused_before_the_run(self, db): @@ -125,7 +142,9 @@ class TestResumeRefusal: def _processor(self, user_id="u1"): from types import SimpleNamespace - return SimpleNamespace(agent_config={}, decoded_token={"sub": user_id}, initial_user_id=user_id) + return SimpleNamespace( + agent_config={}, decoded_token={"sub": user_id}, initial_user_id=user_id, agent_id=None + ) def test_a_refused_resume_releases_its_claim(self, db, flask_app): from docsgpt.api.answer.routes.base import BaseAnswerResource