fix(quotas): keyless agent chat bucket, keep disabled policies disabled, cached rates

- check_usage treats a request through a keyless (draft) agent as agent
  traffic, matching how its usage rows are bucketed and the headless rule
- dashboard edits carry the stored enabled flag instead of re-enabling the
  policy; disabled policies are labelled in the Quotas tab and the editor
- quota 429s send x-should-retry: false so OpenAI SDK clients do not retry
  a refusal that cannot succeed before the reset
- cached-input and cache-write rates for Anthropic, OpenRouter and Groq
  gpt-oss-120b; refresh OpenRouter deepseek-v3.2 list prices
- UsageQuota reuses usagePercent; docs note that a user override needs an
  existing user
This commit is contained in:
Alex committed 2026-09-21 15:26:29 +01:00
1 parent b5296df8a9
commit 1e14605ee7
15 files changed
+87 -22

No files matched your search

+1 -1
View File
@@ -97,7 +97,7 @@ Every admin endpoint requires the admin role, and every change is written to the
| `GET` | `/api/admin/quotas` | All policies by layer, the current window, and used models without a price. |
| `PUT` `DELETE` | `/api/admin/quotas/instance` | The instance default. |
| `GET` `PUT` `DELETE` | `/api/admin/quotas/teams/<team_id>` | A team's per-member allowance. |
| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/<user_id>` | A user's override. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. |
| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/<user_id>` | A user's override; the user must already exist (SCIM-provisioned, or signed in once), otherwise `404`. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. |
| `GET` | `/api/user/quota` | The caller's own limits, usage and reset time. |
A `PUT` body sets, per budget, a limit or the unlimited flag; leave both out to defer to the next layer:
+3 -1
View File
@@ -132,7 +132,9 @@ class AnswerResource(Resource, BaseAnswerResource):
return make_response({"error": "Unauthorized"}, 401)
if error := self.check_usage(
processor.agent_config, processor.decoded_token
processor.agent_config,
processor.decoded_token,
agent_id=processor.agent_id,
):
return error
+10 -3
View File
@@ -115,7 +115,10 @@ class BaseAnswerResource:
return prepared
def check_usage(
self, agent_config: Dict, decoded_token: Optional[Dict] = None
self,
agent_config: Dict,
decoded_token: Optional[Dict] = None,
agent_id: Optional[str] = None,
) -> Optional[Response]:
"""Refuse the request when a usage limit is exhausted.
@@ -127,6 +130,8 @@ class BaseAnswerResource:
agent_config: The config dict of agent instance
decoded_token: The request's resolved identity; its ``sub`` is the
billable user.
agent_id: The agent the request runs through. A draft agent has no
key, but its usage rows carry the agent id, so it is agent traffic.
Returns:
None or Response if either of limits exceeded.
@@ -134,7 +139,7 @@ class BaseAnswerResource:
"""
api_key = agent_config.get("user_api_key")
user_id = (decoded_token or {}).get("sub") or agent_config.get("user_id")
exceeded = QuotaService.check(user_id, "agent" if api_key else "direct")
exceeded = QuotaService.check(user_id, "agent" if api_key or agent_id else "direct")
if exceeded is not None:
return quota_exceeded_response(exceeded)
if not api_key:
@@ -227,7 +232,9 @@ class BaseAnswerResource:
Returns:
None, or the refusal Response.
"""
error = self.check_usage(processor.agent_config, processor.decoded_token)
error = self.check_usage(
processor.agent_config, processor.decoded_token, agent_id=processor.agent_id
)
if error is None or not conversation_id:
return error
user = processor.initial_user_id or (processor.decoded_token or {}).get("sub")
+3 -1
View File
@@ -154,7 +154,9 @@ class StreamResource(Resource, BaseAnswerResource):
)
if error := self.check_usage(
processor.agent_config, processor.decoded_token
processor.agent_config,
processor.decoded_token,
agent_id=processor.agent_id,
):
return error
should_persist, visibility = resolve_persistence(
+3 -1
View File
@@ -344,7 +344,9 @@ def chat_completions():
if claimed_conversation_id:
usage_error = helper.check_usage_on_resume(processor, claimed_conversation_id)
else:
usage_error = helper.check_usage(processor.agent_config, processor.decoded_token)
usage_error = helper.check_usage(
processor.agent_config, processor.decoded_token, agent_id=processor.agent_id
)
if usage_error:
return usage_error
+6
View File
@@ -12,6 +12,8 @@ models:
supports_structured_output: true
input_cost_per_million: 5.0
output_cost_per_million: 25.0
cached_input_cost_per_million: 0.5
cache_write_cost_per_million: 6.25
- id: claude-sonnet-4-6
display_name: Claude Sonnet 4.6
@@ -20,6 +22,8 @@ models:
supports_structured_output: true
input_cost_per_million: 3.0
output_cost_per_million: 15.0
cached_input_cost_per_million: 0.3
cache_write_cost_per_million: 3.75
- id: claude-haiku-4-5
display_name: Claude Haiku 4.5
@@ -27,3 +31,5 @@ models:
supports_structured_output: true
input_cost_per_million: 1.0
output_cost_per_million: 5.0
cached_input_cost_per_million: 0.1
cache_write_cost_per_million: 1.25
+1
View File
@@ -10,6 +10,7 @@ models:
supports_structured_output: true
input_cost_per_million: 0.15
output_cost_per_million: 0.6
cached_input_cost_per_million: 0.075
- id: llama-3.3-70b-versatile
display_name: Llama 3.3 70B Versatile
description: Meta's Llama 3.3 70B for general-purpose chat with parallel tool use
+6 -3
View File
@@ -15,12 +15,13 @@ models:
- id: deepseek/deepseek-v3.2
display_name: DeepSeek V3.2
description: Open-weights reasoning model, very low cost (~$0.23 in / $0.34 out per 1M)
description: Open-weights reasoning model, very low cost (~$0.27 in / $0.40 out per 1M)
context_window: 131072
attachments: []
supports_structured_output: true
input_cost_per_million: 0.23
output_cost_per_million: 0.34
input_cost_per_million: 0.269
output_cost_per_million: 0.4
cached_input_cost_per_million: 0.1345
- id: anthropic/claude-sonnet-4.6
display_name: Claude Sonnet 4.6 (via OpenRouter)
@@ -29,3 +30,5 @@ models:
supports_structured_output: true
input_cost_per_million: 3.0
output_cost_per_million: 15.0
cached_input_cost_per_million: 0.3
cache_write_cost_per_million: 3.75
+2
View File
@@ -11,4 +11,6 @@ def quota_exceeded_response(exceeded: QuotaExceeded) -> Response:
"""Return the 429 for an exhausted quota, with ``Retry-After`` set to the reset."""
response = make_response(jsonify(exceeded.to_payload()), 429)
response.headers["Retry-After"] = str(exceeded.retry_after_seconds)
# The reset can be weeks away; the OpenAI SDKs would otherwise retry with backoff.
response.headers["x-should-retry"] = "false"
return response
+6 -1
View File
@@ -185,7 +185,7 @@ export default function QuotaEditor({
if (policy) remove();
return;
}
const result = formToPolicy(form);
const result = formToPolicy(form, policy);
if (!result.ok) {
setError(result.error);
return;
@@ -196,6 +196,11 @@ export default function QuotaEditor({
return (
<div className="space-y-4">
<p className="text-muted-foreground text-xs">{inheritHint}</p>
{policy && !policy.enabled ? (
<p className="text-xs text-amber-600 dark:text-amber-400">
This policy is disabled and is not enforced. Saving keeps it disabled.
</p>
) : null}
<BudgetField
label="Tokens"
hint="e.g. 2000000"
+7 -3
View File
@@ -116,8 +116,12 @@ export default function Quotas() {
if (data === null && loading) return <Loading />;
if (!data?.success) return <LoadError message="Failed to load quotas." />;
const bucketPill = (policy: QuotaPolicy) =>
isAll(policy) ? null : <Pill tone="muted">{policy.bucket} traffic</Pill>;
const bucketPill = (policy: QuotaPolicy) => (
<>
{isAll(policy) ? null : <Pill tone="muted">{policy.bucket} traffic</Pill>}
{policy.enabled ? null : <Pill tone="muted">Disabled</Pill>}
</>
);
return (
<div className="mt-6 space-y-8">
@@ -163,7 +167,7 @@ export default function Quotas() {
</div>
<p className="text-muted-foreground mt-1 text-sm">
{instancePolicy
? `Tokens: ${describeBudget(instancePolicy.token_limit, instancePolicy.token_unlimited, 'tokens')} · Cost: ${describeBudget(instancePolicy.cost_limit_usd, instancePolicy.cost_unlimited, 'cost')}`
? `${instancePolicy.enabled ? '' : 'Disabled · '}Tokens: ${describeBudget(instancePolicy.token_limit, instancePolicy.token_unlimited, 'tokens')} · Cost: ${describeBudget(instancePolicy.cost_limit_usd, instancePolicy.cost_unlimited, 'cost')}`
: 'No default: users without a team allowance or override are unlimited.'}
</p>
</section>
+8
View File
@@ -54,6 +54,7 @@ describe('formToPolicy', () => {
ok: true,
policy: {
bucket: 'all',
enabled: true,
token_limit: 5000,
token_unlimited: false,
cost_limit_usd: 2.5,
@@ -63,6 +64,13 @@ describe('formToPolicy', () => {
});
});
it('keeps a disabled policy disabled', () => {
const form = { ...base, tokenMode: 'limit' as const, tokenLimit: '10' };
const stored = policy({ enabled: false });
const result = formToPolicy(form, stored);
expect(result.ok && result.policy.enabled).toBe(false);
});
it('sends unlimited without a limit', () => {
const result = formToPolicy({
...base,
+7 -1
View File
@@ -66,9 +66,15 @@ export function isEmptyForm(form: QuotaForm): boolean {
return form.tokenMode === 'inherit' && form.costMode === 'inherit';
}
export function formToPolicy(form: QuotaForm): FormResult {
// ``existing`` carries the stored ``enabled`` flag through an edit: the form has
// no control for it, and a body without it would switch the policy back on.
export function formToPolicy(
form: QuotaForm,
existing?: QuotaPolicy | null,
): FormResult {
const policy: Record<string, unknown> = {
bucket: 'all',
enabled: existing?.enabled ?? true,
token_limit: null,
token_unlimited: form.tokenMode === 'unlimited',
cost_limit_usd: null,
@@ -2,6 +2,7 @@ import { useEffect, useState } from 'react';
import { useTranslation } from 'react-i18next';
import { useSelector } from 'react-redux';
import { usagePercent } from '../../admin/quotaUtils';
import userService from '../../api/services/userService';
import { selectToken } from '../../preferences/preferenceSlice';
@@ -24,10 +25,7 @@ function Meter({
}) {
const { t } = useTranslation();
if (budget.limit === null) return null;
const percent =
budget.limit <= 0
? 100
: Math.min(100, Math.max(0, (budget.used / budget.limit) * 100));
const percent = usagePercent(budget.used, budget.limit);
const tone =
percent >= 100
? 'bg-red-500'
+22 -3
View File
@@ -30,11 +30,11 @@ def _spend(conn, user_id, tokens, api_key=None):
TokenUsageRepository(conn).insert(user_id=user_id, api_key=api_key, prompt_tokens=tokens)
def _check(flask_app, agent_config, decoded_token=None):
def _check(flask_app, agent_config, decoded_token=None, agent_id=None):
from docsgpt.api.answer.routes.base import BaseAnswerResource
with flask_app.app_context():
return BaseAnswerResource().check_usage(agent_config, decoded_token)
return BaseAnswerResource().check_usage(agent_config, decoded_token, agent_id=agent_id)
class TestCheckUsage:
@@ -90,6 +90,23 @@ class TestCheckUsage:
_spend(db, "u1", 500)
assert _check(flask_app, {}, {"sub": "u1"}) is None
def test_a_keyless_agent_chat_is_agent_traffic(self, db, flask_app):
agent_id = str(AgentsRepository(db).create("u1", "draft", "draft")["id"])
QuotaPoliciesRepository(db).upsert(scope="user", subject_id="u1", bucket="agent", token_limit=10)
TokenUsageRepository(db).insert(user_id="u1", agent_id=agent_id, prompt_tokens=10)
response = _check(flask_app, {"user_api_key": None}, {"sub": "u1"}, agent_id=agent_id)
assert response.status_code == 429
assert json.loads(response.data)["bucket"] == "agent"
# The same spend leaves chat without an agent alone.
assert _check(flask_app, {}, {"sub": "u1"}) is None
def test_a_refusal_tells_sdk_clients_not_to_retry(self, db, flask_app):
QuotaPoliciesRepository(db).upsert(scope="user", subject_id="u1", token_limit=0)
response = _check(flask_app, {}, {"sub": "u1"})
assert response.headers["x-should-retry"] == "false"
class TestHeadless:
def test_exhausted_owner_is_refused_before_the_run(self, db):
@@ -125,7 +142,9 @@ class TestResumeRefusal:
def _processor(self, user_id="u1"):
from types import SimpleNamespace
return SimpleNamespace(agent_config={}, decoded_token={"sub": user_id}, initial_user_id=user_id)
return SimpleNamespace(
agent_config={}, decoded_token={"sub": user_id}, initial_user_id=user_id, agent_id=None
)
def test_a_refused_resume_releases_its_claim(self, db, flask_app):
from docsgpt.api.answer.routes.base import BaseAnswerResource