mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 12:11:45 +00:00
fix(quotas): keyless agent chat bucket, keep disabled policies disabled, cached rates
- check_usage treats a request through a keyless (draft) agent as agent traffic, matching how its usage rows are bucketed and the headless rule - dashboard edits carry the stored enabled flag instead of re-enabling the policy; disabled policies are labelled in the Quotas tab and the editor - quota 429s send x-should-retry: false so OpenAI SDK clients do not retry a refusal that cannot succeed before the reset - cached-input and cache-write rates for Anthropic, OpenRouter and Groq gpt-oss-120b; refresh OpenRouter deepseek-v3.2 list prices - UsageQuota reuses usagePercent; docs note that a user override needs an existing user
This commit is contained in:
1 parent
b5296df8a9
commit
1e14605ee7
15 files changed
+87
-22
No files matched your search
@@ -97,7 +97,7 @@ Every admin endpoint requires the admin role, and every change is written to the
|
||||
| `GET` | `/api/admin/quotas` | All policies by layer, the current window, and used models without a price. |
|
||||
| `PUT` `DELETE` | `/api/admin/quotas/instance` | The instance default. |
|
||||
| `GET` `PUT` `DELETE` | `/api/admin/quotas/teams/<team_id>` | A team's per-member allowance. |
|
||||
| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/<user_id>` | A user's override. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. |
|
||||
| `GET` `PUT` `DELETE` | `/api/admin/quotas/users/<user_id>` | A user's override; the user must already exist (SCIM-provisioned, or signed in once), otherwise `404`. `GET` also returns the limits the user ends up with, the layer each came from, and their usage. |
|
||||
| `GET` | `/api/user/quota` | The caller's own limits, usage and reset time. |
|
||||
|
||||
A `PUT` body sets, per budget, a limit or the unlimited flag; leave both out to defer to the next layer:
|
||||
|
||||
@@ -132,7 +132,9 @@ class AnswerResource(Resource, BaseAnswerResource):
|
||||
return make_response({"error": "Unauthorized"}, 401)
|
||||
|
||||
if error := self.check_usage(
|
||||
processor.agent_config, processor.decoded_token
|
||||
processor.agent_config,
|
||||
processor.decoded_token,
|
||||
agent_id=processor.agent_id,
|
||||
):
|
||||
return error
|
||||
|
||||
|
||||
@@ -115,7 +115,10 @@ class BaseAnswerResource:
|
||||
return prepared
|
||||
|
||||
def check_usage(
|
||||
self, agent_config: Dict, decoded_token: Optional[Dict] = None
|
||||
self,
|
||||
agent_config: Dict,
|
||||
decoded_token: Optional[Dict] = None,
|
||||
agent_id: Optional[str] = None,
|
||||
) -> Optional[Response]:
|
||||
"""Refuse the request when a usage limit is exhausted.
|
||||
|
||||
@@ -127,6 +130,8 @@ class BaseAnswerResource:
|
||||
agent_config: The config dict of agent instance
|
||||
decoded_token: The request's resolved identity; its ``sub`` is the
|
||||
billable user.
|
||||
agent_id: The agent the request runs through. A draft agent has no
|
||||
key, but its usage rows carry the agent id, so it is agent traffic.
|
||||
|
||||
Returns:
|
||||
None or Response if either of limits exceeded.
|
||||
@@ -134,7 +139,7 @@ class BaseAnswerResource:
|
||||
"""
|
||||
api_key = agent_config.get("user_api_key")
|
||||
user_id = (decoded_token or {}).get("sub") or agent_config.get("user_id")
|
||||
exceeded = QuotaService.check(user_id, "agent" if api_key else "direct")
|
||||
exceeded = QuotaService.check(user_id, "agent" if api_key or agent_id else "direct")
|
||||
if exceeded is not None:
|
||||
return quota_exceeded_response(exceeded)
|
||||
if not api_key:
|
||||
@@ -227,7 +232,9 @@ class BaseAnswerResource:
|
||||
Returns:
|
||||
None, or the refusal Response.
|
||||
"""
|
||||
error = self.check_usage(processor.agent_config, processor.decoded_token)
|
||||
error = self.check_usage(
|
||||
processor.agent_config, processor.decoded_token, agent_id=processor.agent_id
|
||||
)
|
||||
if error is None or not conversation_id:
|
||||
return error
|
||||
user = processor.initial_user_id or (processor.decoded_token or {}).get("sub")
|
||||
|
||||
@@ -154,7 +154,9 @@ class StreamResource(Resource, BaseAnswerResource):
|
||||
)
|
||||
|
||||
if error := self.check_usage(
|
||||
processor.agent_config, processor.decoded_token
|
||||
processor.agent_config,
|
||||
processor.decoded_token,
|
||||
agent_id=processor.agent_id,
|
||||
):
|
||||
return error
|
||||
should_persist, visibility = resolve_persistence(
|
||||
|
||||
@@ -344,7 +344,9 @@ def chat_completions():
|
||||
if claimed_conversation_id:
|
||||
usage_error = helper.check_usage_on_resume(processor, claimed_conversation_id)
|
||||
else:
|
||||
usage_error = helper.check_usage(processor.agent_config, processor.decoded_token)
|
||||
usage_error = helper.check_usage(
|
||||
processor.agent_config, processor.decoded_token, agent_id=processor.agent_id
|
||||
)
|
||||
if usage_error:
|
||||
return usage_error
|
||||
|
||||
|
||||
@@ -12,6 +12,8 @@ models:
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 5.0
|
||||
output_cost_per_million: 25.0
|
||||
cached_input_cost_per_million: 0.5
|
||||
cache_write_cost_per_million: 6.25
|
||||
|
||||
- id: claude-sonnet-4-6
|
||||
display_name: Claude Sonnet 4.6
|
||||
@@ -20,6 +22,8 @@ models:
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 3.0
|
||||
output_cost_per_million: 15.0
|
||||
cached_input_cost_per_million: 0.3
|
||||
cache_write_cost_per_million: 3.75
|
||||
|
||||
- id: claude-haiku-4-5
|
||||
display_name: Claude Haiku 4.5
|
||||
@@ -27,3 +31,5 @@ models:
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 1.0
|
||||
output_cost_per_million: 5.0
|
||||
cached_input_cost_per_million: 0.1
|
||||
cache_write_cost_per_million: 1.25
|
||||
@@ -10,6 +10,7 @@ models:
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 0.15
|
||||
output_cost_per_million: 0.6
|
||||
cached_input_cost_per_million: 0.075
|
||||
- id: llama-3.3-70b-versatile
|
||||
display_name: Llama 3.3 70B Versatile
|
||||
description: Meta's Llama 3.3 70B for general-purpose chat with parallel tool use
|
||||
|
||||
@@ -15,12 +15,13 @@ models:
|
||||
|
||||
- id: deepseek/deepseek-v3.2
|
||||
display_name: DeepSeek V3.2
|
||||
description: Open-weights reasoning model, very low cost (~$0.23 in / $0.34 out per 1M)
|
||||
description: Open-weights reasoning model, very low cost (~$0.27 in / $0.40 out per 1M)
|
||||
context_window: 131072
|
||||
attachments: []
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 0.23
|
||||
output_cost_per_million: 0.34
|
||||
input_cost_per_million: 0.269
|
||||
output_cost_per_million: 0.4
|
||||
cached_input_cost_per_million: 0.1345
|
||||
|
||||
- id: anthropic/claude-sonnet-4.6
|
||||
display_name: Claude Sonnet 4.6 (via OpenRouter)
|
||||
@@ -29,3 +30,5 @@ models:
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 3.0
|
||||
output_cost_per_million: 15.0
|
||||
cached_input_cost_per_million: 0.3
|
||||
cache_write_cost_per_million: 3.75
|
||||
@@ -11,4 +11,6 @@ def quota_exceeded_response(exceeded: QuotaExceeded) -> Response:
|
||||
"""Return the 429 for an exhausted quota, with ``Retry-After`` set to the reset."""
|
||||
response = make_response(jsonify(exceeded.to_payload()), 429)
|
||||
response.headers["Retry-After"] = str(exceeded.retry_after_seconds)
|
||||
# The reset can be weeks away; the OpenAI SDKs would otherwise retry with backoff.
|
||||
response.headers["x-should-retry"] = "false"
|
||||
return response
|
||||
@@ -185,7 +185,7 @@ export default function QuotaEditor({
|
||||
if (policy) remove();
|
||||
return;
|
||||
}
|
||||
const result = formToPolicy(form);
|
||||
const result = formToPolicy(form, policy);
|
||||
if (!result.ok) {
|
||||
setError(result.error);
|
||||
return;
|
||||
@@ -196,6 +196,11 @@ export default function QuotaEditor({
|
||||
return (
|
||||
<div className="space-y-4">
|
||||
<p className="text-muted-foreground text-xs">{inheritHint}</p>
|
||||
{policy && !policy.enabled ? (
|
||||
<p className="text-xs text-amber-600 dark:text-amber-400">
|
||||
This policy is disabled and is not enforced. Saving keeps it disabled.
|
||||
</p>
|
||||
) : null}
|
||||
<BudgetField
|
||||
label="Tokens"
|
||||
hint="e.g. 2000000"
|
||||
|
||||
@@ -116,8 +116,12 @@ export default function Quotas() {
|
||||
if (data === null && loading) return <Loading />;
|
||||
if (!data?.success) return <LoadError message="Failed to load quotas." />;
|
||||
|
||||
const bucketPill = (policy: QuotaPolicy) =>
|
||||
isAll(policy) ? null : <Pill tone="muted">{policy.bucket} traffic</Pill>;
|
||||
const bucketPill = (policy: QuotaPolicy) => (
|
||||
<>
|
||||
{isAll(policy) ? null : <Pill tone="muted">{policy.bucket} traffic</Pill>}
|
||||
{policy.enabled ? null : <Pill tone="muted">Disabled</Pill>}
|
||||
</>
|
||||
);
|
||||
|
||||
return (
|
||||
<div className="mt-6 space-y-8">
|
||||
@@ -163,7 +167,7 @@ export default function Quotas() {
|
||||
</div>
|
||||
<p className="text-muted-foreground mt-1 text-sm">
|
||||
{instancePolicy
|
||||
? `Tokens: ${describeBudget(instancePolicy.token_limit, instancePolicy.token_unlimited, 'tokens')} · Cost: ${describeBudget(instancePolicy.cost_limit_usd, instancePolicy.cost_unlimited, 'cost')}`
|
||||
? `${instancePolicy.enabled ? '' : 'Disabled · '}Tokens: ${describeBudget(instancePolicy.token_limit, instancePolicy.token_unlimited, 'tokens')} · Cost: ${describeBudget(instancePolicy.cost_limit_usd, instancePolicy.cost_unlimited, 'cost')}`
|
||||
: 'No default: users without a team allowance or override are unlimited.'}
|
||||
</p>
|
||||
</section>
|
||||
|
||||
@@ -54,6 +54,7 @@ describe('formToPolicy', () => {
|
||||
ok: true,
|
||||
policy: {
|
||||
bucket: 'all',
|
||||
enabled: true,
|
||||
token_limit: 5000,
|
||||
token_unlimited: false,
|
||||
cost_limit_usd: 2.5,
|
||||
@@ -63,6 +64,13 @@ describe('formToPolicy', () => {
|
||||
});
|
||||
});
|
||||
|
||||
it('keeps a disabled policy disabled', () => {
|
||||
const form = { ...base, tokenMode: 'limit' as const, tokenLimit: '10' };
|
||||
const stored = policy({ enabled: false });
|
||||
const result = formToPolicy(form, stored);
|
||||
expect(result.ok && result.policy.enabled).toBe(false);
|
||||
});
|
||||
|
||||
it('sends unlimited without a limit', () => {
|
||||
const result = formToPolicy({
|
||||
...base,
|
||||
|
||||
@@ -66,9 +66,15 @@ export function isEmptyForm(form: QuotaForm): boolean {
|
||||
return form.tokenMode === 'inherit' && form.costMode === 'inherit';
|
||||
}
|
||||
|
||||
export function formToPolicy(form: QuotaForm): FormResult {
|
||||
// ``existing`` carries the stored ``enabled`` flag through an edit: the form has
|
||||
// no control for it, and a body without it would switch the policy back on.
|
||||
export function formToPolicy(
|
||||
form: QuotaForm,
|
||||
existing?: QuotaPolicy | null,
|
||||
): FormResult {
|
||||
const policy: Record<string, unknown> = {
|
||||
bucket: 'all',
|
||||
enabled: existing?.enabled ?? true,
|
||||
token_limit: null,
|
||||
token_unlimited: form.tokenMode === 'unlimited',
|
||||
cost_limit_usd: null,
|
||||
|
||||
@@ -2,6 +2,7 @@ import { useEffect, useState } from 'react';
|
||||
import { useTranslation } from 'react-i18next';
|
||||
import { useSelector } from 'react-redux';
|
||||
|
||||
import { usagePercent } from '../../admin/quotaUtils';
|
||||
import userService from '../../api/services/userService';
|
||||
import { selectToken } from '../../preferences/preferenceSlice';
|
||||
|
||||
@@ -24,10 +25,7 @@ function Meter({
|
||||
}) {
|
||||
const { t } = useTranslation();
|
||||
if (budget.limit === null) return null;
|
||||
const percent =
|
||||
budget.limit <= 0
|
||||
? 100
|
||||
: Math.min(100, Math.max(0, (budget.used / budget.limit) * 100));
|
||||
const percent = usagePercent(budget.used, budget.limit);
|
||||
const tone =
|
||||
percent >= 100
|
||||
? 'bg-red-500'
|
||||
|
||||
@@ -30,11 +30,11 @@ def _spend(conn, user_id, tokens, api_key=None):
|
||||
TokenUsageRepository(conn).insert(user_id=user_id, api_key=api_key, prompt_tokens=tokens)
|
||||
|
||||
|
||||
def _check(flask_app, agent_config, decoded_token=None):
|
||||
def _check(flask_app, agent_config, decoded_token=None, agent_id=None):
|
||||
from docsgpt.api.answer.routes.base import BaseAnswerResource
|
||||
|
||||
with flask_app.app_context():
|
||||
return BaseAnswerResource().check_usage(agent_config, decoded_token)
|
||||
return BaseAnswerResource().check_usage(agent_config, decoded_token, agent_id=agent_id)
|
||||
|
||||
|
||||
class TestCheckUsage:
|
||||
@@ -90,6 +90,23 @@ class TestCheckUsage:
|
||||
_spend(db, "u1", 500)
|
||||
assert _check(flask_app, {}, {"sub": "u1"}) is None
|
||||
|
||||
def test_a_keyless_agent_chat_is_agent_traffic(self, db, flask_app):
|
||||
agent_id = str(AgentsRepository(db).create("u1", "draft", "draft")["id"])
|
||||
QuotaPoliciesRepository(db).upsert(scope="user", subject_id="u1", bucket="agent", token_limit=10)
|
||||
TokenUsageRepository(db).insert(user_id="u1", agent_id=agent_id, prompt_tokens=10)
|
||||
|
||||
response = _check(flask_app, {"user_api_key": None}, {"sub": "u1"}, agent_id=agent_id)
|
||||
|
||||
assert response.status_code == 429
|
||||
assert json.loads(response.data)["bucket"] == "agent"
|
||||
# The same spend leaves chat without an agent alone.
|
||||
assert _check(flask_app, {}, {"sub": "u1"}) is None
|
||||
|
||||
def test_a_refusal_tells_sdk_clients_not_to_retry(self, db, flask_app):
|
||||
QuotaPoliciesRepository(db).upsert(scope="user", subject_id="u1", token_limit=0)
|
||||
response = _check(flask_app, {}, {"sub": "u1"})
|
||||
assert response.headers["x-should-retry"] == "false"
|
||||
|
||||
|
||||
class TestHeadless:
|
||||
def test_exhausted_owner_is_refused_before_the_run(self, db):
|
||||
@@ -125,7 +142,9 @@ class TestResumeRefusal:
|
||||
def _processor(self, user_id="u1"):
|
||||
from types import SimpleNamespace
|
||||
|
||||
return SimpleNamespace(agent_config={}, decoded_token={"sub": user_id}, initial_user_id=user_id)
|
||||
return SimpleNamespace(
|
||||
agent_config={}, decoded_token={"sub": user_id}, initial_user_id=user_id, agent_id=None
|
||||
)
|
||||
|
||||
def test_a_refused_resume_releases_its_claim(self, db, flask_app):
|
||||
from docsgpt.api.answer.routes.base import BaseAnswerResource
|
||||
|
||||
Reference in new issue
Block a user