feat(pricing): per-million model rates and a cost module

Rename the unused *_cost_per_token capability fields to USD per 1M tokens,
add prompt-cache read/write rates, and ship list prices for the hosted
catalogs. The old per-token keys still load, scaled, with a warning.

docsgpt/pricing.py turns a call's token bins into a USD cost. Models with
no declared rate cost $0 unless QUOTA_UNPRICED_RATE_PER_MILLION is set.
This commit is contained in:
Alex committed 2026-09-21 11:38:22 +01:00
1 parent 1c1bc2538f
commit b77561288d
16 files changed
+375 -12

No files matched your search

@@ -1451,6 +1451,23 @@ Type `int`, default `30`, must be `>= 1`.
Days guardrail events are kept before the cleanup task removes them.
## Quotas
Quota window and the treatment of unpriced models.
### `QUOTA_PERIOD`
Type `"day" | "week" | "month"`, default `month`.
Window every usage quota is measured over. Windows are calendar-aligned in UTC: a day starts at 00:00, a week on Monday, a month on the 1st.
### `QUOTA_UNPRICED_RATE_PER_MILLION`
Type `list[float]`, default unset.
Fallback `[input, output]` USD rates per 1M tokens for models that declare no price, e.g. `[0.5, 1.5]`. Unset, such calls are recorded at $0 and only count toward token quotas.
## Scheduler
Cadence, quotas and timeouts of scheduled runs.
+9 -2
View File
@@ -32,8 +32,15 @@ class ModelCapabilities:
supports_streaming: bool = True
supported_attachment_types: List[str] = field(default_factory=list)
context_window: int = 128000
input_cost_per_token: Optional[float] = None
output_cost_per_token: Optional[float] = None
# USD per 1M tokens; consumed by ``docsgpt/pricing.py``. ``None`` means
# "not declared": the call is recorded at $0 unless
# ``QUOTA_UNPRICED_RATE_PER_MILLION`` is set.
input_cost_per_million: Optional[float] = None
output_cost_per_million: Optional[float] = None
# Rates for the prompt-cache sub-bins of the prompt total. ``None`` bills
# those tokens at ``input_cost_per_million``.
cached_input_cost_per_million: Optional[float] = None
cache_write_cost_per_million: Optional[float] = None
# OpenAI reasoning-model effort hint (none/minimal/low/medium/high/xhigh;
# the accepted subset is model-dependent). Consumed by OpenAILLM — sent
# top-level on Chat Completions and nested under ``reasoning`` on the
+29 -5
View File
@@ -18,7 +18,7 @@ from pathlib import Path
from typing import Dict, List, Optional, Sequence
import yaml
from pydantic import BaseModel, ConfigDict, Field, field_validator
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
from docsgpt.core.model_settings import (
AvailableModel,
@@ -65,11 +65,33 @@ class _CapabilityFields(BaseModel):
supports_streaming: Optional[bool] = None
attachments: Optional[List[str]] = None
context_window: Optional[int] = None
input_cost_per_token: Optional[float] = None
output_cost_per_token: Optional[float] = None
input_cost_per_million: Optional[float] = Field(default=None, ge=0)
output_cost_per_million: Optional[float] = Field(default=None, ge=0)
cached_input_cost_per_million: Optional[float] = Field(default=None, ge=0)
cache_write_cost_per_million: Optional[float] = Field(default=None, ge=0)
reasoning_effort: Optional[str] = None
api_flavor: Optional[str] = None
@model_validator(mode="before")
@classmethod
def _per_token_alias(cls, data):
"""Accept the deprecated ``*_cost_per_token`` keys, scaled to per-1M."""
if not isinstance(data, dict):
return data
data = dict(data)
for side in ("input", "output"):
old, new = f"{side}_cost_per_token", f"{side}_cost_per_million"
if old not in data:
continue
value = data.pop(old)
if new in data:
raise ValueError(f"set only one of {old} and {new}")
if isinstance(value, bool) or not isinstance(value, (int, float)):
raise ValueError(f"{old} must be a number")
logger.warning("%s is deprecated; use %s (USD per 1M tokens)", old, new)
data[new] = value * 1_000_000
return data
@field_validator("reasoning_effort")
@classmethod
def _valid_reasoning_effort(cls, v: Optional[str]) -> Optional[str]:
@@ -237,8 +259,10 @@ def _build_model(
supports_streaming=pick("supports_streaming", True),
supported_attachment_types=expanded,
context_window=pick("context_window", 128000),
input_cost_per_token=pick("input_cost_per_token", None),
output_cost_per_token=pick("output_cost_per_token", None),
input_cost_per_million=pick("input_cost_per_million", None),
output_cost_per_million=pick("output_cost_per_million", None),
cached_input_cost_per_million=pick("cached_input_cost_per_million", None),
cache_write_cost_per_million=pick("cache_write_cost_per_million", None),
reasoning_effort=pick("reasoning_effort", None),
api_flavor=pick("api_flavor", "chat_completions"),
)
+4 -2
View File
@@ -108,8 +108,10 @@ defaults: # optional, applied to every model below
supports_streaming: bool # default true
attachments: [<alias-or-mime>, ...] # default []
context_window: int # default 128000
input_cost_per_token: float # default null
output_cost_per_token: float # default null
input_cost_per_million: float # USD per 1M prompt tokens; default null (unpriced)
output_cost_per_million: float # USD per 1M generated tokens; default null
cached_input_cost_per_million: float # prompt-cache reads; default: the input rate
cache_write_cost_per_million: float # prompt-cache writes; default: the input rate
reasoning_effort: <string> # default null; none|minimal|low|medium|high|xhigh (subset is model-dependent)
api_flavor: <string> # chat_completions (default) or responses
+6
View File
@@ -10,14 +10,20 @@ models:
description: Most capable Claude model for complex reasoning and agentic coding
context_window: 1000000
supports_structured_output: true
input_cost_per_million: 5.0
output_cost_per_million: 25.0
- id: claude-sonnet-4-6
display_name: Claude Sonnet 4.6
description: Best balance of speed and intelligence with extended thinking
context_window: 1000000
supports_structured_output: true
input_cost_per_million: 3.0
output_cost_per_million: 15.0
- id: claude-haiku-4-5
display_name: Claude Haiku 4.5
description: Fastest Claude model with near-frontier intelligence
supports_structured_output: true
input_cost_per_million: 1.0
output_cost_per_million: 5.0
+4
View File
@@ -12,7 +12,11 @@ models:
- id: deepseek-v4-flash
display_name: DeepSeek V4 Flash
description: Cost-efficient 1M-context model with hybrid thinking / non-thinking modes, tool calling and FIM completion
input_cost_per_million: 0.14
output_cost_per_million: 0.28
- id: deepseek-v4-pro
display_name: DeepSeek V4 Pro
description: Frontier 1M-context model with hybrid thinking / non-thinking modes for advanced reasoning and agentic coding
input_cost_per_million: 0.435
output_cost_per_million: 0.87
+7
View File
@@ -9,9 +9,16 @@ models:
- id: gemini-3.1-pro-preview
display_name: Gemini 3.1 Pro (preview)
description: Most capable Gemini 3 model with advanced reasoning and agentic coding (preview)
# Priced at the >200k-token tier; long prompts are common with attachments.
input_cost_per_million: 4.0
output_cost_per_million: 18.0
- id: gemini-3.5-flash
display_name: Gemini 3.5 Flash
description: Frontier-class Flash for sustained performance on agentic and coding tasks
input_cost_per_million: 1.5
output_cost_per_million: 9
- id: gemini-3.1-flash-lite
display_name: Gemini 3.1 Flash-Lite
description: Cost-efficient frontier-class multimodal model for high-throughput workloads
input_cost_per_million: 0.25
output_cost_per_million: 1.5
+6
View File
@@ -8,9 +8,15 @@ models:
display_name: GPT-OSS 120B
description: OpenAI's open-weight 120B flagship served on Groq's LPU hardware; strong general reasoning with strict structured output support
supports_structured_output: true
input_cost_per_million: 0.15
output_cost_per_million: 0.6
- id: llama-3.3-70b-versatile
display_name: Llama 3.3 70B Versatile
description: Meta's Llama 3.3 70B for general-purpose chat with parallel tool use
input_cost_per_million: 0.59
output_cost_per_million: 0.79
- id: llama-3.1-8b-instant
display_name: Llama 3.1 8B Instant
description: Small, very low-latency Llama model (~560 tok/s) with parallel tool use
input_cost_per_million: 0.05
output_cost_per_million: 0.08
+6
View File
@@ -8,14 +8,20 @@ models:
display_name: DeepSeek V4 Pro
description: 1.6T MoE (49B active) with 1M context, hybrid CSA/HCA attention, top-tier reasoning and agentic coding
context_window: 1048576
input_cost_per_million: 1.6
output_cost_per_million: 3.2
- id: moonshotai/kimi-k2.6
display_name: Kimi K2.6
description: 1T-parameter open-weight MoE with native vision/video, multi-step tool calling, and agentic long-horizon execution
attachments: [image]
context_window: 262144
input_cost_per_million: 0.8
output_cost_per_million: 3.4
- id: zai-org/glm-5
display_name: GLM-5
description: Z.AI 754B-parameter MoE with strong general reasoning, function calling, and structured output
context_window: 202800
input_cost_per_million: 1.0
output_cost_per_million: 3.2
+7
View File
@@ -12,9 +12,16 @@ models:
context_window: 1050000
api_flavor: responses
reasoning_effort: medium
input_cost_per_million: 5.0
output_cost_per_million: 30.0
cached_input_cost_per_million: 0.5
- id: gpt-5.4-mini
display_name: GPT-5.4 Mini
description: Cost-efficient GPT-5.4-class model for high-volume coding, computer use, and subagent workloads
input_cost_per_million: 0.75
output_cost_per_million: 4.5
- id: gpt-5.4-nano
display_name: GPT-5.4 Nano
description: Cheapest GPT-5.4-class model, optimized for simple high-volume tasks where speed and cost matter most
input_cost_per_million: 0.2
output_cost_per_million: 1.25
+6
View File
@@ -10,6 +10,8 @@ models:
description: Free-tier 480B MoE coder model with strong agentic tool use; rate-limited
context_window: 262000
attachments: []
input_cost_per_million: 0.0
output_cost_per_million: 0.0
- id: deepseek/deepseek-v3.2
display_name: DeepSeek V3.2
@@ -17,9 +19,13 @@ models:
context_window: 131072
attachments: []
supports_structured_output: true
input_cost_per_million: 0.23
output_cost_per_million: 0.34
- id: anthropic/claude-sonnet-4.6
display_name: Claude Sonnet 4.6 (via OpenRouter)
description: Frontier Sonnet-class model with 1M context, vision, and extended thinking
context_window: 1000000
supports_structured_output: true
input_cost_per_million: 3.0
output_cost_per_million: 15.0
+2
View File
@@ -28,6 +28,7 @@ from docsgpt.core.settings.guardrails import GuardrailSettings
from docsgpt.core.settings.ingestion import IngestionSettings
from docsgpt.core.settings.llm import LLMSettings
from docsgpt.core.settings.ocr import OCRSettings
from docsgpt.core.settings.quotas import QuotaSettings
from docsgpt.core.settings.retrieval import RetrievalSettings
from docsgpt.core.settings.sandbox import SandboxSettings
from docsgpt.core.settings.scheduler import SchedulerSettings
@@ -54,6 +55,7 @@ SETTINGS_GROUPS: tuple[tuple[str, type[SettingsGroup]], ...] = (
("Events and devices", EventsSettings),
("Agents", AgentSettings),
("Guardrails", GuardrailSettings),
("Quotas", QuotaSettings),
("Scheduler", SchedulerSettings),
("Sandbox", SandboxSettings),
("Speech", SpeechSettings),
+36
View File
@@ -0,0 +1,36 @@
"""Admin-set usage quotas and the pricing that feeds their cost budgets."""
from __future__ import annotations
from typing import Literal, Optional
from pydantic import Field, field_validator
from docsgpt.core.settings._shared import SettingsGroup
class QuotaSettings(SettingsGroup):
"""Quota window and the treatment of unpriced models."""
QUOTA_PERIOD: Literal["day", "week", "month"] = Field(
default="month",
description=(
"Window every usage quota is measured over. Windows are calendar-aligned in UTC: "
"a day starts at 00:00, a week on Monday, a month on the 1st."
),
)
QUOTA_UNPRICED_RATE_PER_MILLION: Optional[list[float]] = Field(
default=None,
description=(
"Fallback `[input, output]` USD rates per 1M tokens for models that declare no price, "
"e.g. `[0.5, 1.5]`. Unset, such calls are recorded at $0 and only count toward token quotas."
),
)
@field_validator("QUOTA_UNPRICED_RATE_PER_MILLION")
@classmethod
def _two_non_negative_rates(cls, v: Optional[list[float]]) -> Optional[list[float]]:
if v is None:
return None
if len(v) != 2 or any(rate < 0 for rate in v):
raise ValueError("QUOTA_UNPRICED_RATE_PER_MILLION must be two non-negative numbers")
return v
+103
View File
@@ -0,0 +1,103 @@
"""USD cost of LLM calls, from the per-model rates in the model catalogs."""
from __future__ import annotations
from dataclasses import dataclass
from typing import Optional
from docsgpt.core.settings import settings
@dataclass(frozen=True)
class ModelRates:
"""USD-per-1M rates for one model; ``None`` cache rates bill at the prompt rate."""
prompt: float
generated: float
cached_input: Optional[float] = None
cache_write: Optional[float] = None
def _unpriced_rates() -> Optional[ModelRates]:
"""Return the operator's fallback rates for undeclared models, if configured."""
fallback = settings.QUOTA_UNPRICED_RATE_PER_MILLION
if not fallback:
return None
return ModelRates(prompt=float(fallback[0]), generated=float(fallback[1]))
def resolve_model_rates(model: Optional[str]) -> Optional[ModelRates]:
"""Return the rates for a registry model id.
Args:
model: Canonical registry id (catalog id, or the UUID of a BYOM record).
Returns:
The declared rates, the ``QUOTA_UNPRICED_RATE_PER_MILLION`` fallback when the
model declares none, or ``None`` when there is no fallback either.
"""
# Imported lazily: the registry pulls in the provider plugins, whose LLM
# classes import ``docsgpt.usage`` and, through it, this module.
from docsgpt.core.model_registry import ModelRegistry
entry = ModelRegistry.get_instance().models.get(str(model)) if model else None
if entry is None:
return _unpriced_rates()
caps = entry.capabilities
if caps.input_cost_per_million is None or caps.output_cost_per_million is None:
return _unpriced_rates()
cached = caps.cached_input_cost_per_million
written = caps.cache_write_cost_per_million
return ModelRates(
prompt=float(caps.input_cost_per_million),
generated=float(caps.output_cost_per_million),
cached_input=float(cached) if cached is not None else None,
cache_write=float(written) if written is not None else None,
)
def is_priced(model: Optional[str]) -> bool:
"""Return whether calls to ``model`` are recorded with a cost."""
return resolve_model_rates(model) is not None
def cost_from_rates(
rates: ModelRates,
prompt_tokens: int,
generated_tokens: int,
cached_tokens: Optional[int] = 0,
cache_write_tokens: Optional[int] = 0,
) -> float:
"""Return the USD cost of one call at ``rates``.
``prompt_tokens`` is the provider's billing total; ``cached_tokens`` and
``cache_write_tokens`` are the parts of it read from or written to the prompt
cache. The sub-bins are clamped to the prompt total, so a malformed report can
never price a call below "everything cached".
"""
prompt_total = max(int(prompt_tokens or 0), 0)
cached = min(max(int(cached_tokens or 0), 0), prompt_total)
written = min(max(int(cache_write_tokens or 0), 0), prompt_total - cached)
regular = prompt_total - cached - written
cached_rate = rates.cached_input if rates.cached_input is not None else rates.prompt
write_rate = rates.cache_write if rates.cache_write is not None else rates.prompt
return (
regular * rates.prompt
+ cached * cached_rate
+ written * write_rate
+ max(int(generated_tokens or 0), 0) * rates.generated
) / 1_000_000.0
def compute_cost_usd(
model: Optional[str],
prompt_tokens: int,
generated_tokens: int,
cached_tokens: Optional[int] = 0,
cache_write_tokens: Optional[int] = 0,
) -> float:
"""Return the USD cost of one call to ``model``; ``0.0`` when it has no rates."""
rates = resolve_model_rates(model)
if rates is None:
return 0.0
return cost_from_rates(rates, prompt_tokens, generated_tokens, cached_tokens, cache_write_tokens)
+3 -3
View File
@@ -46,8 +46,8 @@ class TestModelCapabilities:
assert caps.supports_streaming is True
assert caps.supported_attachment_types == []
assert caps.context_window == 128000
assert caps.input_cost_per_token is None
assert caps.output_cost_per_token is None
assert caps.input_cost_per_million is None
assert caps.output_cost_per_million is None
@pytest.mark.unit
def test_custom_values(self):
@@ -55,7 +55,7 @@ class TestModelCapabilities:
supports_tools=True,
supports_structured_output=True,
context_window=32000,
input_cost_per_token=0.001,
input_cost_per_million=1.0,
)
assert caps.supports_tools is True
assert caps.context_window == 32000
+130
View File
@@ -0,0 +1,130 @@
"""Tests for docsgpt/pricing.py and the per-million catalog fields."""
from __future__ import annotations
from types import SimpleNamespace
from unittest.mock import patch
import pytest
from docsgpt import pricing
from docsgpt.core.model_settings import ModelCapabilities
from docsgpt.core.model_yaml import (
BUILTIN_MODELS_DIR,
ModelYAMLError,
load_model_yamls,
)
from docsgpt.pricing import ModelRates, compute_cost_usd, cost_from_rates
def _registry(**models):
entries = {k: SimpleNamespace(capabilities=v) for k, v in models.items()}
return SimpleNamespace(models=entries)
@pytest.fixture
def priced_registry():
caps = ModelCapabilities(
input_cost_per_million=2.0,
output_cost_per_million=10.0,
cached_input_cost_per_million=0.2,
cache_write_cost_per_million=2.5,
)
bare = ModelCapabilities()
with patch(
"docsgpt.core.model_registry.ModelRegistry.get_instance",
return_value=_registry(priced=caps, bare=bare),
):
yield
@pytest.mark.unit
class TestCostFromRates:
def test_prompt_and_generated(self):
rates = ModelRates(prompt=2.0, generated=10.0)
assert cost_from_rates(rates, 1_000_000, 500_000) == pytest.approx(7.0)
def test_cache_bins_use_their_rates(self):
rates = ModelRates(prompt=2.0, generated=10.0, cached_input=0.2, cache_write=2.5)
cost = cost_from_rates(rates, 1000, 0, cached_tokens=600, cache_write_tokens=100)
assert cost == pytest.approx((300 * 2.0 + 600 * 0.2 + 100 * 2.5) / 1e6)
def test_missing_cache_rates_bill_at_prompt_rate(self):
rates = ModelRates(prompt=2.0, generated=10.0)
assert cost_from_rates(rates, 1000, 0, cached_tokens=900) == pytest.approx(1000 * 2.0 / 1e6)
def test_cache_bins_clamped_to_prompt_total(self):
rates = ModelRates(prompt=2.0, generated=0.0, cached_input=0.0, cache_write=0.0)
assert cost_from_rates(rates, 100, 0, cached_tokens=5000, cache_write_tokens=5000) == 0.0
assert cost_from_rates(rates, 100, 0, cached_tokens=-5) == pytest.approx(100 * 2.0 / 1e6)
def test_none_and_negative_counts(self):
rates = ModelRates(prompt=2.0, generated=10.0)
assert cost_from_rates(rates, None, -3, None, None) == 0.0
@pytest.mark.unit
class TestComputeCost:
def test_priced_model(self, priced_registry):
assert compute_cost_usd("priced", 1_000_000, 0) == pytest.approx(2.0)
@pytest.mark.parametrize("model", ["bare", "unknown", None])
def test_unpriced_model_is_free_without_fallback(self, priced_registry, model):
with patch.object(pricing.settings, "QUOTA_UNPRICED_RATE_PER_MILLION", None):
assert compute_cost_usd(model, 1_000_000, 1_000_000) == 0.0
assert pricing.is_priced(model) is False
def test_unpriced_model_uses_fallback(self, priced_registry):
with patch.object(pricing.settings, "QUOTA_UNPRICED_RATE_PER_MILLION", [0.5, 1.5]):
assert compute_cost_usd("bare", 1_000_000, 1_000_000) == pytest.approx(2.0)
assert pricing.is_priced("bare") is True
@pytest.mark.unit
class TestCatalogFields:
def _load(self, tmp_path, body):
(tmp_path / "p.yaml").write_text(body)
return load_model_yamls([tmp_path])[0].models[0].capabilities
def test_per_million_fields(self, tmp_path):
caps = self._load(
tmp_path,
"provider: openai\nmodels:\n - id: m\n input_cost_per_million: 3\n"
" output_cost_per_million: 15\n cached_input_cost_per_million: 0.3\n",
)
assert (caps.input_cost_per_million, caps.output_cost_per_million) == (3, 15)
assert caps.cached_input_cost_per_million == 0.3
assert caps.cache_write_cost_per_million is None
def test_per_token_alias_is_scaled(self, tmp_path):
caps = self._load(
tmp_path,
"provider: openai\ndefaults:\n input_cost_per_token: 0.000003\n"
"models:\n - id: m\n output_cost_per_token: 0.000015\n",
)
assert caps.input_cost_per_million == pytest.approx(3.0)
assert caps.output_cost_per_million == pytest.approx(15.0)
def test_both_spellings_rejected(self, tmp_path):
with pytest.raises(ModelYAMLError):
self._load(
tmp_path,
"provider: openai\nmodels:\n - id: m\n input_cost_per_token: 0.1\n"
" input_cost_per_million: 1\n",
)
def test_negative_rate_rejected(self, tmp_path):
with pytest.raises(ModelYAMLError):
self._load(tmp_path, "provider: openai\nmodels:\n - id: m\n input_cost_per_million: -1\n")
def test_hosted_builtin_models_are_priced(self):
hosted = {"anthropic", "deepseek", "google", "groq", "novita", "openai", "openrouter"}
catalogs = [
c for c in load_model_yamls([BUILTIN_MODELS_DIR]) if c.source_path.stem in hosted
]
assert {c.source_path.stem for c in catalogs} == hosted
for catalog in catalogs:
for model in catalog.models:
caps = model.capabilities
assert caps.input_cost_per_million is not None, model.id
assert caps.output_cost_per_million is not None, model.id