mirror of
https://github.com/tiennm99/DocsGPT.git
synced 2026-10-11 03:12:55 +00:00
feat(pricing): per-million model rates and a cost module
Rename the unused *_cost_per_token capability fields to USD per 1M tokens, add prompt-cache read/write rates, and ship list prices for the hosted catalogs. The old per-token keys still load, scaled, with a warning. docsgpt/pricing.py turns a call's token bins into a USD cost. Models with no declared rate cost $0 unless QUOTA_UNPRICED_RATE_PER_MILLION is set.
This commit is contained in:
1 parent
1c1bc2538f
commit
b77561288d
16 files changed
+375
-12
No files matched your search
@@ -1451,6 +1451,23 @@ Type `int`, default `30`, must be `>= 1`.
|
||||
Days guardrail events are kept before the cleanup task removes them.
|
||||
|
||||
|
||||
## Quotas
|
||||
|
||||
Quota window and the treatment of unpriced models.
|
||||
|
||||
### `QUOTA_PERIOD`
|
||||
|
||||
Type `"day" | "week" | "month"`, default `month`.
|
||||
|
||||
Window every usage quota is measured over. Windows are calendar-aligned in UTC: a day starts at 00:00, a week on Monday, a month on the 1st.
|
||||
|
||||
### `QUOTA_UNPRICED_RATE_PER_MILLION`
|
||||
|
||||
Type `list[float]`, default unset.
|
||||
|
||||
Fallback `[input, output]` USD rates per 1M tokens for models that declare no price, e.g. `[0.5, 1.5]`. Unset, such calls are recorded at $0 and only count toward token quotas.
|
||||
|
||||
|
||||
## Scheduler
|
||||
|
||||
Cadence, quotas and timeouts of scheduled runs.
|
||||
|
||||
@@ -32,8 +32,15 @@ class ModelCapabilities:
|
||||
supports_streaming: bool = True
|
||||
supported_attachment_types: List[str] = field(default_factory=list)
|
||||
context_window: int = 128000
|
||||
input_cost_per_token: Optional[float] = None
|
||||
output_cost_per_token: Optional[float] = None
|
||||
# USD per 1M tokens; consumed by ``docsgpt/pricing.py``. ``None`` means
|
||||
# "not declared": the call is recorded at $0 unless
|
||||
# ``QUOTA_UNPRICED_RATE_PER_MILLION`` is set.
|
||||
input_cost_per_million: Optional[float] = None
|
||||
output_cost_per_million: Optional[float] = None
|
||||
# Rates for the prompt-cache sub-bins of the prompt total. ``None`` bills
|
||||
# those tokens at ``input_cost_per_million``.
|
||||
cached_input_cost_per_million: Optional[float] = None
|
||||
cache_write_cost_per_million: Optional[float] = None
|
||||
# OpenAI reasoning-model effort hint (none/minimal/low/medium/high/xhigh;
|
||||
# the accepted subset is model-dependent). Consumed by OpenAILLM — sent
|
||||
# top-level on Chat Completions and nested under ``reasoning`` on the
|
||||
|
||||
@@ -18,7 +18,7 @@ from pathlib import Path
|
||||
from typing import Dict, List, Optional, Sequence
|
||||
|
||||
import yaml
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator, model_validator
|
||||
|
||||
from docsgpt.core.model_settings import (
|
||||
AvailableModel,
|
||||
@@ -65,11 +65,33 @@ class _CapabilityFields(BaseModel):
|
||||
supports_streaming: Optional[bool] = None
|
||||
attachments: Optional[List[str]] = None
|
||||
context_window: Optional[int] = None
|
||||
input_cost_per_token: Optional[float] = None
|
||||
output_cost_per_token: Optional[float] = None
|
||||
input_cost_per_million: Optional[float] = Field(default=None, ge=0)
|
||||
output_cost_per_million: Optional[float] = Field(default=None, ge=0)
|
||||
cached_input_cost_per_million: Optional[float] = Field(default=None, ge=0)
|
||||
cache_write_cost_per_million: Optional[float] = Field(default=None, ge=0)
|
||||
reasoning_effort: Optional[str] = None
|
||||
api_flavor: Optional[str] = None
|
||||
|
||||
@model_validator(mode="before")
|
||||
@classmethod
|
||||
def _per_token_alias(cls, data):
|
||||
"""Accept the deprecated ``*_cost_per_token`` keys, scaled to per-1M."""
|
||||
if not isinstance(data, dict):
|
||||
return data
|
||||
data = dict(data)
|
||||
for side in ("input", "output"):
|
||||
old, new = f"{side}_cost_per_token", f"{side}_cost_per_million"
|
||||
if old not in data:
|
||||
continue
|
||||
value = data.pop(old)
|
||||
if new in data:
|
||||
raise ValueError(f"set only one of {old} and {new}")
|
||||
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
||||
raise ValueError(f"{old} must be a number")
|
||||
logger.warning("%s is deprecated; use %s (USD per 1M tokens)", old, new)
|
||||
data[new] = value * 1_000_000
|
||||
return data
|
||||
|
||||
@field_validator("reasoning_effort")
|
||||
@classmethod
|
||||
def _valid_reasoning_effort(cls, v: Optional[str]) -> Optional[str]:
|
||||
@@ -237,8 +259,10 @@ def _build_model(
|
||||
supports_streaming=pick("supports_streaming", True),
|
||||
supported_attachment_types=expanded,
|
||||
context_window=pick("context_window", 128000),
|
||||
input_cost_per_token=pick("input_cost_per_token", None),
|
||||
output_cost_per_token=pick("output_cost_per_token", None),
|
||||
input_cost_per_million=pick("input_cost_per_million", None),
|
||||
output_cost_per_million=pick("output_cost_per_million", None),
|
||||
cached_input_cost_per_million=pick("cached_input_cost_per_million", None),
|
||||
cache_write_cost_per_million=pick("cache_write_cost_per_million", None),
|
||||
reasoning_effort=pick("reasoning_effort", None),
|
||||
api_flavor=pick("api_flavor", "chat_completions"),
|
||||
)
|
||||
|
||||
@@ -108,8 +108,10 @@ defaults: # optional, applied to every model below
|
||||
supports_streaming: bool # default true
|
||||
attachments: [<alias-or-mime>, ...] # default []
|
||||
context_window: int # default 128000
|
||||
input_cost_per_token: float # default null
|
||||
output_cost_per_token: float # default null
|
||||
input_cost_per_million: float # USD per 1M prompt tokens; default null (unpriced)
|
||||
output_cost_per_million: float # USD per 1M generated tokens; default null
|
||||
cached_input_cost_per_million: float # prompt-cache reads; default: the input rate
|
||||
cache_write_cost_per_million: float # prompt-cache writes; default: the input rate
|
||||
reasoning_effort: <string> # default null; none|minimal|low|medium|high|xhigh (subset is model-dependent)
|
||||
api_flavor: <string> # chat_completions (default) or responses
|
||||
|
||||
|
||||
@@ -10,14 +10,20 @@ models:
|
||||
description: Most capable Claude model for complex reasoning and agentic coding
|
||||
context_window: 1000000
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 5.0
|
||||
output_cost_per_million: 25.0
|
||||
|
||||
- id: claude-sonnet-4-6
|
||||
display_name: Claude Sonnet 4.6
|
||||
description: Best balance of speed and intelligence with extended thinking
|
||||
context_window: 1000000
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 3.0
|
||||
output_cost_per_million: 15.0
|
||||
|
||||
- id: claude-haiku-4-5
|
||||
display_name: Claude Haiku 4.5
|
||||
description: Fastest Claude model with near-frontier intelligence
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 1.0
|
||||
output_cost_per_million: 5.0
|
||||
@@ -12,7 +12,11 @@ models:
|
||||
- id: deepseek-v4-flash
|
||||
display_name: DeepSeek V4 Flash
|
||||
description: Cost-efficient 1M-context model with hybrid thinking / non-thinking modes, tool calling and FIM completion
|
||||
input_cost_per_million: 0.14
|
||||
output_cost_per_million: 0.28
|
||||
|
||||
- id: deepseek-v4-pro
|
||||
display_name: DeepSeek V4 Pro
|
||||
description: Frontier 1M-context model with hybrid thinking / non-thinking modes for advanced reasoning and agentic coding
|
||||
input_cost_per_million: 0.435
|
||||
output_cost_per_million: 0.87
|
||||
@@ -9,9 +9,16 @@ models:
|
||||
- id: gemini-3.1-pro-preview
|
||||
display_name: Gemini 3.1 Pro (preview)
|
||||
description: Most capable Gemini 3 model with advanced reasoning and agentic coding (preview)
|
||||
# Priced at the >200k-token tier; long prompts are common with attachments.
|
||||
input_cost_per_million: 4.0
|
||||
output_cost_per_million: 18.0
|
||||
- id: gemini-3.5-flash
|
||||
display_name: Gemini 3.5 Flash
|
||||
description: Frontier-class Flash for sustained performance on agentic and coding tasks
|
||||
input_cost_per_million: 1.5
|
||||
output_cost_per_million: 9
|
||||
- id: gemini-3.1-flash-lite
|
||||
display_name: Gemini 3.1 Flash-Lite
|
||||
description: Cost-efficient frontier-class multimodal model for high-throughput workloads
|
||||
input_cost_per_million: 0.25
|
||||
output_cost_per_million: 1.5
|
||||
@@ -8,9 +8,15 @@ models:
|
||||
display_name: GPT-OSS 120B
|
||||
description: OpenAI's open-weight 120B flagship served on Groq's LPU hardware; strong general reasoning with strict structured output support
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 0.15
|
||||
output_cost_per_million: 0.6
|
||||
- id: llama-3.3-70b-versatile
|
||||
display_name: Llama 3.3 70B Versatile
|
||||
description: Meta's Llama 3.3 70B for general-purpose chat with parallel tool use
|
||||
input_cost_per_million: 0.59
|
||||
output_cost_per_million: 0.79
|
||||
- id: llama-3.1-8b-instant
|
||||
display_name: Llama 3.1 8B Instant
|
||||
description: Small, very low-latency Llama model (~560 tok/s) with parallel tool use
|
||||
input_cost_per_million: 0.05
|
||||
output_cost_per_million: 0.08
|
||||
@@ -8,14 +8,20 @@ models:
|
||||
display_name: DeepSeek V4 Pro
|
||||
description: 1.6T MoE (49B active) with 1M context, hybrid CSA/HCA attention, top-tier reasoning and agentic coding
|
||||
context_window: 1048576
|
||||
input_cost_per_million: 1.6
|
||||
output_cost_per_million: 3.2
|
||||
|
||||
- id: moonshotai/kimi-k2.6
|
||||
display_name: Kimi K2.6
|
||||
description: 1T-parameter open-weight MoE with native vision/video, multi-step tool calling, and agentic long-horizon execution
|
||||
attachments: [image]
|
||||
context_window: 262144
|
||||
input_cost_per_million: 0.8
|
||||
output_cost_per_million: 3.4
|
||||
|
||||
- id: zai-org/glm-5
|
||||
display_name: GLM-5
|
||||
description: Z.AI 754B-parameter MoE with strong general reasoning, function calling, and structured output
|
||||
context_window: 202800
|
||||
input_cost_per_million: 1.0
|
||||
output_cost_per_million: 3.2
|
||||
@@ -12,9 +12,16 @@ models:
|
||||
context_window: 1050000
|
||||
api_flavor: responses
|
||||
reasoning_effort: medium
|
||||
input_cost_per_million: 5.0
|
||||
output_cost_per_million: 30.0
|
||||
cached_input_cost_per_million: 0.5
|
||||
- id: gpt-5.4-mini
|
||||
display_name: GPT-5.4 Mini
|
||||
description: Cost-efficient GPT-5.4-class model for high-volume coding, computer use, and subagent workloads
|
||||
input_cost_per_million: 0.75
|
||||
output_cost_per_million: 4.5
|
||||
- id: gpt-5.4-nano
|
||||
display_name: GPT-5.4 Nano
|
||||
description: Cheapest GPT-5.4-class model, optimized for simple high-volume tasks where speed and cost matter most
|
||||
input_cost_per_million: 0.2
|
||||
output_cost_per_million: 1.25
|
||||
@@ -10,6 +10,8 @@ models:
|
||||
description: Free-tier 480B MoE coder model with strong agentic tool use; rate-limited
|
||||
context_window: 262000
|
||||
attachments: []
|
||||
input_cost_per_million: 0.0
|
||||
output_cost_per_million: 0.0
|
||||
|
||||
- id: deepseek/deepseek-v3.2
|
||||
display_name: DeepSeek V3.2
|
||||
@@ -17,9 +19,13 @@ models:
|
||||
context_window: 131072
|
||||
attachments: []
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 0.23
|
||||
output_cost_per_million: 0.34
|
||||
|
||||
- id: anthropic/claude-sonnet-4.6
|
||||
display_name: Claude Sonnet 4.6 (via OpenRouter)
|
||||
description: Frontier Sonnet-class model with 1M context, vision, and extended thinking
|
||||
context_window: 1000000
|
||||
supports_structured_output: true
|
||||
input_cost_per_million: 3.0
|
||||
output_cost_per_million: 15.0
|
||||
@@ -28,6 +28,7 @@ from docsgpt.core.settings.guardrails import GuardrailSettings
|
||||
from docsgpt.core.settings.ingestion import IngestionSettings
|
||||
from docsgpt.core.settings.llm import LLMSettings
|
||||
from docsgpt.core.settings.ocr import OCRSettings
|
||||
from docsgpt.core.settings.quotas import QuotaSettings
|
||||
from docsgpt.core.settings.retrieval import RetrievalSettings
|
||||
from docsgpt.core.settings.sandbox import SandboxSettings
|
||||
from docsgpt.core.settings.scheduler import SchedulerSettings
|
||||
@@ -54,6 +55,7 @@ SETTINGS_GROUPS: tuple[tuple[str, type[SettingsGroup]], ...] = (
|
||||
("Events and devices", EventsSettings),
|
||||
("Agents", AgentSettings),
|
||||
("Guardrails", GuardrailSettings),
|
||||
("Quotas", QuotaSettings),
|
||||
("Scheduler", SchedulerSettings),
|
||||
("Sandbox", SandboxSettings),
|
||||
("Speech", SpeechSettings),
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
"""Admin-set usage quotas and the pricing that feeds their cost budgets."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Literal, Optional
|
||||
|
||||
from pydantic import Field, field_validator
|
||||
|
||||
from docsgpt.core.settings._shared import SettingsGroup
|
||||
|
||||
|
||||
class QuotaSettings(SettingsGroup):
|
||||
"""Quota window and the treatment of unpriced models."""
|
||||
|
||||
QUOTA_PERIOD: Literal["day", "week", "month"] = Field(
|
||||
default="month",
|
||||
description=(
|
||||
"Window every usage quota is measured over. Windows are calendar-aligned in UTC: "
|
||||
"a day starts at 00:00, a week on Monday, a month on the 1st."
|
||||
),
|
||||
)
|
||||
QUOTA_UNPRICED_RATE_PER_MILLION: Optional[list[float]] = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Fallback `[input, output]` USD rates per 1M tokens for models that declare no price, "
|
||||
"e.g. `[0.5, 1.5]`. Unset, such calls are recorded at $0 and only count toward token quotas."
|
||||
),
|
||||
)
|
||||
@field_validator("QUOTA_UNPRICED_RATE_PER_MILLION")
|
||||
@classmethod
|
||||
def _two_non_negative_rates(cls, v: Optional[list[float]]) -> Optional[list[float]]:
|
||||
if v is None:
|
||||
return None
|
||||
if len(v) != 2 or any(rate < 0 for rate in v):
|
||||
raise ValueError("QUOTA_UNPRICED_RATE_PER_MILLION must be two non-negative numbers")
|
||||
return v
|
||||
@@ -0,0 +1,103 @@
|
||||
"""USD cost of LLM calls, from the per-model rates in the model catalogs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Optional
|
||||
|
||||
from docsgpt.core.settings import settings
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ModelRates:
|
||||
"""USD-per-1M rates for one model; ``None`` cache rates bill at the prompt rate."""
|
||||
|
||||
prompt: float
|
||||
generated: float
|
||||
cached_input: Optional[float] = None
|
||||
cache_write: Optional[float] = None
|
||||
|
||||
|
||||
def _unpriced_rates() -> Optional[ModelRates]:
|
||||
"""Return the operator's fallback rates for undeclared models, if configured."""
|
||||
fallback = settings.QUOTA_UNPRICED_RATE_PER_MILLION
|
||||
if not fallback:
|
||||
return None
|
||||
return ModelRates(prompt=float(fallback[0]), generated=float(fallback[1]))
|
||||
|
||||
|
||||
def resolve_model_rates(model: Optional[str]) -> Optional[ModelRates]:
|
||||
"""Return the rates for a registry model id.
|
||||
|
||||
Args:
|
||||
model: Canonical registry id (catalog id, or the UUID of a BYOM record).
|
||||
|
||||
Returns:
|
||||
The declared rates, the ``QUOTA_UNPRICED_RATE_PER_MILLION`` fallback when the
|
||||
model declares none, or ``None`` when there is no fallback either.
|
||||
"""
|
||||
# Imported lazily: the registry pulls in the provider plugins, whose LLM
|
||||
# classes import ``docsgpt.usage`` and, through it, this module.
|
||||
from docsgpt.core.model_registry import ModelRegistry
|
||||
|
||||
entry = ModelRegistry.get_instance().models.get(str(model)) if model else None
|
||||
if entry is None:
|
||||
return _unpriced_rates()
|
||||
caps = entry.capabilities
|
||||
if caps.input_cost_per_million is None or caps.output_cost_per_million is None:
|
||||
return _unpriced_rates()
|
||||
cached = caps.cached_input_cost_per_million
|
||||
written = caps.cache_write_cost_per_million
|
||||
return ModelRates(
|
||||
prompt=float(caps.input_cost_per_million),
|
||||
generated=float(caps.output_cost_per_million),
|
||||
cached_input=float(cached) if cached is not None else None,
|
||||
cache_write=float(written) if written is not None else None,
|
||||
)
|
||||
|
||||
|
||||
def is_priced(model: Optional[str]) -> bool:
|
||||
"""Return whether calls to ``model`` are recorded with a cost."""
|
||||
return resolve_model_rates(model) is not None
|
||||
|
||||
|
||||
def cost_from_rates(
|
||||
rates: ModelRates,
|
||||
prompt_tokens: int,
|
||||
generated_tokens: int,
|
||||
cached_tokens: Optional[int] = 0,
|
||||
cache_write_tokens: Optional[int] = 0,
|
||||
) -> float:
|
||||
"""Return the USD cost of one call at ``rates``.
|
||||
|
||||
``prompt_tokens`` is the provider's billing total; ``cached_tokens`` and
|
||||
``cache_write_tokens`` are the parts of it read from or written to the prompt
|
||||
cache. The sub-bins are clamped to the prompt total, so a malformed report can
|
||||
never price a call below "everything cached".
|
||||
"""
|
||||
prompt_total = max(int(prompt_tokens or 0), 0)
|
||||
cached = min(max(int(cached_tokens or 0), 0), prompt_total)
|
||||
written = min(max(int(cache_write_tokens or 0), 0), prompt_total - cached)
|
||||
regular = prompt_total - cached - written
|
||||
cached_rate = rates.cached_input if rates.cached_input is not None else rates.prompt
|
||||
write_rate = rates.cache_write if rates.cache_write is not None else rates.prompt
|
||||
return (
|
||||
regular * rates.prompt
|
||||
+ cached * cached_rate
|
||||
+ written * write_rate
|
||||
+ max(int(generated_tokens or 0), 0) * rates.generated
|
||||
) / 1_000_000.0
|
||||
|
||||
|
||||
def compute_cost_usd(
|
||||
model: Optional[str],
|
||||
prompt_tokens: int,
|
||||
generated_tokens: int,
|
||||
cached_tokens: Optional[int] = 0,
|
||||
cache_write_tokens: Optional[int] = 0,
|
||||
) -> float:
|
||||
"""Return the USD cost of one call to ``model``; ``0.0`` when it has no rates."""
|
||||
rates = resolve_model_rates(model)
|
||||
if rates is None:
|
||||
return 0.0
|
||||
return cost_from_rates(rates, prompt_tokens, generated_tokens, cached_tokens, cache_write_tokens)
|
||||
@@ -46,8 +46,8 @@ class TestModelCapabilities:
|
||||
assert caps.supports_streaming is True
|
||||
assert caps.supported_attachment_types == []
|
||||
assert caps.context_window == 128000
|
||||
assert caps.input_cost_per_token is None
|
||||
assert caps.output_cost_per_token is None
|
||||
assert caps.input_cost_per_million is None
|
||||
assert caps.output_cost_per_million is None
|
||||
|
||||
@pytest.mark.unit
|
||||
def test_custom_values(self):
|
||||
@@ -55,7 +55,7 @@ class TestModelCapabilities:
|
||||
supports_tools=True,
|
||||
supports_structured_output=True,
|
||||
context_window=32000,
|
||||
input_cost_per_token=0.001,
|
||||
input_cost_per_million=1.0,
|
||||
)
|
||||
assert caps.supports_tools is True
|
||||
assert caps.context_window == 32000
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
"""Tests for docsgpt/pricing.py and the per-million catalog fields."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from docsgpt import pricing
|
||||
from docsgpt.core.model_settings import ModelCapabilities
|
||||
from docsgpt.core.model_yaml import (
|
||||
BUILTIN_MODELS_DIR,
|
||||
ModelYAMLError,
|
||||
load_model_yamls,
|
||||
)
|
||||
from docsgpt.pricing import ModelRates, compute_cost_usd, cost_from_rates
|
||||
|
||||
|
||||
def _registry(**models):
|
||||
entries = {k: SimpleNamespace(capabilities=v) for k, v in models.items()}
|
||||
return SimpleNamespace(models=entries)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def priced_registry():
|
||||
caps = ModelCapabilities(
|
||||
input_cost_per_million=2.0,
|
||||
output_cost_per_million=10.0,
|
||||
cached_input_cost_per_million=0.2,
|
||||
cache_write_cost_per_million=2.5,
|
||||
)
|
||||
bare = ModelCapabilities()
|
||||
with patch(
|
||||
"docsgpt.core.model_registry.ModelRegistry.get_instance",
|
||||
return_value=_registry(priced=caps, bare=bare),
|
||||
):
|
||||
yield
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestCostFromRates:
|
||||
def test_prompt_and_generated(self):
|
||||
rates = ModelRates(prompt=2.0, generated=10.0)
|
||||
assert cost_from_rates(rates, 1_000_000, 500_000) == pytest.approx(7.0)
|
||||
|
||||
def test_cache_bins_use_their_rates(self):
|
||||
rates = ModelRates(prompt=2.0, generated=10.0, cached_input=0.2, cache_write=2.5)
|
||||
cost = cost_from_rates(rates, 1000, 0, cached_tokens=600, cache_write_tokens=100)
|
||||
assert cost == pytest.approx((300 * 2.0 + 600 * 0.2 + 100 * 2.5) / 1e6)
|
||||
|
||||
def test_missing_cache_rates_bill_at_prompt_rate(self):
|
||||
rates = ModelRates(prompt=2.0, generated=10.0)
|
||||
assert cost_from_rates(rates, 1000, 0, cached_tokens=900) == pytest.approx(1000 * 2.0 / 1e6)
|
||||
|
||||
def test_cache_bins_clamped_to_prompt_total(self):
|
||||
rates = ModelRates(prompt=2.0, generated=0.0, cached_input=0.0, cache_write=0.0)
|
||||
assert cost_from_rates(rates, 100, 0, cached_tokens=5000, cache_write_tokens=5000) == 0.0
|
||||
assert cost_from_rates(rates, 100, 0, cached_tokens=-5) == pytest.approx(100 * 2.0 / 1e6)
|
||||
|
||||
def test_none_and_negative_counts(self):
|
||||
rates = ModelRates(prompt=2.0, generated=10.0)
|
||||
assert cost_from_rates(rates, None, -3, None, None) == 0.0
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestComputeCost:
|
||||
def test_priced_model(self, priced_registry):
|
||||
assert compute_cost_usd("priced", 1_000_000, 0) == pytest.approx(2.0)
|
||||
|
||||
@pytest.mark.parametrize("model", ["bare", "unknown", None])
|
||||
def test_unpriced_model_is_free_without_fallback(self, priced_registry, model):
|
||||
with patch.object(pricing.settings, "QUOTA_UNPRICED_RATE_PER_MILLION", None):
|
||||
assert compute_cost_usd(model, 1_000_000, 1_000_000) == 0.0
|
||||
assert pricing.is_priced(model) is False
|
||||
|
||||
def test_unpriced_model_uses_fallback(self, priced_registry):
|
||||
with patch.object(pricing.settings, "QUOTA_UNPRICED_RATE_PER_MILLION", [0.5, 1.5]):
|
||||
assert compute_cost_usd("bare", 1_000_000, 1_000_000) == pytest.approx(2.0)
|
||||
assert pricing.is_priced("bare") is True
|
||||
|
||||
|
||||
@pytest.mark.unit
|
||||
class TestCatalogFields:
|
||||
def _load(self, tmp_path, body):
|
||||
(tmp_path / "p.yaml").write_text(body)
|
||||
return load_model_yamls([tmp_path])[0].models[0].capabilities
|
||||
|
||||
def test_per_million_fields(self, tmp_path):
|
||||
caps = self._load(
|
||||
tmp_path,
|
||||
"provider: openai\nmodels:\n - id: m\n input_cost_per_million: 3\n"
|
||||
" output_cost_per_million: 15\n cached_input_cost_per_million: 0.3\n",
|
||||
)
|
||||
assert (caps.input_cost_per_million, caps.output_cost_per_million) == (3, 15)
|
||||
assert caps.cached_input_cost_per_million == 0.3
|
||||
assert caps.cache_write_cost_per_million is None
|
||||
|
||||
def test_per_token_alias_is_scaled(self, tmp_path):
|
||||
caps = self._load(
|
||||
tmp_path,
|
||||
"provider: openai\ndefaults:\n input_cost_per_token: 0.000003\n"
|
||||
"models:\n - id: m\n output_cost_per_token: 0.000015\n",
|
||||
)
|
||||
assert caps.input_cost_per_million == pytest.approx(3.0)
|
||||
assert caps.output_cost_per_million == pytest.approx(15.0)
|
||||
|
||||
def test_both_spellings_rejected(self, tmp_path):
|
||||
with pytest.raises(ModelYAMLError):
|
||||
self._load(
|
||||
tmp_path,
|
||||
"provider: openai\nmodels:\n - id: m\n input_cost_per_token: 0.1\n"
|
||||
" input_cost_per_million: 1\n",
|
||||
)
|
||||
|
||||
def test_negative_rate_rejected(self, tmp_path):
|
||||
with pytest.raises(ModelYAMLError):
|
||||
self._load(tmp_path, "provider: openai\nmodels:\n - id: m\n input_cost_per_million: -1\n")
|
||||
|
||||
def test_hosted_builtin_models_are_priced(self):
|
||||
hosted = {"anthropic", "deepseek", "google", "groq", "novita", "openai", "openrouter"}
|
||||
catalogs = [
|
||||
c for c in load_model_yamls([BUILTIN_MODELS_DIR]) if c.source_path.stem in hosted
|
||||
]
|
||||
assert {c.source_path.stem for c in catalogs} == hosted
|
||||
for catalog in catalogs:
|
||||
for model in catalog.models:
|
||||
caps = model.capabilities
|
||||
assert caps.input_cost_per_million is not None, model.id
|
||||
assert caps.output_cost_per_million is not None, model.id
|
||||
Reference in new issue
Block a user