Files
Duc Nguyen 747b58d24c fix(usage): repair cost analytics and display precision (#1330)
fix(usage): repair cost analytics and display precision

- Automatic OpenRouter pricing sync with cost backfill for traces/snapshots/events
- Live usage data merging for current-hour dashboard accuracy
- 2-decimal API cost formatting across usage/overview pages
- Comprehensive test coverage (PG, SQLite, HTTP, UI)

Merged by github-maintain automation.
2026-07-03 06:24:21 +07:00

82 lines
3.2 KiB
Go

package tracing
import (
"github.com/nextlevelbuilder/goclaw/internal/config"
"github.com/nextlevelbuilder/goclaw/internal/providers"
"github.com/nextlevelbuilder/goclaw/internal/store"
usagepricing "github.com/nextlevelbuilder/goclaw/internal/usage/pricing"
)
// CalculateCost computes the USD cost for a single LLM call based on token usage and pricing.
// Returns 0 if pricing is nil.
//
// Semantics for reasoning/thinking tokens:
//
// All supported providers (OpenAI o3/o4-mini, Codex/GPT-5 Responses API, Anthropic
// extended thinking) report Usage.ThinkingTokens as a SUB-COUNT of Usage.CompletionTokens:
// - OpenAI: completion_tokens includes reasoning; completion_tokens_details.reasoning_tokens is the breakdown.
// - Codex: output_tokens includes reasoning; output_tokens_details.reasoning_tokens is the breakdown.
// - Anthropic: output_tokens is total output; we estimate thinking tokens from streamed thinking_delta chars.
//
// Thus CompletionTokens already bills thinking at the output rate by default. Only
// when an explicit ReasoningPerMillion rate is configured do we split the two and
// price the thinking portion separately — otherwise we'd double-count.
func CalculateCost(pricing *config.ModelPricing, usage *providers.Usage) float64 {
if pricing == nil || usage == nil {
return 0
}
cost := float64(usage.PromptTokens) * pricing.InputPerMillion / 1_000_000
// Split completion tokens into visible output + thinking only when a distinct
// ReasoningPerMillion rate is set. Otherwise price the full CompletionTokens
// at OutputPerMillion — matches the provider billing semantics described above.
if pricing.ReasoningPerMillion > 0 && usage.ThinkingTokens > 0 {
visible := max(usage.CompletionTokens-usage.ThinkingTokens,
// Defensive: thinkingChars/4 estimate for Anthropic may exceed OutputTokens
// under unusual streaming conditions. Clamp to zero instead of going negative.
0)
cost += float64(visible) * pricing.OutputPerMillion / 1_000_000
cost += float64(usage.ThinkingTokens) * pricing.ReasoningPerMillion / 1_000_000
} else {
cost += float64(usage.CompletionTokens) * pricing.OutputPerMillion / 1_000_000
}
if pricing.CacheReadPerMillion > 0 && usage.CacheReadTokens > 0 {
cost += float64(usage.CacheReadTokens) * pricing.CacheReadPerMillion / 1_000_000
}
if pricing.CacheCreatePerMillion > 0 && usage.CacheCreationTokens > 0 {
cost += float64(usage.CacheCreationTokens) * pricing.CacheCreatePerMillion / 1_000_000
}
return cost
}
func CalculateCostFromUsagePricing(fields store.UsagePricingFields, usage *providers.Usage) (float64, error) {
if usage == nil {
return 0, nil
}
billable := usagepricing.FromProviderUsage(usage)
if fields.Request == nil {
billable.RequestCount = 0
}
micros, err := usagepricing.CostMicros(fields, billable)
if err != nil {
return 0, err
}
return float64(micros) / 1_000_000, nil
}
// LookupPricing finds the model pricing from config.
// Tries "provider/model" first, then just "model".
func LookupPricing(pricingMap map[string]*config.ModelPricing, provider, model string) *config.ModelPricing {
if pricingMap == nil {
return nil
}
if p, ok := pricingMap[provider+"/"+model]; ok {
return p
}
if p, ok := pricingMap[model]; ok {
return p
}
return nil
}