mirror of
https://github.com/tiennm99/goclaw.git
synced 2026-10-11 03:13:24 +00:00
fix(usage): repair cost analytics and display precision - Automatic OpenRouter pricing sync with cost backfill for traces/snapshots/events - Live usage data merging for current-hour dashboard accuracy - 2-decimal API cost formatting across usage/overview pages - Comprehensive test coverage (PG, SQLite, HTTP, UI) Merged by github-maintain automation.
82 lines
3.2 KiB
Go
82 lines
3.2 KiB
Go
package tracing
|
|
|
|
import (
|
|
"github.com/nextlevelbuilder/goclaw/internal/config"
|
|
"github.com/nextlevelbuilder/goclaw/internal/providers"
|
|
"github.com/nextlevelbuilder/goclaw/internal/store"
|
|
usagepricing "github.com/nextlevelbuilder/goclaw/internal/usage/pricing"
|
|
)
|
|
|
|
// CalculateCost computes the USD cost for a single LLM call based on token usage and pricing.
|
|
// Returns 0 if pricing is nil.
|
|
//
|
|
// Semantics for reasoning/thinking tokens:
|
|
//
|
|
// All supported providers (OpenAI o3/o4-mini, Codex/GPT-5 Responses API, Anthropic
|
|
// extended thinking) report Usage.ThinkingTokens as a SUB-COUNT of Usage.CompletionTokens:
|
|
// - OpenAI: completion_tokens includes reasoning; completion_tokens_details.reasoning_tokens is the breakdown.
|
|
// - Codex: output_tokens includes reasoning; output_tokens_details.reasoning_tokens is the breakdown.
|
|
// - Anthropic: output_tokens is total output; we estimate thinking tokens from streamed thinking_delta chars.
|
|
//
|
|
// Thus CompletionTokens already bills thinking at the output rate by default. Only
|
|
// when an explicit ReasoningPerMillion rate is configured do we split the two and
|
|
// price the thinking portion separately — otherwise we'd double-count.
|
|
func CalculateCost(pricing *config.ModelPricing, usage *providers.Usage) float64 {
|
|
if pricing == nil || usage == nil {
|
|
return 0
|
|
}
|
|
cost := float64(usage.PromptTokens) * pricing.InputPerMillion / 1_000_000
|
|
|
|
// Split completion tokens into visible output + thinking only when a distinct
|
|
// ReasoningPerMillion rate is set. Otherwise price the full CompletionTokens
|
|
// at OutputPerMillion — matches the provider billing semantics described above.
|
|
if pricing.ReasoningPerMillion > 0 && usage.ThinkingTokens > 0 {
|
|
visible := max(usage.CompletionTokens-usage.ThinkingTokens,
|
|
// Defensive: thinkingChars/4 estimate for Anthropic may exceed OutputTokens
|
|
// under unusual streaming conditions. Clamp to zero instead of going negative.
|
|
0)
|
|
cost += float64(visible) * pricing.OutputPerMillion / 1_000_000
|
|
cost += float64(usage.ThinkingTokens) * pricing.ReasoningPerMillion / 1_000_000
|
|
} else {
|
|
cost += float64(usage.CompletionTokens) * pricing.OutputPerMillion / 1_000_000
|
|
}
|
|
|
|
if pricing.CacheReadPerMillion > 0 && usage.CacheReadTokens > 0 {
|
|
cost += float64(usage.CacheReadTokens) * pricing.CacheReadPerMillion / 1_000_000
|
|
}
|
|
if pricing.CacheCreatePerMillion > 0 && usage.CacheCreationTokens > 0 {
|
|
cost += float64(usage.CacheCreationTokens) * pricing.CacheCreatePerMillion / 1_000_000
|
|
}
|
|
return cost
|
|
}
|
|
|
|
func CalculateCostFromUsagePricing(fields store.UsagePricingFields, usage *providers.Usage) (float64, error) {
|
|
if usage == nil {
|
|
return 0, nil
|
|
}
|
|
billable := usagepricing.FromProviderUsage(usage)
|
|
if fields.Request == nil {
|
|
billable.RequestCount = 0
|
|
}
|
|
micros, err := usagepricing.CostMicros(fields, billable)
|
|
if err != nil {
|
|
return 0, err
|
|
}
|
|
return float64(micros) / 1_000_000, nil
|
|
}
|
|
|
|
// LookupPricing finds the model pricing from config.
|
|
// Tries "provider/model" first, then just "model".
|
|
func LookupPricing(pricingMap map[string]*config.ModelPricing, provider, model string) *config.ModelPricing {
|
|
if pricingMap == nil {
|
|
return nil
|
|
}
|
|
if p, ok := pricingMap[provider+"/"+model]; ok {
|
|
return p
|
|
}
|
|
if p, ok := pricingMap[model]; ok {
|
|
return p
|
|
}
|
|
return nil
|
|
}
|