Fix cache token accounting and tiered pricing

This commit is contained in:
elky
2026-07-10 15:13:12 +08:00
parent 736fc76345
commit 4bf5d4c044
19 changed files with 506 additions and 264 deletions
+43
View File
@@ -133,6 +133,7 @@ fn build_dimensions(
let normalized_input_tokens = normalize_input_tokens_for_billing(
input.api_format.as_deref(),
input.input_tokens,
input.cache_creation_tokens,
input.cache_read_tokens,
);
let classified_cache_creation_tokens = input
@@ -687,6 +688,48 @@ mod tests {
);
}
#[test]
fn openai_cache_write_and_read_are_billed_separately() {
let result = BillingService::new()
.calculate(
&pricing(),
&BillingUsageInput {
task_type: "chat".to_string(),
api_format: Some("openai:responses".to_string()),
request_count: 1,
input_tokens: 1_000,
output_tokens: 10,
cache_creation_tokens: 100,
cache_creation_ephemeral_5m_tokens: 0,
cache_creation_ephemeral_1h_tokens: 0,
cache_read_tokens: 800,
image_count: 0,
image_size: None,
image_quality: None,
image_output_format: None,
cache_ttl_minutes: Some(60),
},
)
.expect("billing should calculate");
let dimensions = &result.cost_result.snapshot.resolved_dimensions;
assert_eq!(dimensions.get("input_tokens"), Some(&json!(100)));
assert_eq!(dimensions.get("cache_creation_tokens"), Some(&json!(100)));
assert_eq!(dimensions.get("cache_read_tokens"), Some(&json!(800)));
assert_eq!(dimensions.get("total_input_context"), Some(&json!(1_000)));
let costs = &result.cost_result.snapshot.cost_breakdown;
assert!(costs.get("input_cost").copied().unwrap_or_default() > 0.0);
assert!(
costs
.get("cache_creation_uncategorized_cost")
.copied()
.unwrap_or_default()
> 0.0
);
assert!(costs.get("cache_read_cost").copied().unwrap_or_default() > 0.0);
}
#[test]
fn image_token_usage_without_image_output_price_bills_tokens_only() {
let pricing = BillingModelPricingSnapshot {
@@ -27,18 +27,23 @@ fn parse_api_family(api_format: Option<&str>) -> ApiFamily {
pub fn normalize_input_tokens_for_billing(
api_format: Option<&str>,
input_tokens: i64,
cache_creation_tokens: i64,
cache_read_tokens: i64,
) -> i64 {
if input_tokens <= 0 {
return input_tokens.max(0);
}
if cache_read_tokens <= 0 {
if cache_creation_tokens <= 0 && cache_read_tokens <= 0 {
return input_tokens;
}
match parse_api_family(api_format) {
ApiFamily::Claude => input_tokens,
ApiFamily::OpenAi | ApiFamily::Gemini => (input_tokens - cache_read_tokens).max(0),
ApiFamily::OpenAi => input_tokens
.saturating_sub(cache_creation_tokens.max(0))
.saturating_sub(cache_read_tokens.max(0))
.max(0),
ApiFamily::Gemini => (input_tokens - cache_read_tokens).max(0),
ApiFamily::Unknown => input_tokens,
}
}
@@ -57,9 +62,17 @@ pub fn normalize_total_input_context_for_cache_hit_rate(
ApiFamily::Claude => {
normalized_input_tokens.saturating_add(normalized_cache_creation_tokens)
}
ApiFamily::OpenAi | ApiFamily::Gemini => normalize_input_tokens_for_billing(
ApiFamily::OpenAi => normalize_input_tokens_for_billing(
api_format,
normalized_input_tokens,
normalized_cache_creation_tokens,
normalized_cache_read_tokens,
)
.saturating_add(normalized_cache_creation_tokens),
ApiFamily::Gemini => normalize_input_tokens_for_billing(
api_format,
normalized_input_tokens,
0,
normalized_cache_read_tokens,
),
ApiFamily::Unknown => {
@@ -83,11 +96,11 @@ mod tests {
#[test]
fn subtracts_cache_tokens_for_openai_and_gemini() {
assert_eq!(
normalize_input_tokens_for_billing(Some("openai:chat"), 100, 20),
80
normalize_input_tokens_for_billing(Some("openai:chat"), 100, 10, 20),
70
);
assert_eq!(
normalize_input_tokens_for_billing(Some("gemini:generate_content"), 100, 20),
normalize_input_tokens_for_billing(Some("gemini:generate_content"), 100, 10, 20),
80
);
}
@@ -95,7 +108,7 @@ mod tests {
#[test]
fn keeps_input_tokens_for_claude() {
assert_eq!(
normalize_input_tokens_for_billing(Some("claude:messages"), 100, 20),
normalize_input_tokens_for_billing(Some("claude:messages"), 100, 10, 20),
100
);
}