fix: 统一缓存token计费口径,避免OpenAI错误计费 (#144)

* fix: 统一缓存token计费口径,避免OpenAI错误计费

* fix: 添加 Gemini 缓存 token 归一化支持,优化代码结构

- Gemini 的 promptTokenCount 包含 cachedContentTokenCount,需要扣除
- 合并重复的 helper 函数为 _get_api_family()
- 将 import 移到文件顶部
- 补充 Gemini 测试用例

---------

Co-authored-by: fawney19 <elky0401@gmail.com>
This commit is contained in:
Yorha
2026-02-04 16:04:09 +08:00
committed by GitHub
parent f64631a3a3
commit 26a0f99f8f
5 changed files with 130 additions and 32 deletions

View File

@@ -0,0 +1,44 @@
from src.services.billing.token_normalization import normalize_input_tokens_for_billing
from src.services.billing.usage_mapper import UsageMapper
class TestNormalizeInputTokensForBilling:
def test_openai_family_subtracts_cached_tokens(self) -> None:
assert normalize_input_tokens_for_billing("openai:cli", 160_070, 81_664) == 78_406
def test_claude_family_does_not_change(self) -> None:
assert normalize_input_tokens_for_billing("claude:cli", 160_070, 81_664) == 160_070
def test_gemini_family_subtracts_cached_tokens(self) -> None:
# Gemini 的 promptTokenCount 包含 cachedContentTokenCount需要扣除
assert normalize_input_tokens_for_billing("gemini:chat", 323_392, 323_384) == 8
assert normalize_input_tokens_for_billing("gemini:cli", 100, 20) == 80
def test_missing_format_does_not_change(self) -> None:
assert normalize_input_tokens_for_billing(None, 100, 20) == 100
assert normalize_input_tokens_for_billing("", 100, 20) == 100
def test_clamps_when_cached_tokens_exceed_input(self) -> None:
assert normalize_input_tokens_for_billing("openai:cli", 10, 20) == 0
assert normalize_input_tokens_for_billing("gemini:chat", 10, 20) == 0
class TestUsageMapperOpenAICacheTokens:
def test_openai_mapping_maps_cached_tokens_details(self) -> None:
raw_usage = {
"prompt_tokens": 100,
"completion_tokens": 50,
"prompt_tokens_details": {"cached_tokens": 20},
}
usage = UsageMapper.map(raw_usage, api_format="openai:chat")
assert usage.input_tokens == 100
assert usage.output_tokens == 50
assert usage.cache_read_tokens == 20
def test_openai_mapping_without_cached_tokens_is_unchanged(self) -> None:
raw_usage = {"prompt_tokens": 100, "completion_tokens": 50}
usage = UsageMapper.map(raw_usage, api_format="openai:chat")
assert usage.input_tokens == 100
assert usage.output_tokens == 50
assert usage.cache_read_tokens == 0