mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
fix(billing): recognize cached tokens in OpenAI Responses API usage
The Responses API reports usage in its own dialect: input_tokens: 9434 (inclusive grand total) input_tokens_details.cached_tokens: 8704 (cached subset) normalize_usage only knew prompt_tokens_details (chat completions), Anthropic top-level fields, and DeepSeek hit/miss. For /v1/responses payloads it found no cache fields and, seeing no prompt_tokens key, fell into the Anthropic-native branch that treats input_tokens as excluding cache — so cached tokens were billed at the full input rate and cache_read_input_tokens was recorded as 0. Two changes: * _extract_cache_tokens also reads input_tokens_details.cached_tokens and input_tokens_details.cache_write_tokens. * normalize_usage treats input_tokens as an inclusive grand total when input_tokens_details is present (Anthropic native never sends that object, so it safely disambiguates the field-name collision). Fixes both streaming and non-streaming /v1/responses billing, which share calculate_cost -> normalize_usage.
This commit is contained in:
@@ -15,6 +15,9 @@ names and in whether cached tokens are included in the input count:
|
||||
top-level, additive to (not included in) ``input_tokens``.
|
||||
* DeepSeek: ``prompt_cache_hit_tokens`` / ``prompt_cache_miss_tokens``, with
|
||||
``prompt_tokens = hit + miss``.
|
||||
* OpenAI Responses API: ``input_tokens_details.cached_tokens`` (and
|
||||
``cache_write_tokens``), included in ``input_tokens`` — same inclusive
|
||||
semantics as ``prompt_tokens``, but under the Responses API field names.
|
||||
|
||||
What decides whether cached tokens must be subtracted out of the input count is
|
||||
**which prompt field the vendor uses**, not which cache field appears:
|
||||
@@ -22,6 +25,10 @@ What decides whether cached tokens must be subtracted out of the input count is
|
||||
* ``prompt_tokens`` present -> cached + cache-write tokens are *included* in it
|
||||
(OpenAI family, DeepSeek, OpenRouter, litellm); subtract both so
|
||||
``input_tokens`` holds only the regular-rate portion.
|
||||
* ``input_tokens_details`` present -> OpenAI Responses API; ``input_tokens``
|
||||
*includes* cached + cache-write tokens, so subtract both. This disambiguates
|
||||
the field-name collision with Anthropic native, which never sends
|
||||
``input_tokens_details``.
|
||||
* only ``input_tokens`` (Anthropic native) -> cached tokens are *additive*;
|
||||
leave ``input_tokens`` untouched.
|
||||
|
||||
@@ -78,6 +85,8 @@ def _extract_cache_tokens(usage_data: dict) -> tuple[int, int]:
|
||||
* Nested ``prompt_tokens_details``: ``cached_tokens`` for reads;
|
||||
``cache_creation_tokens`` (litellm) or ``cache_write_tokens``
|
||||
(OpenRouter) for writes.
|
||||
* Nested ``input_tokens_details`` (OpenAI Responses API): ``cached_tokens``
|
||||
for reads, ``cache_write_tokens`` for writes.
|
||||
* DeepSeek: ``prompt_cache_hit_tokens`` for reads (no write concept).
|
||||
"""
|
||||
cache_read = parse_token_count(usage_data.get("cache_read_input_tokens", 0))
|
||||
@@ -92,6 +101,13 @@ def _extract_cache_tokens(usage_data: dict) -> tuple[int, int]:
|
||||
prompt_details, "cache_creation_tokens", "cache_write_tokens"
|
||||
)
|
||||
|
||||
input_details = usage_data.get("input_tokens_details")
|
||||
if isinstance(input_details, dict):
|
||||
if not cache_read:
|
||||
cache_read = parse_token_count(input_details.get("cached_tokens", 0))
|
||||
if not cache_write:
|
||||
cache_write = parse_token_count(input_details.get("cache_write_tokens", 0))
|
||||
|
||||
if not cache_read:
|
||||
# DeepSeek: prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens
|
||||
cache_read = parse_token_count(usage_data.get("prompt_cache_hit_tokens", 0))
|
||||
@@ -103,16 +119,16 @@ def normalize_usage(usage_data: object) -> NormalizedUsage | None:
|
||||
"""Map a vendor usage dict onto the canonical shape, or None if absent.
|
||||
|
||||
Cached reads and writes are subtracted from the input count exactly once,
|
||||
only for dialects that report a ``prompt_tokens`` grand total that already
|
||||
includes them (OpenAI family, DeepSeek, OpenRouter, litellm). Anthropic
|
||||
native reports them additively under ``input_tokens`` and is left untouched.
|
||||
only for dialects whose input grand total already includes them: the
|
||||
``prompt_tokens`` family (OpenAI chat completions, DeepSeek, OpenRouter,
|
||||
litellm) and the OpenAI Responses API (``input_tokens`` inclusive,
|
||||
identified by the presence of ``input_tokens_details``). Anthropic native
|
||||
reports them additively under ``input_tokens`` and is left untouched.
|
||||
"""
|
||||
if not isinstance(usage_data, dict):
|
||||
return None
|
||||
|
||||
output_tokens = _first_token_count(
|
||||
usage_data, "completion_tokens", "output_tokens"
|
||||
)
|
||||
output_tokens = _first_token_count(usage_data, "completion_tokens", "output_tokens")
|
||||
cache_read, cache_write = _extract_cache_tokens(usage_data)
|
||||
|
||||
# ``prompt_tokens`` is the inclusive grand total; ``input_tokens`` (Anthropic
|
||||
@@ -120,6 +136,12 @@ def normalize_usage(usage_data: object) -> NormalizedUsage | None:
|
||||
if "prompt_tokens" in usage_data:
|
||||
input_tokens = parse_token_count(usage_data.get("prompt_tokens", 0))
|
||||
input_tokens = max(0, input_tokens - cache_read - cache_write)
|
||||
elif isinstance(usage_data.get("input_tokens_details"), dict):
|
||||
# OpenAI Responses API: ``input_tokens`` is also an inclusive grand
|
||||
# total (cached tokens are a subset of it), signalled by the nested
|
||||
# ``input_tokens_details`` object Anthropic native never sends.
|
||||
input_tokens = parse_token_count(usage_data.get("input_tokens", 0))
|
||||
input_tokens = max(0, input_tokens - cache_read - cache_write)
|
||||
else:
|
||||
input_tokens = parse_token_count(usage_data.get("input_tokens", 0))
|
||||
|
||||
|
||||
@@ -92,6 +92,58 @@ from routstr.payment.usage import NormalizedUsage, normalize_usage
|
||||
cache_write_tokens=2000,
|
||||
),
|
||||
),
|
||||
# OpenAI Responses API: cached tokens nested under input_tokens_details
|
||||
# and INCLUDED in input_tokens (same semantics as prompt_tokens) →
|
||||
# subtracted. Real payload shape from /v1/responses response.completed.
|
||||
(
|
||||
{
|
||||
"input_tokens": 9434,
|
||||
"input_tokens_details": {
|
||||
"cached_tokens": 8704,
|
||||
"cache_write_tokens": 0,
|
||||
},
|
||||
"output_tokens": 9,
|
||||
"output_tokens_details": {"reasoning_tokens": 0},
|
||||
"total_tokens": 9443,
|
||||
},
|
||||
NormalizedUsage(
|
||||
input_tokens=730,
|
||||
output_tokens=9,
|
||||
cache_read_tokens=8704,
|
||||
cache_write_tokens=0,
|
||||
),
|
||||
),
|
||||
# OpenAI Responses API without a cache hit: input_tokens untouched.
|
||||
(
|
||||
{
|
||||
"input_tokens": 9412,
|
||||
"input_tokens_details": {
|
||||
"cached_tokens": 0,
|
||||
"cache_write_tokens": 0,
|
||||
},
|
||||
"output_tokens": 11,
|
||||
"total_tokens": 9423,
|
||||
},
|
||||
NormalizedUsage(input_tokens=9412, output_tokens=11),
|
||||
),
|
||||
# OpenAI Responses API with cache writes: both reads and writes are
|
||||
# included in input_tokens → both subtracted.
|
||||
(
|
||||
{
|
||||
"input_tokens": 1000,
|
||||
"input_tokens_details": {
|
||||
"cached_tokens": 400,
|
||||
"cache_write_tokens": 200,
|
||||
},
|
||||
"output_tokens": 50,
|
||||
},
|
||||
NormalizedUsage(
|
||||
input_tokens=400,
|
||||
output_tokens=50,
|
||||
cache_read_tokens=400,
|
||||
cache_write_tokens=200,
|
||||
),
|
||||
),
|
||||
# litellm-normalized Anthropic: prompt_tokens is the grand total and the
|
||||
# write field is named cache_creation_tokens; top-level fields mirror it.
|
||||
# prompt_tokens present → both subtracted (NOT additive like native).
|
||||
|
||||
Reference in New Issue
Block a user