mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
fix(billing): recognize cached tokens in OpenAI Responses API usage
The Responses API reports usage in its own dialect: input_tokens: 9434 (inclusive grand total) input_tokens_details.cached_tokens: 8704 (cached subset) normalize_usage only knew prompt_tokens_details (chat completions), Anthropic top-level fields, and DeepSeek hit/miss. For /v1/responses payloads it found no cache fields and, seeing no prompt_tokens key, fell into the Anthropic-native branch that treats input_tokens as excluding cache — so cached tokens were billed at the full input rate and cache_read_input_tokens was recorded as 0. Two changes: * _extract_cache_tokens also reads input_tokens_details.cached_tokens and input_tokens_details.cache_write_tokens. * normalize_usage treats input_tokens as an inclusive grand total when input_tokens_details is present (Anthropic native never sends that object, so it safely disambiguates the field-name collision). Fixes both streaming and non-streaming /v1/responses billing, which share calculate_cost -> normalize_usage.
This commit is contained in:
@@ -15,6 +15,9 @@ names and in whether cached tokens are included in the input count:
|
|||||||
top-level, additive to (not included in) ``input_tokens``.
|
top-level, additive to (not included in) ``input_tokens``.
|
||||||
* DeepSeek: ``prompt_cache_hit_tokens`` / ``prompt_cache_miss_tokens``, with
|
* DeepSeek: ``prompt_cache_hit_tokens`` / ``prompt_cache_miss_tokens``, with
|
||||||
``prompt_tokens = hit + miss``.
|
``prompt_tokens = hit + miss``.
|
||||||
|
* OpenAI Responses API: ``input_tokens_details.cached_tokens`` (and
|
||||||
|
``cache_write_tokens``), included in ``input_tokens`` — same inclusive
|
||||||
|
semantics as ``prompt_tokens``, but under the Responses API field names.
|
||||||
|
|
||||||
What decides whether cached tokens must be subtracted out of the input count is
|
What decides whether cached tokens must be subtracted out of the input count is
|
||||||
**which prompt field the vendor uses**, not which cache field appears:
|
**which prompt field the vendor uses**, not which cache field appears:
|
||||||
@@ -22,6 +25,10 @@ What decides whether cached tokens must be subtracted out of the input count is
|
|||||||
* ``prompt_tokens`` present -> cached + cache-write tokens are *included* in it
|
* ``prompt_tokens`` present -> cached + cache-write tokens are *included* in it
|
||||||
(OpenAI family, DeepSeek, OpenRouter, litellm); subtract both so
|
(OpenAI family, DeepSeek, OpenRouter, litellm); subtract both so
|
||||||
``input_tokens`` holds only the regular-rate portion.
|
``input_tokens`` holds only the regular-rate portion.
|
||||||
|
* ``input_tokens_details`` present -> OpenAI Responses API; ``input_tokens``
|
||||||
|
*includes* cached + cache-write tokens, so subtract both. This disambiguates
|
||||||
|
the field-name collision with Anthropic native, which never sends
|
||||||
|
``input_tokens_details``.
|
||||||
* only ``input_tokens`` (Anthropic native) -> cached tokens are *additive*;
|
* only ``input_tokens`` (Anthropic native) -> cached tokens are *additive*;
|
||||||
leave ``input_tokens`` untouched.
|
leave ``input_tokens`` untouched.
|
||||||
|
|
||||||
@@ -78,6 +85,8 @@ def _extract_cache_tokens(usage_data: dict) -> tuple[int, int]:
|
|||||||
* Nested ``prompt_tokens_details``: ``cached_tokens`` for reads;
|
* Nested ``prompt_tokens_details``: ``cached_tokens`` for reads;
|
||||||
``cache_creation_tokens`` (litellm) or ``cache_write_tokens``
|
``cache_creation_tokens`` (litellm) or ``cache_write_tokens``
|
||||||
(OpenRouter) for writes.
|
(OpenRouter) for writes.
|
||||||
|
* Nested ``input_tokens_details`` (OpenAI Responses API): ``cached_tokens``
|
||||||
|
for reads, ``cache_write_tokens`` for writes.
|
||||||
* DeepSeek: ``prompt_cache_hit_tokens`` for reads (no write concept).
|
* DeepSeek: ``prompt_cache_hit_tokens`` for reads (no write concept).
|
||||||
"""
|
"""
|
||||||
cache_read = parse_token_count(usage_data.get("cache_read_input_tokens", 0))
|
cache_read = parse_token_count(usage_data.get("cache_read_input_tokens", 0))
|
||||||
@@ -92,6 +101,13 @@ def _extract_cache_tokens(usage_data: dict) -> tuple[int, int]:
|
|||||||
prompt_details, "cache_creation_tokens", "cache_write_tokens"
|
prompt_details, "cache_creation_tokens", "cache_write_tokens"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
input_details = usage_data.get("input_tokens_details")
|
||||||
|
if isinstance(input_details, dict):
|
||||||
|
if not cache_read:
|
||||||
|
cache_read = parse_token_count(input_details.get("cached_tokens", 0))
|
||||||
|
if not cache_write:
|
||||||
|
cache_write = parse_token_count(input_details.get("cache_write_tokens", 0))
|
||||||
|
|
||||||
if not cache_read:
|
if not cache_read:
|
||||||
# DeepSeek: prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens
|
# DeepSeek: prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens
|
||||||
cache_read = parse_token_count(usage_data.get("prompt_cache_hit_tokens", 0))
|
cache_read = parse_token_count(usage_data.get("prompt_cache_hit_tokens", 0))
|
||||||
@@ -103,16 +119,16 @@ def normalize_usage(usage_data: object) -> NormalizedUsage | None:
|
|||||||
"""Map a vendor usage dict onto the canonical shape, or None if absent.
|
"""Map a vendor usage dict onto the canonical shape, or None if absent.
|
||||||
|
|
||||||
Cached reads and writes are subtracted from the input count exactly once,
|
Cached reads and writes are subtracted from the input count exactly once,
|
||||||
only for dialects that report a ``prompt_tokens`` grand total that already
|
only for dialects whose input grand total already includes them: the
|
||||||
includes them (OpenAI family, DeepSeek, OpenRouter, litellm). Anthropic
|
``prompt_tokens`` family (OpenAI chat completions, DeepSeek, OpenRouter,
|
||||||
native reports them additively under ``input_tokens`` and is left untouched.
|
litellm) and the OpenAI Responses API (``input_tokens`` inclusive,
|
||||||
|
identified by the presence of ``input_tokens_details``). Anthropic native
|
||||||
|
reports them additively under ``input_tokens`` and is left untouched.
|
||||||
"""
|
"""
|
||||||
if not isinstance(usage_data, dict):
|
if not isinstance(usage_data, dict):
|
||||||
return None
|
return None
|
||||||
|
|
||||||
output_tokens = _first_token_count(
|
output_tokens = _first_token_count(usage_data, "completion_tokens", "output_tokens")
|
||||||
usage_data, "completion_tokens", "output_tokens"
|
|
||||||
)
|
|
||||||
cache_read, cache_write = _extract_cache_tokens(usage_data)
|
cache_read, cache_write = _extract_cache_tokens(usage_data)
|
||||||
|
|
||||||
# ``prompt_tokens`` is the inclusive grand total; ``input_tokens`` (Anthropic
|
# ``prompt_tokens`` is the inclusive grand total; ``input_tokens`` (Anthropic
|
||||||
@@ -120,6 +136,12 @@ def normalize_usage(usage_data: object) -> NormalizedUsage | None:
|
|||||||
if "prompt_tokens" in usage_data:
|
if "prompt_tokens" in usage_data:
|
||||||
input_tokens = parse_token_count(usage_data.get("prompt_tokens", 0))
|
input_tokens = parse_token_count(usage_data.get("prompt_tokens", 0))
|
||||||
input_tokens = max(0, input_tokens - cache_read - cache_write)
|
input_tokens = max(0, input_tokens - cache_read - cache_write)
|
||||||
|
elif isinstance(usage_data.get("input_tokens_details"), dict):
|
||||||
|
# OpenAI Responses API: ``input_tokens`` is also an inclusive grand
|
||||||
|
# total (cached tokens are a subset of it), signalled by the nested
|
||||||
|
# ``input_tokens_details`` object Anthropic native never sends.
|
||||||
|
input_tokens = parse_token_count(usage_data.get("input_tokens", 0))
|
||||||
|
input_tokens = max(0, input_tokens - cache_read - cache_write)
|
||||||
else:
|
else:
|
||||||
input_tokens = parse_token_count(usage_data.get("input_tokens", 0))
|
input_tokens = parse_token_count(usage_data.get("input_tokens", 0))
|
||||||
|
|
||||||
|
|||||||
@@ -92,6 +92,58 @@ from routstr.payment.usage import NormalizedUsage, normalize_usage
|
|||||||
cache_write_tokens=2000,
|
cache_write_tokens=2000,
|
||||||
),
|
),
|
||||||
),
|
),
|
||||||
|
# OpenAI Responses API: cached tokens nested under input_tokens_details
|
||||||
|
# and INCLUDED in input_tokens (same semantics as prompt_tokens) →
|
||||||
|
# subtracted. Real payload shape from /v1/responses response.completed.
|
||||||
|
(
|
||||||
|
{
|
||||||
|
"input_tokens": 9434,
|
||||||
|
"input_tokens_details": {
|
||||||
|
"cached_tokens": 8704,
|
||||||
|
"cache_write_tokens": 0,
|
||||||
|
},
|
||||||
|
"output_tokens": 9,
|
||||||
|
"output_tokens_details": {"reasoning_tokens": 0},
|
||||||
|
"total_tokens": 9443,
|
||||||
|
},
|
||||||
|
NormalizedUsage(
|
||||||
|
input_tokens=730,
|
||||||
|
output_tokens=9,
|
||||||
|
cache_read_tokens=8704,
|
||||||
|
cache_write_tokens=0,
|
||||||
|
),
|
||||||
|
),
|
||||||
|
# OpenAI Responses API without a cache hit: input_tokens untouched.
|
||||||
|
(
|
||||||
|
{
|
||||||
|
"input_tokens": 9412,
|
||||||
|
"input_tokens_details": {
|
||||||
|
"cached_tokens": 0,
|
||||||
|
"cache_write_tokens": 0,
|
||||||
|
},
|
||||||
|
"output_tokens": 11,
|
||||||
|
"total_tokens": 9423,
|
||||||
|
},
|
||||||
|
NormalizedUsage(input_tokens=9412, output_tokens=11),
|
||||||
|
),
|
||||||
|
# OpenAI Responses API with cache writes: both reads and writes are
|
||||||
|
# included in input_tokens → both subtracted.
|
||||||
|
(
|
||||||
|
{
|
||||||
|
"input_tokens": 1000,
|
||||||
|
"input_tokens_details": {
|
||||||
|
"cached_tokens": 400,
|
||||||
|
"cache_write_tokens": 200,
|
||||||
|
},
|
||||||
|
"output_tokens": 50,
|
||||||
|
},
|
||||||
|
NormalizedUsage(
|
||||||
|
input_tokens=400,
|
||||||
|
output_tokens=50,
|
||||||
|
cache_read_tokens=400,
|
||||||
|
cache_write_tokens=200,
|
||||||
|
),
|
||||||
|
),
|
||||||
# litellm-normalized Anthropic: prompt_tokens is the grand total and the
|
# litellm-normalized Anthropic: prompt_tokens is the grand total and the
|
||||||
# write field is named cache_creation_tokens; top-level fields mirror it.
|
# write field is named cache_creation_tokens; top-level fields mirror it.
|
||||||
# prompt_tokens present → both subtracted (NOT additive like native).
|
# prompt_tokens present → both subtracted (NOT additive like native).
|
||||||
|
|||||||
Reference in New Issue
Block a user