From 71c8ae3c847b9cc973f2b389a152cce76edf3076 Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:09:16 +0300 Subject: [PATCH] fix(billing): recognize cached tokens in OpenAI Responses API usage MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The Responses API reports usage in its own dialect: input_tokens: 9434 (inclusive grand total) input_tokens_details.cached_tokens: 8704 (cached subset) normalize_usage only knew prompt_tokens_details (chat completions), Anthropic top-level fields, and DeepSeek hit/miss. For /v1/responses payloads it found no cache fields and, seeing no prompt_tokens key, fell into the Anthropic-native branch that treats input_tokens as excluding cache — so cached tokens were billed at the full input rate and cache_read_input_tokens was recorded as 0. Two changes: * _extract_cache_tokens also reads input_tokens_details.cached_tokens and input_tokens_details.cache_write_tokens. * normalize_usage treats input_tokens as an inclusive grand total when input_tokens_details is present (Anthropic native never sends that object, so it safely disambiguates the field-name collision). Fixes both streaming and non-streaming /v1/responses billing, which share calculate_cost -> normalize_usage. --- routstr/payment/usage.py | 34 ++++++++++++++--- tests/unit/test_usage_normalization.py | 52 ++++++++++++++++++++++++++ 2 files changed, 80 insertions(+), 6 deletions(-) diff --git a/routstr/payment/usage.py b/routstr/payment/usage.py index 02c90055..08d675ad 100644 --- a/routstr/payment/usage.py +++ b/routstr/payment/usage.py @@ -15,6 +15,9 @@ names and in whether cached tokens are included in the input count: top-level, additive to (not included in) ``input_tokens``. * DeepSeek: ``prompt_cache_hit_tokens`` / ``prompt_cache_miss_tokens``, with ``prompt_tokens = hit + miss``. +* OpenAI Responses API: ``input_tokens_details.cached_tokens`` (and + ``cache_write_tokens``), included in ``input_tokens`` — same inclusive + semantics as ``prompt_tokens``, but under the Responses API field names. What decides whether cached tokens must be subtracted out of the input count is **which prompt field the vendor uses**, not which cache field appears: @@ -22,6 +25,10 @@ What decides whether cached tokens must be subtracted out of the input count is * ``prompt_tokens`` present -> cached + cache-write tokens are *included* in it (OpenAI family, DeepSeek, OpenRouter, litellm); subtract both so ``input_tokens`` holds only the regular-rate portion. +* ``input_tokens_details`` present -> OpenAI Responses API; ``input_tokens`` + *includes* cached + cache-write tokens, so subtract both. This disambiguates + the field-name collision with Anthropic native, which never sends + ``input_tokens_details``. * only ``input_tokens`` (Anthropic native) -> cached tokens are *additive*; leave ``input_tokens`` untouched. @@ -78,6 +85,8 @@ def _extract_cache_tokens(usage_data: dict) -> tuple[int, int]: * Nested ``prompt_tokens_details``: ``cached_tokens`` for reads; ``cache_creation_tokens`` (litellm) or ``cache_write_tokens`` (OpenRouter) for writes. + * Nested ``input_tokens_details`` (OpenAI Responses API): ``cached_tokens`` + for reads, ``cache_write_tokens`` for writes. * DeepSeek: ``prompt_cache_hit_tokens`` for reads (no write concept). """ cache_read = parse_token_count(usage_data.get("cache_read_input_tokens", 0)) @@ -92,6 +101,13 @@ def _extract_cache_tokens(usage_data: dict) -> tuple[int, int]: prompt_details, "cache_creation_tokens", "cache_write_tokens" ) + input_details = usage_data.get("input_tokens_details") + if isinstance(input_details, dict): + if not cache_read: + cache_read = parse_token_count(input_details.get("cached_tokens", 0)) + if not cache_write: + cache_write = parse_token_count(input_details.get("cache_write_tokens", 0)) + if not cache_read: # DeepSeek: prompt_tokens = prompt_cache_hit_tokens + prompt_cache_miss_tokens cache_read = parse_token_count(usage_data.get("prompt_cache_hit_tokens", 0)) @@ -103,16 +119,16 @@ def normalize_usage(usage_data: object) -> NormalizedUsage | None: """Map a vendor usage dict onto the canonical shape, or None if absent. Cached reads and writes are subtracted from the input count exactly once, - only for dialects that report a ``prompt_tokens`` grand total that already - includes them (OpenAI family, DeepSeek, OpenRouter, litellm). Anthropic - native reports them additively under ``input_tokens`` and is left untouched. + only for dialects whose input grand total already includes them: the + ``prompt_tokens`` family (OpenAI chat completions, DeepSeek, OpenRouter, + litellm) and the OpenAI Responses API (``input_tokens`` inclusive, + identified by the presence of ``input_tokens_details``). Anthropic native + reports them additively under ``input_tokens`` and is left untouched. """ if not isinstance(usage_data, dict): return None - output_tokens = _first_token_count( - usage_data, "completion_tokens", "output_tokens" - ) + output_tokens = _first_token_count(usage_data, "completion_tokens", "output_tokens") cache_read, cache_write = _extract_cache_tokens(usage_data) # ``prompt_tokens`` is the inclusive grand total; ``input_tokens`` (Anthropic @@ -120,6 +136,12 @@ def normalize_usage(usage_data: object) -> NormalizedUsage | None: if "prompt_tokens" in usage_data: input_tokens = parse_token_count(usage_data.get("prompt_tokens", 0)) input_tokens = max(0, input_tokens - cache_read - cache_write) + elif isinstance(usage_data.get("input_tokens_details"), dict): + # OpenAI Responses API: ``input_tokens`` is also an inclusive grand + # total (cached tokens are a subset of it), signalled by the nested + # ``input_tokens_details`` object Anthropic native never sends. + input_tokens = parse_token_count(usage_data.get("input_tokens", 0)) + input_tokens = max(0, input_tokens - cache_read - cache_write) else: input_tokens = parse_token_count(usage_data.get("input_tokens", 0)) diff --git a/tests/unit/test_usage_normalization.py b/tests/unit/test_usage_normalization.py index 50c6c645..66a44dc4 100644 --- a/tests/unit/test_usage_normalization.py +++ b/tests/unit/test_usage_normalization.py @@ -92,6 +92,58 @@ from routstr.payment.usage import NormalizedUsage, normalize_usage cache_write_tokens=2000, ), ), + # OpenAI Responses API: cached tokens nested under input_tokens_details + # and INCLUDED in input_tokens (same semantics as prompt_tokens) → + # subtracted. Real payload shape from /v1/responses response.completed. + ( + { + "input_tokens": 9434, + "input_tokens_details": { + "cached_tokens": 8704, + "cache_write_tokens": 0, + }, + "output_tokens": 9, + "output_tokens_details": {"reasoning_tokens": 0}, + "total_tokens": 9443, + }, + NormalizedUsage( + input_tokens=730, + output_tokens=9, + cache_read_tokens=8704, + cache_write_tokens=0, + ), + ), + # OpenAI Responses API without a cache hit: input_tokens untouched. + ( + { + "input_tokens": 9412, + "input_tokens_details": { + "cached_tokens": 0, + "cache_write_tokens": 0, + }, + "output_tokens": 11, + "total_tokens": 9423, + }, + NormalizedUsage(input_tokens=9412, output_tokens=11), + ), + # OpenAI Responses API with cache writes: both reads and writes are + # included in input_tokens → both subtracted. + ( + { + "input_tokens": 1000, + "input_tokens_details": { + "cached_tokens": 400, + "cache_write_tokens": 200, + }, + "output_tokens": 50, + }, + NormalizedUsage( + input_tokens=400, + output_tokens=50, + cache_read_tokens=400, + cache_write_tokens=200, + ), + ), # litellm-normalized Anthropic: prompt_tokens is the grand total and the # write field is named cache_creation_tokens; top-level fields mirror it. # prompt_tokens present → both subtracted (NOT additive like native).