From bc8172783d748893e5192a105b44912b2bf98ef3 Mon Sep 17 00:00:00 2001 From: redshift <213178690+1ftredsh@users.noreply.github.com> Date: Mon, 21 Sep 2026 16:06:48 +0300 Subject: [PATCH] fix(usage): don't fold cache tokens into inclusive prompt_tokens MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit _fold_cache_into_input_tokens rolled cache_read/cache_creation tokens into the visible prompt_tokens for every dialect. In the OpenAI family (Venice, OpenAI, DeepSeek, OpenRouter, litellm) prompt_tokens already includes the cached portion, so the client-visible prompt count double-counted cache hits (Venice: 14075-token prompt shown as 27997 after a 13922-token cache read). Billing was unaffected — normalize_usage already subtracts cache exactly once and the fold runs after cost calculation. The fold now mirrors normalize_usage: only Anthropic-native input_tokens (which excludes cache) gets the roll-up; prompt_tokens is left untouched. Adds dict-based regression tests for the Venice/OpenAI, Anthropic-native and litellm-mirror shapes (existing Mock tests never exercised the logic). --- routstr/upstream/base.py | 17 +++++++---- tests/unit/test_coverage_base2.py | 48 +++++++++++++++++++++++++++++++ 2 files changed, 60 insertions(+), 5 deletions(-) diff --git a/routstr/upstream/base.py b/routstr/upstream/base.py index 1a080eaf..0b6be7f1 100644 --- a/routstr/upstream/base.py +++ b/routstr/upstream/base.py @@ -433,6 +433,14 @@ class BaseUpstreamProvider: ``cache_creation_input_tokens`` fields are left in place for clients that want the breakdown. + Which field may be folded mirrors ``normalize_usage`` exactly: + Anthropic-native ``input_tokens`` *excludes* the cached portion and + needs the roll-up, while a ``prompt_tokens`` grand total (OpenAI + family, DeepSeek, OpenRouter, litellm) *already includes* it — + folding there double-counts the cache in the visible prompt total + (Venice showed 27997 prompt tokens for a 14075-token prompt after a + 13922-token cache read). + For Anthropic-shaped responses (``input_tokens`` present), the cache fields are forced to ``0`` when the upstream omitted them, so the client always sees a consistent shape. @@ -459,11 +467,10 @@ class BaseUpstreamProvider: usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra except (TypeError, ValueError): pass - if "prompt_tokens" in usage: - try: - usage["prompt_tokens"] = int(usage.get("prompt_tokens") or 0) + extra - except (TypeError, ValueError): - pass + # ``prompt_tokens`` is deliberately left untouched: in every dialect + # that reports it, it is an inclusive grand total that already + # contains the cached portion — the same assumption + # ``normalize_usage`` subtracts against when billing. def _apply_provider_field(self, response_json: object) -> None: """Stamp the routstr ``provider`` field onto an upstream response payload. diff --git a/tests/unit/test_coverage_base2.py b/tests/unit/test_coverage_base2.py index 1c6a7e46..96bcd7e2 100644 --- a/tests/unit/test_coverage_base2.py +++ b/tests/unit/test_coverage_base2.py @@ -116,6 +116,54 @@ def test_fold_cache_preserves_total() -> None: assert usage.prompt_tokens == 100 +def test_fold_cache_openai_dialect_prompt_tokens_untouched() -> None: + """Venice/OpenAI shape: prompt_tokens already includes cached tokens. + + Regression: folding cache_read into prompt_tokens double-counted the + cache (14075 real prompt shown as 27997 after a 13922-token cache read). + """ + from routstr.upstream.base import BaseUpstreamProvider + + usage = { + "prompt_tokens": 14075, + "completion_tokens": 24, + "total_tokens": 14099, + "prompt_tokens_details": {"cached_tokens": 13922}, + "cache_read_input_tokens": 13922, + } + BaseUpstreamProvider._fold_cache_into_input_tokens(usage) + assert usage["prompt_tokens"] == 14075 + assert usage["cache_read_input_tokens"] == 13922 + + +def test_fold_cache_anthropic_dialect_folds_input_tokens() -> None: + """Anthropic-native shape: input_tokens excludes cache, so it is folded.""" + from routstr.upstream.base import BaseUpstreamProvider + + usage = { + "input_tokens": 153, + "output_tokens": 24, + "cache_read_input_tokens": 13922, + "cache_creation_input_tokens": 0, + } + BaseUpstreamProvider._fold_cache_into_input_tokens(usage) + assert usage["input_tokens"] == 153 + 13922 + + +def test_fold_cache_litellm_mirror_folds_only_input_tokens() -> None: + """Both fields present (litellm mirror): fold input_tokens only.""" + from routstr.upstream.base import BaseUpstreamProvider + + usage = { + "prompt_tokens": 14075, # inclusive grand total + "input_tokens": 153, # additive Anthropic mirror + "cache_read_input_tokens": 13922, + } + BaseUpstreamProvider._fold_cache_into_input_tokens(usage) + assert usage["prompt_tokens"] == 14075 + assert usage["input_tokens"] == 153 + 13922 + + # =========================================================================== # get_cached_models / get_cached_model_by_id # ===========================================================================