diff --git a/routstr/upstream/base.py b/routstr/upstream/base.py index 1a080eaf..0b6be7f1 100644 --- a/routstr/upstream/base.py +++ b/routstr/upstream/base.py @@ -433,6 +433,14 @@ class BaseUpstreamProvider: ``cache_creation_input_tokens`` fields are left in place for clients that want the breakdown. + Which field may be folded mirrors ``normalize_usage`` exactly: + Anthropic-native ``input_tokens`` *excludes* the cached portion and + needs the roll-up, while a ``prompt_tokens`` grand total (OpenAI + family, DeepSeek, OpenRouter, litellm) *already includes* it — + folding there double-counts the cache in the visible prompt total + (Venice showed 27997 prompt tokens for a 14075-token prompt after a + 13922-token cache read). + For Anthropic-shaped responses (``input_tokens`` present), the cache fields are forced to ``0`` when the upstream omitted them, so the client always sees a consistent shape. @@ -459,11 +467,10 @@ class BaseUpstreamProvider: usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra except (TypeError, ValueError): pass - if "prompt_tokens" in usage: - try: - usage["prompt_tokens"] = int(usage.get("prompt_tokens") or 0) + extra - except (TypeError, ValueError): - pass + # ``prompt_tokens`` is deliberately left untouched: in every dialect + # that reports it, it is an inclusive grand total that already + # contains the cached portion — the same assumption + # ``normalize_usage`` subtracts against when billing. def _apply_provider_field(self, response_json: object) -> None: """Stamp the routstr ``provider`` field onto an upstream response payload. diff --git a/tests/unit/test_coverage_base2.py b/tests/unit/test_coverage_base2.py index 1c6a7e46..96bcd7e2 100644 --- a/tests/unit/test_coverage_base2.py +++ b/tests/unit/test_coverage_base2.py @@ -116,6 +116,54 @@ def test_fold_cache_preserves_total() -> None: assert usage.prompt_tokens == 100 +def test_fold_cache_openai_dialect_prompt_tokens_untouched() -> None: + """Venice/OpenAI shape: prompt_tokens already includes cached tokens. + + Regression: folding cache_read into prompt_tokens double-counted the + cache (14075 real prompt shown as 27997 after a 13922-token cache read). + """ + from routstr.upstream.base import BaseUpstreamProvider + + usage = { + "prompt_tokens": 14075, + "completion_tokens": 24, + "total_tokens": 14099, + "prompt_tokens_details": {"cached_tokens": 13922}, + "cache_read_input_tokens": 13922, + } + BaseUpstreamProvider._fold_cache_into_input_tokens(usage) + assert usage["prompt_tokens"] == 14075 + assert usage["cache_read_input_tokens"] == 13922 + + +def test_fold_cache_anthropic_dialect_folds_input_tokens() -> None: + """Anthropic-native shape: input_tokens excludes cache, so it is folded.""" + from routstr.upstream.base import BaseUpstreamProvider + + usage = { + "input_tokens": 153, + "output_tokens": 24, + "cache_read_input_tokens": 13922, + "cache_creation_input_tokens": 0, + } + BaseUpstreamProvider._fold_cache_into_input_tokens(usage) + assert usage["input_tokens"] == 153 + 13922 + + +def test_fold_cache_litellm_mirror_folds_only_input_tokens() -> None: + """Both fields present (litellm mirror): fold input_tokens only.""" + from routstr.upstream.base import BaseUpstreamProvider + + usage = { + "prompt_tokens": 14075, # inclusive grand total + "input_tokens": 153, # additive Anthropic mirror + "cache_read_input_tokens": 13922, + } + BaseUpstreamProvider._fold_cache_into_input_tokens(usage) + assert usage["prompt_tokens"] == 14075 + assert usage["input_tokens"] == 153 + 13922 + + # =========================================================================== # get_cached_models / get_cached_model_by_id # ===========================================================================