Merge pull request #752 from Routstr/fix/cache-fold-double-count

fix(usage): don't fold cache tokens into inclusive prompt_tokens
This commit is contained in:
9qeklajc
2026-09-21 16:12:09 +02:00
committed by GitHub
2 changed files with 60 additions and 5 deletions
+12 -5
View File
@@ -433,6 +433,14 @@ class BaseUpstreamProvider:
``cache_creation_input_tokens`` fields are left in place for clients ``cache_creation_input_tokens`` fields are left in place for clients
that want the breakdown. that want the breakdown.
Which field may be folded mirrors ``normalize_usage`` exactly:
Anthropic-native ``input_tokens`` *excludes* the cached portion and
needs the roll-up, while a ``prompt_tokens`` grand total (OpenAI
family, DeepSeek, OpenRouter, litellm) *already includes* it —
folding there double-counts the cache in the visible prompt total
(Venice showed 27997 prompt tokens for a 14075-token prompt after a
13922-token cache read).
For Anthropic-shaped responses (``input_tokens`` present), the cache For Anthropic-shaped responses (``input_tokens`` present), the cache
fields are forced to ``0`` when the upstream omitted them, so the fields are forced to ``0`` when the upstream omitted them, so the
client always sees a consistent shape. client always sees a consistent shape.
@@ -459,11 +467,10 @@ class BaseUpstreamProvider:
usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra
except (TypeError, ValueError): except (TypeError, ValueError):
pass pass
if "prompt_tokens" in usage: # ``prompt_tokens`` is deliberately left untouched: in every dialect
try: # that reports it, it is an inclusive grand total that already
usage["prompt_tokens"] = int(usage.get("prompt_tokens") or 0) + extra # contains the cached portion — the same assumption
except (TypeError, ValueError): # ``normalize_usage`` subtracts against when billing.
pass
def _apply_provider_field(self, response_json: object) -> None: def _apply_provider_field(self, response_json: object) -> None:
"""Stamp the routstr ``provider`` field onto an upstream response payload. """Stamp the routstr ``provider`` field onto an upstream response payload.
+48
View File
@@ -116,6 +116,54 @@ def test_fold_cache_preserves_total() -> None:
assert usage.prompt_tokens == 100 assert usage.prompt_tokens == 100
def test_fold_cache_openai_dialect_prompt_tokens_untouched() -> None:
"""Venice/OpenAI shape: prompt_tokens already includes cached tokens.
Regression: folding cache_read into prompt_tokens double-counted the
cache (14075 real prompt shown as 27997 after a 13922-token cache read).
"""
from routstr.upstream.base import BaseUpstreamProvider
usage = {
"prompt_tokens": 14075,
"completion_tokens": 24,
"total_tokens": 14099,
"prompt_tokens_details": {"cached_tokens": 13922},
"cache_read_input_tokens": 13922,
}
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
assert usage["prompt_tokens"] == 14075
assert usage["cache_read_input_tokens"] == 13922
def test_fold_cache_anthropic_dialect_folds_input_tokens() -> None:
"""Anthropic-native shape: input_tokens excludes cache, so it is folded."""
from routstr.upstream.base import BaseUpstreamProvider
usage = {
"input_tokens": 153,
"output_tokens": 24,
"cache_read_input_tokens": 13922,
"cache_creation_input_tokens": 0,
}
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
assert usage["input_tokens"] == 153 + 13922
def test_fold_cache_litellm_mirror_folds_only_input_tokens() -> None:
"""Both fields present (litellm mirror): fold input_tokens only."""
from routstr.upstream.base import BaseUpstreamProvider
usage = {
"prompt_tokens": 14075, # inclusive grand total
"input_tokens": 153, # additive Anthropic mirror
"cache_read_input_tokens": 13922,
}
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
assert usage["prompt_tokens"] == 14075
assert usage["input_tokens"] == 153 + 13922
# =========================================================================== # ===========================================================================
# get_cached_models / get_cached_model_by_id # get_cached_models / get_cached_model_by_id
# =========================================================================== # ===========================================================================