mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 20:28:23 +00:00
Merge pull request #752 from Routstr/fix/cache-fold-double-count
fix(usage): don't fold cache tokens into inclusive prompt_tokens
This commit is contained in:
@@ -433,6 +433,14 @@ class BaseUpstreamProvider:
|
||||
``cache_creation_input_tokens`` fields are left in place for clients
|
||||
that want the breakdown.
|
||||
|
||||
Which field may be folded mirrors ``normalize_usage`` exactly:
|
||||
Anthropic-native ``input_tokens`` *excludes* the cached portion and
|
||||
needs the roll-up, while a ``prompt_tokens`` grand total (OpenAI
|
||||
family, DeepSeek, OpenRouter, litellm) *already includes* it —
|
||||
folding there double-counts the cache in the visible prompt total
|
||||
(Venice showed 27997 prompt tokens for a 14075-token prompt after a
|
||||
13922-token cache read).
|
||||
|
||||
For Anthropic-shaped responses (``input_tokens`` present), the cache
|
||||
fields are forced to ``0`` when the upstream omitted them, so the
|
||||
client always sees a consistent shape.
|
||||
@@ -459,11 +467,10 @@ class BaseUpstreamProvider:
|
||||
usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
if "prompt_tokens" in usage:
|
||||
try:
|
||||
usage["prompt_tokens"] = int(usage.get("prompt_tokens") or 0) + extra
|
||||
except (TypeError, ValueError):
|
||||
pass
|
||||
# ``prompt_tokens`` is deliberately left untouched: in every dialect
|
||||
# that reports it, it is an inclusive grand total that already
|
||||
# contains the cached portion — the same assumption
|
||||
# ``normalize_usage`` subtracts against when billing.
|
||||
|
||||
def _apply_provider_field(self, response_json: object) -> None:
|
||||
"""Stamp the routstr ``provider`` field onto an upstream response payload.
|
||||
|
||||
@@ -116,6 +116,54 @@ def test_fold_cache_preserves_total() -> None:
|
||||
assert usage.prompt_tokens == 100
|
||||
|
||||
|
||||
def test_fold_cache_openai_dialect_prompt_tokens_untouched() -> None:
|
||||
"""Venice/OpenAI shape: prompt_tokens already includes cached tokens.
|
||||
|
||||
Regression: folding cache_read into prompt_tokens double-counted the
|
||||
cache (14075 real prompt shown as 27997 after a 13922-token cache read).
|
||||
"""
|
||||
from routstr.upstream.base import BaseUpstreamProvider
|
||||
|
||||
usage = {
|
||||
"prompt_tokens": 14075,
|
||||
"completion_tokens": 24,
|
||||
"total_tokens": 14099,
|
||||
"prompt_tokens_details": {"cached_tokens": 13922},
|
||||
"cache_read_input_tokens": 13922,
|
||||
}
|
||||
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
|
||||
assert usage["prompt_tokens"] == 14075
|
||||
assert usage["cache_read_input_tokens"] == 13922
|
||||
|
||||
|
||||
def test_fold_cache_anthropic_dialect_folds_input_tokens() -> None:
|
||||
"""Anthropic-native shape: input_tokens excludes cache, so it is folded."""
|
||||
from routstr.upstream.base import BaseUpstreamProvider
|
||||
|
||||
usage = {
|
||||
"input_tokens": 153,
|
||||
"output_tokens": 24,
|
||||
"cache_read_input_tokens": 13922,
|
||||
"cache_creation_input_tokens": 0,
|
||||
}
|
||||
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
|
||||
assert usage["input_tokens"] == 153 + 13922
|
||||
|
||||
|
||||
def test_fold_cache_litellm_mirror_folds_only_input_tokens() -> None:
|
||||
"""Both fields present (litellm mirror): fold input_tokens only."""
|
||||
from routstr.upstream.base import BaseUpstreamProvider
|
||||
|
||||
usage = {
|
||||
"prompt_tokens": 14075, # inclusive grand total
|
||||
"input_tokens": 153, # additive Anthropic mirror
|
||||
"cache_read_input_tokens": 13922,
|
||||
}
|
||||
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
|
||||
assert usage["prompt_tokens"] == 14075
|
||||
assert usage["input_tokens"] == 153 + 13922
|
||||
|
||||
|
||||
# ===========================================================================
|
||||
# get_cached_models / get_cached_model_by_id
|
||||
# ===========================================================================
|
||||
|
||||
Reference in New Issue
Block a user