mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 20:28:23 +00:00
Merge pull request #752 from Routstr/fix/cache-fold-double-count
fix(usage): don't fold cache tokens into inclusive prompt_tokens
This commit is contained in:
@@ -433,6 +433,14 @@ class BaseUpstreamProvider:
|
|||||||
``cache_creation_input_tokens`` fields are left in place for clients
|
``cache_creation_input_tokens`` fields are left in place for clients
|
||||||
that want the breakdown.
|
that want the breakdown.
|
||||||
|
|
||||||
|
Which field may be folded mirrors ``normalize_usage`` exactly:
|
||||||
|
Anthropic-native ``input_tokens`` *excludes* the cached portion and
|
||||||
|
needs the roll-up, while a ``prompt_tokens`` grand total (OpenAI
|
||||||
|
family, DeepSeek, OpenRouter, litellm) *already includes* it —
|
||||||
|
folding there double-counts the cache in the visible prompt total
|
||||||
|
(Venice showed 27997 prompt tokens for a 14075-token prompt after a
|
||||||
|
13922-token cache read).
|
||||||
|
|
||||||
For Anthropic-shaped responses (``input_tokens`` present), the cache
|
For Anthropic-shaped responses (``input_tokens`` present), the cache
|
||||||
fields are forced to ``0`` when the upstream omitted them, so the
|
fields are forced to ``0`` when the upstream omitted them, so the
|
||||||
client always sees a consistent shape.
|
client always sees a consistent shape.
|
||||||
@@ -459,11 +467,10 @@ class BaseUpstreamProvider:
|
|||||||
usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra
|
usage["input_tokens"] = int(usage.get("input_tokens") or 0) + extra
|
||||||
except (TypeError, ValueError):
|
except (TypeError, ValueError):
|
||||||
pass
|
pass
|
||||||
if "prompt_tokens" in usage:
|
# ``prompt_tokens`` is deliberately left untouched: in every dialect
|
||||||
try:
|
# that reports it, it is an inclusive grand total that already
|
||||||
usage["prompt_tokens"] = int(usage.get("prompt_tokens") or 0) + extra
|
# contains the cached portion — the same assumption
|
||||||
except (TypeError, ValueError):
|
# ``normalize_usage`` subtracts against when billing.
|
||||||
pass
|
|
||||||
|
|
||||||
def _apply_provider_field(self, response_json: object) -> None:
|
def _apply_provider_field(self, response_json: object) -> None:
|
||||||
"""Stamp the routstr ``provider`` field onto an upstream response payload.
|
"""Stamp the routstr ``provider`` field onto an upstream response payload.
|
||||||
|
|||||||
@@ -116,6 +116,54 @@ def test_fold_cache_preserves_total() -> None:
|
|||||||
assert usage.prompt_tokens == 100
|
assert usage.prompt_tokens == 100
|
||||||
|
|
||||||
|
|
||||||
|
def test_fold_cache_openai_dialect_prompt_tokens_untouched() -> None:
|
||||||
|
"""Venice/OpenAI shape: prompt_tokens already includes cached tokens.
|
||||||
|
|
||||||
|
Regression: folding cache_read into prompt_tokens double-counted the
|
||||||
|
cache (14075 real prompt shown as 27997 after a 13922-token cache read).
|
||||||
|
"""
|
||||||
|
from routstr.upstream.base import BaseUpstreamProvider
|
||||||
|
|
||||||
|
usage = {
|
||||||
|
"prompt_tokens": 14075,
|
||||||
|
"completion_tokens": 24,
|
||||||
|
"total_tokens": 14099,
|
||||||
|
"prompt_tokens_details": {"cached_tokens": 13922},
|
||||||
|
"cache_read_input_tokens": 13922,
|
||||||
|
}
|
||||||
|
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
|
||||||
|
assert usage["prompt_tokens"] == 14075
|
||||||
|
assert usage["cache_read_input_tokens"] == 13922
|
||||||
|
|
||||||
|
|
||||||
|
def test_fold_cache_anthropic_dialect_folds_input_tokens() -> None:
|
||||||
|
"""Anthropic-native shape: input_tokens excludes cache, so it is folded."""
|
||||||
|
from routstr.upstream.base import BaseUpstreamProvider
|
||||||
|
|
||||||
|
usage = {
|
||||||
|
"input_tokens": 153,
|
||||||
|
"output_tokens": 24,
|
||||||
|
"cache_read_input_tokens": 13922,
|
||||||
|
"cache_creation_input_tokens": 0,
|
||||||
|
}
|
||||||
|
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
|
||||||
|
assert usage["input_tokens"] == 153 + 13922
|
||||||
|
|
||||||
|
|
||||||
|
def test_fold_cache_litellm_mirror_folds_only_input_tokens() -> None:
|
||||||
|
"""Both fields present (litellm mirror): fold input_tokens only."""
|
||||||
|
from routstr.upstream.base import BaseUpstreamProvider
|
||||||
|
|
||||||
|
usage = {
|
||||||
|
"prompt_tokens": 14075, # inclusive grand total
|
||||||
|
"input_tokens": 153, # additive Anthropic mirror
|
||||||
|
"cache_read_input_tokens": 13922,
|
||||||
|
}
|
||||||
|
BaseUpstreamProvider._fold_cache_into_input_tokens(usage)
|
||||||
|
assert usage["prompt_tokens"] == 14075
|
||||||
|
assert usage["input_tokens"] == 153 + 13922
|
||||||
|
|
||||||
|
|
||||||
# ===========================================================================
|
# ===========================================================================
|
||||||
# get_cached_models / get_cached_model_by_id
|
# get_cached_models / get_cached_model_by_id
|
||||||
# ===========================================================================
|
# ===========================================================================
|
||||||
|
|||||||
Reference in New Issue
Block a user