fix(generic): carry Venice's native cache_input price into input_cache_read

The generic provider's _native_pricing parses Venice's bespoke model_spec
schema but read only pricing.input/output.usd, dropping cache_input.usd
(e.g. deepseek-v4-1-flash: $0.0075/1M cache reads vs $0.375/1M input).
With input_cache_read left at 0, billing's fallback priced cache reads at
the FULL input rate — a ~50x overcharge on cache hits compared to what the
upstream charges.

cache_input.usd is now coerced with the same rules as the token rates;
a malformed/negative cache rate is treated as absent (never carried),
mirroring the OpenRouter rung's drop-don't-carry behaviour.

Adds regression tests: cache rate carried through, and malformed cache
rates (-, Infinity, non-numeric) dropped while the model still resolves.
This commit is contained in:
redshift
2026-09-21 16:26:01 +03:00
parent f5add71b0b
commit bc5b41b3bd
2 changed files with 93 additions and 0 deletions
+14
View File
@@ -91,6 +91,19 @@ class GenericUpstreamProvider(BaseUpstreamProvider):
if input_usd < 0 or output_usd < 0 or (input_usd == 0 and output_usd == 0):
return None
# Venice ships a discounted cache-read rate as ``cache_input`` (e.g.
# deepseek-v4-1-flash: $0.0075/1M vs $0.375/1M input). Dropping it
# left ``input_cache_read`` at 0, which billing reads as "no cache
# rate" and falls back to the FULL input rate — a 50x overcharge on
# cache hits. A malformed/negative cache rate coerces to None and is
# treated as absent, never carried (same rule as the OpenRouter rung).
cache_read_usd = _as_float(pricing_info.get("cache_input", {}).get("usd"))
input_cache_read = (
cache_read_usd / 1_000_000
if cache_read_usd is not None and cache_read_usd > 0
else 0.0
)
capabilities = model_spec.get("capabilities", {})
input_modalities = ["text"]
if capabilities.get("supportsVision", False):
@@ -101,6 +114,7 @@ class GenericUpstreamProvider(BaseUpstreamProvider):
completion=output_usd / 1_000_000,
context_length=model_spec.get("availableContextTokens"),
source="native",
input_cache_read=input_cache_read,
input_modalities=input_modalities,
)
+79
View File
@@ -111,6 +111,85 @@ async def test_native_model_spec_resolves_and_captures_metadata() -> None:
or_feed.assert_not_awaited()
@pytest.mark.asyncio
async def test_native_model_spec_carries_cache_read_price() -> None:
"""Venice's ``cache_input`` rate must reach ``input_cache_read``.
Regression: the native parser read only ``input``/``output`` and dropped
``cache_input``, leaving ``input_cache_read`` at 0 — which billing reads
as "no cache rate" and falls back to the FULL input rate, a ~50x
overcharge on cache hits (deepseek-v4-1-flash: $0.0075 vs $0.375 per 1M).
"""
payload = {
"data": [
{
"id": "venice-flash",
"owned_by": "venice",
"model_spec": {
"name": "Venice Flash",
"availableContextTokens": 1000000,
"pricing": {
"input": {"usd": 0.375},
"cache_input": {"usd": 0.0075},
"output": {"usd": 1.5},
},
},
}
]
}
with _patch_models_endpoint(payload):
with patch(
"routstr.payment.models.async_fetch_openrouter_models",
AsyncMock(return_value=[]),
):
models = await GenericUpstreamProvider(base_url="http://x").fetch_models()
model = _model_by_id(models, "venice-flash")
assert model.enabled is True
assert model.pricing.prompt == pytest.approx(0.375 / 1_000_000)
assert model.pricing.input_cache_read == pytest.approx(0.0075 / 1_000_000)
@pytest.mark.asyncio
@pytest.mark.parametrize("bad_rate", [-0.0075, "Infinity", "not-a-number"])
async def test_native_model_spec_drops_malformed_cache_rate(bad_rate: object) -> None:
"""A malformed ``cache_input`` costs the cache rate, not the model.
The token prices still resolve; the cache rate falls back to 0 (absent),
mirroring the OpenRouter rung's drop-don't-carry rule.
"""
payload = {
"data": [
{
"id": "venice-badcache",
"owned_by": "venice",
"model_spec": {
"name": "Venice Bad Cache",
"availableContextTokens": 65536,
"pricing": {
"input": {"usd": 0.375},
"cache_input": {"usd": bad_rate},
"output": {"usd": 1.5},
},
},
}
]
}
with _patch_models_endpoint(payload):
with patch(
"routstr.payment.models.async_fetch_openrouter_models",
AsyncMock(return_value=[]),
):
models = await GenericUpstreamProvider(base_url="http://x").fetch_models()
model = _model_by_id(models, "venice-badcache")
assert model.enabled is True
assert model.pricing.prompt == pytest.approx(0.375 / 1_000_000)
assert model.pricing.input_cache_read == 0.0
# ---------------------------------------------------------------------------
# native model_spec validation — a bogus native price is not authoritative
# ---------------------------------------------------------------------------