fix(pricing): ask the shared rate question when reading a feed price

The feed-price coercion rejected `NaN`/`inf` but returned a negative unchanged,
and the catalog import filter that would have caught one inspects only prompt
and completion. A negative cache rate was the one malformed value still
reaching a stored price, where it prices cached input tokens at a credit.

Both resolver rungs now ask `is_usable_rate` rather than spelling the rule out
again, so a feed price is held to the same standard as one arriving at the
admin edge. The both-zero rejection stays: that is a free price, not a
malformed one, and it is a separate question.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011cKHVF5LA7TR5QuYi6ErLM
This commit is contained in:
Jeroen Ubbink
2026-08-25 20:47:48 +02:00
co-authored by Claude Opus 5
parent b68086c592
commit fec229e1f4
2 changed files with 56 additions and 14 deletions
+18 -14
View File
@@ -15,7 +15,6 @@ it into the base provider unchanged.
from __future__ import annotations
import math
from dataclasses import dataclass, field
@@ -66,19 +65,23 @@ def estimate_context_length(model_id: str) -> int:
def _as_float(value: object) -> float | None:
"""OpenRouter reports prices as strings; coerce, ``None`` if not a real number.
"""OpenRouter reports prices as strings; coerce, ``None`` if not a real rate.
Non-finite values are rejected as unparseable: ``float("Infinity")`` and
``float("NaN")`` parse happily from a feed string, and ``json.loads``
accepts the bare literals and overflows ``1e999`` to ``inf``. An oversized
integer raises ``OverflowError`` rather than ``ValueError``, so that is
caught too.
Every caller reads a *price* out of a feed, so this asks the shared
billable-rate question rather than merely parsing: ``float("Infinity")`` and
``float("NaN")`` parse happily from a feed string, ``json.loads`` accepts the
bare literals and overflows ``1e999`` to ``inf``, and a negative parses
cleanly into a rate that credits the caller. An oversized integer raises
``OverflowError`` rather than ``ValueError``, so that is caught too.
"""
# Lazy, like the litellm lookup below: the resolver stays import-light.
from ..payment.models import is_usable_rate
try:
parsed = float(value) # type: ignore[arg-type]
except (TypeError, ValueError, OverflowError):
return None
return parsed if math.isfinite(parsed) else None
return parsed if is_usable_rate(parsed) else None
def _as_int(value: object) -> int | None:
@@ -89,7 +92,7 @@ def _as_int(value: object) -> int | None:
def _from_litellm(model_id: str) -> ResolvedPricing | None:
# Lazy import so the resolver stays import-light and shares the exact
# lookup semantics used by cache-rate backfill.
from ..payment.models import litellm_cost_entry
from ..payment.models import is_usable_rate, litellm_cost_entry
info = litellm_cost_entry(model_id)
if info is None:
@@ -103,12 +106,13 @@ def _from_litellm(model_id: str) -> ResolvedPricing | None:
# moderation/rerank tiers do this) — treating 0/0 as resolved would serve
# the model for free. Reject it (and any negative) so the caller falls
# through, mirroring async_fetch_openrouter_models' _has_valid_pricing.
# A non-finite entry is junk, not a price: `inf` would bill an infinite
# amount and `NaN` poisons every total it enters (and defeats the `< 0` and
# `== 0` guards below, since both comparisons are False for `NaN`).
if not math.isfinite(prompt) or not math.isfinite(completion):
# A malformed entry is junk, not a price: `inf` would bill an infinite
# amount, `NaN` poisons every total it enters, and a negative credits the
# caller. `NaN` also defeats the both-zero guard below on its own, since
# every comparison against it is False.
if not is_usable_rate(prompt) or not is_usable_rate(completion):
return None
if prompt < 0 or completion < 0 or (prompt == 0 and completion == 0):
if prompt == 0 and completion == 0:
return None
input_modalities = ["text"]
+38
View File
@@ -629,3 +629,41 @@ async def test_non_finite_openrouter_cache_rate_is_dropped_not_carried() -> None
assert model.enabled is True
assert model.pricing.prompt == pytest.approx(1e-06)
assert model.pricing.input_cache_read == 0.0
@pytest.mark.asyncio
async def test_negative_openrouter_cache_rate_is_dropped_not_carried() -> None:
"""A negative rate is as unusable as a non-finite one, and arrives the same way.
The coercion that reads a feed price rejected ``NaN``/``inf`` but returned a
negative unchanged, and the catalog import filter that would have caught one
inspects only prompt and completion. So a negative cache rate was the single
malformed value that still reached a stored price — where it prices cached
input tokens at a credit rather than a charge.
"""
payload = {
"data": [
{"id": "or-negcache-xyz", "object": "model", "owned_by": "mystery"},
]
}
feed = [
{
"id": "or-negcache-xyz",
"pricing": {
"prompt": "0.000001",
"completion": "0.000002",
"input_cache_read": "-0.0000005",
},
"context_length": 8192,
}
]
with _patch_models_endpoint(payload):
or_feed = AsyncMock(return_value=feed)
with patch("routstr.payment.models.async_fetch_openrouter_models", or_feed):
models = await GenericUpstreamProvider(base_url="http://x").fetch_models()
model = _model_by_id(models, "or-negcache-xyz")
assert model.enabled is True
assert model.pricing.prompt == pytest.approx(1e-06)
assert model.pricing.input_cache_read == 0.0