mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 20:28:23 +00:00
fix(billing): a rate of zero is a price, not a missing price
The gate that decides a model cannot be priced on tokens asked `input_rate and output_rate`, so a rate of zero read as an absent one and the request was billed the whole reservation. The catalog serves such a model and the router routes it — this PR says so in as many words — and then billing charged it as if it had no price at all. For a model that is free on both sides the reservation is the minimum-request floor, so the overcharge is a msat. The bite is a model free on one side only: the reservation there is the context window priced at the paid rate, charged flat on every request no matter how few tokens it used. The catalog import filter rejects only a both-zero price, so those models are ingested, served and routed today. Ask only whether each rate is usable. A zero rate now bills as what it is. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_014X8RZzzbAuQCbavhFjTvJ4
This commit is contained in:
co-authored by
Claude Opus 5
parent
dff164b980
commit
337be4895e
@@ -270,11 +270,11 @@ async def calculate_cost(
|
||||
input_rate, output_rate, cache_read_rate, cache_creation_rate = pricing_rates
|
||||
|
||||
# Truthiness is not the question: `NaN` and a negative rate are both truthy
|
||||
# and sailed past this gate into the token math.
|
||||
# and sailed past this gate into the token math, while a rate of zero is a
|
||||
# price — free — and reading it as a missing one charged the whole
|
||||
# reservation for a request the model serves for nothing.
|
||||
rates = (input_rate, output_rate, cache_read_rate, cache_creation_rate)
|
||||
if not all(is_usable_rate(rate) for rate in rates) or not (
|
||||
input_rate and output_rate
|
||||
):
|
||||
if not all(is_usable_rate(rate) for rate in rates):
|
||||
logger.warning(
|
||||
"No usable token pricing — billing at flat MaxCostData. "
|
||||
"Token counts %s in the upstream response but cannot be "
|
||||
|
||||
@@ -92,6 +92,30 @@ async def test_unusable_token_rate_falls_back_to_max_cost(bad_rate: float) -> No
|
||||
assert cost.total_msats == 1234
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("prompt", "completion", "expected_msats"),
|
||||
[(0.0, 0.0, 0), (0.0, 2e-06, 1), (1e-06, 0.0, 1)],
|
||||
ids=["free", "free-input", "free-output"],
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_a_rate_of_zero_is_billed_as_free_not_as_missing(
|
||||
prompt: float, completion: float, expected_msats: int
|
||||
) -> None:
|
||||
"""Zero is a price, and the request must be billed on it.
|
||||
|
||||
The gate that decides a model has no token pricing was a truthiness test, so
|
||||
a free rate read as an absent one and the request was charged the whole
|
||||
reservation instead — on a model priced at zero for that side, which is a
|
||||
price the catalog serves and the router routes.
|
||||
"""
|
||||
model = _model(Pricing(prompt=prompt, completion=completion))
|
||||
|
||||
cost = await calculate_cost(_usage_response(), max_cost=1234, model_obj=model)
|
||||
|
||||
assert isinstance(cost, CostData)
|
||||
assert cost.total_msats == expected_msats
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"junk", [float("inf"), float("nan"), "Infinity"], ids=["inf", "nan", "inf-string"]
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user