From ef4c92c7365a9e58e1fa0c920a7856b3957582b3 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Sat, 3 Oct 2026 14:58:01 +0200 Subject: [PATCH 1/4] fix: bill endpoint-pinned requests at the pinned endpoint's own pricing --- routstr/core/admin.py | 9 +- routstr/proxy.py | 46 +++++++ routstr/upstream/certification.py | 6 +- routstr/upstream/certification_cache.py | 34 +---- routstr/upstream/model_paths.py | 6 +- tests/integration/test_certify_endpoint.py | 21 ++-- tests/unit/test_certification_cache.py | 41 ------ tests/unit/test_model_path_routing.py | 138 ++++++++++++++++++++- 8 files changed, 201 insertions(+), 100 deletions(-) diff --git a/routstr/core/admin.py b/routstr/core/admin.py index 2041f409..a36aa7bf 100644 --- a/routstr/core/admin.py +++ b/routstr/core/admin.py @@ -1724,14 +1724,12 @@ async def certify_upstream_provider( status_code=503, detail="sats/USD price is not initialized yet; retry shortly", ) - # The proxy reserves and token-bills a pinned request with the model's - # own pricing, so the cost rows use it too; the path's advertised - # endpoint rates are only compared against it in the margin row. - advertised_model = None + # The proxy reserves and token-bills a pinned endpoint at that + # endpoint's own rates, so the cost rows price it the same way. if selected_path is not None: from ..upstream.model_paths import apply_model_path_pricing - advertised_model = apply_model_path_pricing( + model_obj = apply_model_path_pricing( model_obj, selected_path, provider.provider_fee, @@ -1750,7 +1748,6 @@ async def certify_upstream_provider( check_cache=payload.check_cache, endpoint_tag=endpoint_tag, upstream=upstream_obj, - advertised_model=advertised_model, ) rows = pricing_rows + live_rows diff --git a/routstr/proxy.py b/routstr/proxy.py index a2cf9492..9faa1acb 100644 --- a/routstr/proxy.py +++ b/routstr/proxy.py @@ -19,6 +19,7 @@ from .core import get_logger from .core.db import ( ApiKey, AsyncSession, + ModelPathRow, ModelRow, UpstreamProviderRow, create_session, @@ -52,6 +53,7 @@ from .upstream.ehbp import forward_ehbp_request, forward_ehbp_x_cashu_request from .upstream.helpers import init_upstreams from .upstream.model_paths import ( ModelPathSelector, + apply_model_path_pricing, decode_model_path, is_openrouter_base_url, public_model_id, @@ -157,6 +159,44 @@ def _candidate_for_selector( return None +async def _price_pinned_endpoint( + session: AsyncSession, + selector: ModelPathSelector, + model_obj: Model, + upstream: BaseUpstreamProvider, +) -> Model: + """Reprice ``model_obj`` with the pinned endpoint's own rates. + + An OpenRouter endpoint can cost more than the model's default listing, and + ``/v1/models/paths`` quotes that endpoint's price, so the reservation and + token billing must use it too. Without a stored path row or a sats price + the request keeps the model's default pricing. + """ + from .payment import price as price_module + + sats_to_usd = price_module.SATS_USD_PRICE + if upstream.db_id is None or not sats_to_usd: + return model_obj + rows = ( + await session.exec( + select(ModelPathRow).where( + ModelPathRow.upstream_provider_id == upstream.db_id, + ModelPathRow.endpoint_tag == selector.endpoint_tag, + ) + ) + ).all() + row = next( + (r for r in rows if _model_ids_match(r.model_id, selector.model_id)), None + ) + if row is None: + logger.warning( + "No stored path for pinned endpoint; billing the model's default pricing", + extra={"model": selector.model_id, "endpoint": selector.endpoint_tag}, + ) + return model_obj + return apply_model_path_pricing(model_obj, row, upstream.provider_fee, sats_to_usd) + + def get_model_instance(model_id: str) -> Model | None: """Get the best-ranked Model instance for a model ID.""" candidates = get_candidates(model_id) @@ -702,6 +742,12 @@ async def _proxy( }, } request_body = json.dumps(request_body_dict).encode() + candidates = [ + ( + await _price_pinned_endpoint(session, selector, *pinned), + pinned[1], + ) + ] if is_ehbp: candidates = [ diff --git a/routstr/upstream/certification.py b/routstr/upstream/certification.py index 180ab239..9b00660b 100644 --- a/routstr/upstream/certification.py +++ b/routstr/upstream/certification.py @@ -861,15 +861,14 @@ async def run_live_checks( check_cache: bool = True, endpoint_tag: str | None = None, upstream: "BaseUpstreamProvider | None" = None, - advertised_model: "Model | None" = None, ) -> list[dict[str, Any]]: """Probe one upstream and build the live/derived rows. ``check_cache`` adds the prompt-cache and margin rows, which cost two or three more completions against a long prompt. ``upstream`` shapes the probes like the proxy's own requests; without it they assume a plain - OpenAI-compatible base URL. ``advertised_model`` carries a pinned path's - own endpoint rates for the margin row to compare against ``model``'s. + OpenAI-compatible base URL. On a pinned endpoint ``model`` carries that + endpoint's own rates, as the proxy bills it. """ probe = await probe_upstream( base_url, @@ -973,7 +972,6 @@ async def run_live_checks( pricing_known=pricing_known, endpoint_tag=endpoint_tag, upstream=upstream, - advertised_model=advertised_model, token_limit_field=probe.token_limit_field, ) ) diff --git a/routstr/upstream/certification_cache.py b/routstr/upstream/certification_cache.py index fef1d8a4..24751712 100644 --- a/routstr/upstream/certification_cache.py +++ b/routstr/upstream/certification_cache.py @@ -446,7 +446,6 @@ def cost_margin_row( provider_fee: float, sats_to_usd: float, pricing_known: bool = True, - advertised_model: Model | None = None, ) -> dict[str, Any]: """Configured token pricing must cover what the upstream reports charging. @@ -457,14 +456,9 @@ def cost_margin_row( falls below the fee-adjusted upstream cost means those paths underprice. Upstreams that report no cost give no sample and the row stays a warn. - ``model`` carries the pricing the proxy reserves and token-bills with. On - a pinned path, ``advertised_model`` carries the endpoint's own rates; a - covered margin whose advertised rates differ from the billed ones is a - warn, since ``/v1/models/paths`` then shows a price the node does not bill. + ``model`` carries the pricing the proxy reserves and token-bills with, + which on a pinned endpoint is that endpoint's own rates. """ - advertised_pricing = ( - advertised_model.sats_pricing if advertised_model is not None else None - ) evidence: dict[str, Any] = { "model_id": model.id, "provider_fee": provider_fee, @@ -483,7 +477,6 @@ def cost_margin_row( samples: list[dict[str, Any]] = [] short: list[str] = [] - mismatched: list[str] = [] for payload in payloads: if not isinstance(payload, dict): continue @@ -496,11 +489,6 @@ def cost_margin_row( upstream_total = _expected_usd_msats( reported_usd, provider_fee, sats_to_usd ) - advertised_total = ( - _expected_token_msats(advertised_pricing, usage)[0] - if advertised_pricing is not None - else None - ) except (ValueError, OverflowError) as exc: evidence["error"] = f"{type(exc).__name__}: {exc}" return certification_row( @@ -516,10 +504,6 @@ def cost_margin_row( "upstream_msats_with_fee": upstream_total, "configured_msats": configured_total, } - if advertised_total is not None: - sample["advertised_msats"] = advertised_total - if abs(advertised_total - configured_total) > COST_TOLERANCE_MSATS: - mismatched.append(f"{advertised_total} vs {configured_total}") samples.append(sample) if configured_total + COST_TOLERANCE_MSATS < upstream_total: short.append(f"{configured_total} < {upstream_total}") @@ -545,18 +529,6 @@ def cost_margin_row( "requests lose money.", evidence, ) - if mismatched: - return certification_row( - ROW_MARGIN, - STATUS_WARN, - TITLE_MARGIN, - f"Configured pricing covers the upstream's reported cost on " - f"{len(samples)} sampled completion(s), but this path advertises " - f"different endpoint rates (advertised vs billed msats: " - f"{'; '.join(mismatched)}); the proxy reserves and token-bills " - "pinned requests with the model's own pricing.", - evidence, - ) return certification_row( ROW_MARGIN, STATUS_OK, @@ -597,7 +569,6 @@ async def run_cache_checks( pricing_known: bool = True, endpoint_tag: str | None = None, upstream: "BaseUpstreamProvider | None" = None, - advertised_model: Model | None = None, token_limit_field: str = "max_tokens", ) -> list[dict[str, Any]]: """Run the cache probe and build the three cache/margin rows.""" @@ -634,7 +605,6 @@ async def run_cache_checks( provider_fee=provider_fee, sats_to_usd=sats_to_usd, pricing_known=pricing_known, - advertised_model=advertised_model, ), ), ] diff --git a/routstr/upstream/model_paths.py b/routstr/upstream/model_paths.py index bcda8d39..de0e7f81 100644 --- a/routstr/upstream/model_paths.py +++ b/routstr/upstream/model_paths.py @@ -898,8 +898,8 @@ def apply_model_path_pricing( Direct paths already use the provider model cache and therefore carry the same pricing as ``model``. OpenRouter endpoint rows instead contain raw, - endpoint-specific USD rates; certification compares them against the - model's own pricing, which the proxy reserves and token-bills with. + endpoint-specific USD rates, which can differ from the model's default + listing; the proxy reserves and token-bills a pinned endpoint with them. """ if row.endpoint_tag is None: return model @@ -933,7 +933,7 @@ def apply_model_path_pricing( return _update_model_sats_pricing(priced, sats_to_usd) except Exception as exc: logger.warning( - "Could not apply model-path pricing for certification", + "Could not apply model-path pricing", extra={"model_id": model.id, "path": row.path, "error": str(exc)}, ) return model diff --git a/tests/integration/test_certify_endpoint.py b/tests/integration/test_certify_endpoint.py index 9c8b2125..74b25b77 100644 --- a/tests/integration/test_certify_endpoint.py +++ b/tests/integration/test_certify_endpoint.py @@ -328,7 +328,7 @@ async def test_certify_model_path_pins_every_completion( @pytest.mark.integration @pytest.mark.asyncio @respx.mock -async def test_certify_margin_bills_model_pricing_and_reports_path_pricing( +async def test_certify_margin_bills_pinned_path_pricing( integration_client: AsyncClient, integration_session: AsyncSession ) -> None: base_url = "https://openrouter.ai/api/v1" @@ -409,20 +409,15 @@ async def test_certify_margin_bills_model_pricing_and_reports_path_pricing( assert resp.status_code == 200, resp.text margin = _find_row(resp.json()["rows"], "cost.margin") - # The proxy reserves and token-bills a pinned request with the model's own - # pricing (``configured_msats``); the path's endpoint rates are reported - # alongside (``advertised_msats``) and differ, so the covered margin warns. + # The proxy bills a pinned endpoint at its own rates, not the model's + # (which would give 3, 356 and 43); the endpoint's cache-read rate falls + # short of what it reported charging on the cached call. assert [ - ( - sample["upstream_msats_with_fee"], - sample["configured_msats"], - sample["advertised_msats"], - ) + (sample["upstream_msats_with_fee"], sample["configured_msats"]) for sample in margin["evidence"]["samples"] - ] == [(3, 3, 3), (269, 356, 289), (26, 43, 15)] - assert margin["status"] == "warn" - assert "advertises different endpoint rates" in margin["detail"] - assert "289 vs 356" in margin["detail"] + ] == [(3, 3), (269, 289), (26, 15)] + assert margin["status"] == "fail" + assert "15 < 26" in margin["detail"] @pytest.mark.integration diff --git a/tests/unit/test_certification_cache.py b/tests/unit/test_certification_cache.py index a1681aeb..2d87a8cf 100644 --- a/tests/unit/test_certification_cache.py +++ b/tests/unit/test_certification_cache.py @@ -333,47 +333,6 @@ class TestCostMarginRow: assert row["status"] == STATUS_OK - def test_pinned_path_fails_when_billed_pricing_misses_cost(self) -> None: - """The path's endpoint rates cover the cost but the model pricing the - proxy actually bills with does not: the margin must fail.""" - payload = _payload({"prompt_tokens": 5, "completion_tokens": 1, "cost": 4e-6}) - row = cost_margin_row( - model=_model(), - payloads=[payload], - provider_fee=1.0, - sats_to_usd=SATS_USD, - advertised_model=_model(prompt=1e-6, completion=2e-6), - ) - assert row["status"] == STATUS_FAIL - sample = row["evidence"]["samples"][0] - assert sample["advertised_msats"] >= sample["upstream_msats_with_fee"] - assert sample["configured_msats"] < sample["upstream_msats_with_fee"] - - def test_pinned_path_warns_when_advertised_rates_differ(self) -> None: - payload = _payload({"prompt_tokens": 5, "completion_tokens": 1, "cost": 9e-7}) - row = cost_margin_row( - model=_model(), - payloads=[payload], - provider_fee=1.0, - sats_to_usd=SATS_USD, - advertised_model=_model(prompt=1e-6, completion=2e-6), - ) - assert row["status"] == STATUS_WARN - assert "advertises different endpoint rates" in row["detail"] - - def test_pinned_path_ok_when_advertised_rates_match(self) -> None: - payload = _payload({"prompt_tokens": 5, "completion_tokens": 1, "cost": 9e-7}) - row = cost_margin_row( - model=_model(), - payloads=[payload], - provider_fee=1.0, - sats_to_usd=SATS_USD, - advertised_model=_model(), - ) - assert row["status"] == STATUS_OK - sample = row["evidence"]["samples"][0] - assert sample["advertised_msats"] == sample["configured_msats"] - def test_warn_when_pricing_unknown(self) -> None: payload = _payload({"prompt_tokens": 5, "completion_tokens": 1, "cost": 9e-7}) row = cost_margin_row( diff --git a/tests/unit/test_model_path_routing.py b/tests/unit/test_model_path_routing.py index cb764289..4cadacd7 100644 --- a/tests/unit/test_model_path_routing.py +++ b/tests/unit/test_model_path_routing.py @@ -50,6 +50,7 @@ async def _run_proxy( request: MagicMock, candidates: list[tuple[Any, Any]], path: str = "v1/chat/completions", + session: Any = None, ) -> Any: key = ApiKey(hashed_key="mpkey", balance=10_000) reservation = ReservationSnapshot( @@ -74,7 +75,7 @@ async def _run_proxy( proxy_module, "pay_for_request", AsyncMock(return_value=reservation) ), patch.object(proxy_module, "revert_pay_for_request", AsyncMock()), - patch_proxy_session(MagicMock()), + patch_proxy_session(session if session is not None else MagicMock()), ): return await proxy_module.proxy(request, path) @@ -845,3 +846,138 @@ async def test_node_fault_stays_500_without_scope_header() -> None: assert ERROR_SCOPE_HEADER not in response.headers body = json.loads(bytes(response.body)) assert body["error"]["code"] != UPSTREAM_UNAVAILABLE + + +_OPENROUTER = "https://openrouter.ai/api/v1" +_SATS_USD = 0.001 +_ENDPOINT_PRICING = {"prompt": 2e-6, "completion": 4e-6} + + +def _priced_model() -> Any: + from routstr.payment.models import ( + Architecture, + Model, + Pricing, + _calculate_usd_max_costs, + _update_model_sats_pricing, + ) + + model = Model( + id=MODEL_ID, + name=MODEL_ID, + created=0, + description="", + context_length=8192, + architecture=Architecture( + modality="text", + input_modalities=["text"], + output_modalities=["text"], + tokenizer="unknown", + instruct_type=None, + ), + pricing=Pricing(prompt=1e-6, completion=2e-6), + ) + ( + model.pricing.max_prompt_cost, + model.pricing.max_completion_cost, + model.pricing.max_cost, + ) = _calculate_usd_max_costs(model) + return _update_model_sats_pricing(model, _SATS_USD) + + +def _endpoint_row(model_id: str = MODEL_ID, endpoint_tag: str = "deepinfra/fp8") -> Any: + from routstr.core.db import ModelPathRow + + return ModelPathRow( + model_id=model_id, + path=encode_model_path(_OPENROUTER, model_id, endpoint_tag), + provider_slug="openrouter", + provider_type="openrouter", + endpoint_tag=endpoint_tag, + model_metadata=json.dumps({"id": model_id, "pricing": _ENDPOINT_PRICING}), + upstream_provider_id=1, + ) + + +def _session_with_rows(rows: list[Any]) -> MagicMock: + session = MagicMock() + session.exec = AsyncMock(return_value=MagicMock(all=MagicMock(return_value=rows))) + return session + + +def _endpoint_selector() -> Any: + selector = decode_model_path( + encode_model_path(_OPENROUTER, MODEL_ID, "deepinfra/fp8") + ) + assert selector is not None + return selector + + +def _openrouter_upstream() -> MagicMock: + upstream = _make_upstream(1) + upstream.base_url = _OPENROUTER + upstream.provider_fee = 1.0 + return upstream + + +@pytest.mark.asyncio +async def test_endpoint_pin_bills_the_endpoint_pricing() -> None: + """A pinned endpoint is reserved and billed at the rates + ``/v1/models/paths`` quotes for it, not the model's default listing.""" + model = _priced_model() + upstream = _openrouter_upstream() + request = _make_request( + { + "authorization": "Bearer sk-mpkey", + "x-routstr-model-path": encode_model_path( + _OPENROUTER, MODEL_ID, "deepinfra/fp8" + ), + }, + json.dumps({"model": MODEL_ID}).encode(), + ) + + with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): + await _run_proxy( + request, [(model, upstream)], session=_session_with_rows([_endpoint_row()]) + ) + + billed = upstream.forward_request.await_args.args[7] + assert billed.sats_pricing.prompt == pytest.approx(2e-6 / _SATS_USD) + assert billed.sats_pricing.completion == pytest.approx(4e-6 / _SATS_USD) + assert billed.sats_pricing.max_cost > model.sats_pricing.max_cost + assert model.sats_pricing.prompt == pytest.approx(1e-6 / _SATS_USD) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "rows", + [[], [_endpoint_row(model_id="other-model")]], + ids=["no-stored-path", "other-model"], +) +async def test_endpoint_pin_without_a_stored_path_keeps_model_pricing( + rows: list[Any], +) -> None: + model = _priced_model() + with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): + priced = await proxy_module._price_pinned_endpoint( + _session_with_rows(rows), + _endpoint_selector(), + model, + _openrouter_upstream(), + ) + assert priced is model + + +@pytest.mark.asyncio +async def test_endpoint_pin_without_sats_price_keeps_model_pricing() -> None: + model = _priced_model() + session = _session_with_rows([_endpoint_row()]) + with patch("routstr.payment.price.SATS_USD_PRICE", None): + priced = await proxy_module._price_pinned_endpoint( + session, + _endpoint_selector(), + model, + _openrouter_upstream(), + ) + assert priced is model + session.exec.assert_not_awaited() From a09a5f1b38239c381b111f47431747c9a87f717d Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Sat, 3 Oct 2026 15:55:49 +0200 Subject: [PATCH 2/4] fix: reserve pinned endpoints at their own limits and floor them at operator overrides --- routstr/core/admin.py | 16 ++--- routstr/proxy.py | 6 +- routstr/upstream/model_paths.py | 69 +++++++++++++++++--- tests/integration/test_certify_endpoint.py | 13 ++-- tests/unit/test_model_path_routing.py | 73 ++++++++++++++++++++-- 5 files changed, 149 insertions(+), 28 deletions(-) diff --git a/routstr/core/admin.py b/routstr/core/admin.py index a36aa7bf..f6f24db6 100644 --- a/routstr/core/admin.py +++ b/routstr/core/admin.py @@ -1727,14 +1727,16 @@ async def certify_upstream_provider( # The proxy reserves and token-bills a pinned endpoint at that # endpoint's own rates, so the cost rows price it the same way. if selected_path is not None: - from ..upstream.model_paths import apply_model_path_pricing + from ..upstream.model_paths import price_pinned_endpoint - model_obj = apply_model_path_pricing( - model_obj, - selected_path, - provider.provider_fee, - sats_to_usd, - ) + async with create_session() as session: + model_obj = await price_pinned_endpoint( + session, + model_obj, + selected_path, + provider.provider_fee, + sats_to_usd, + ) # The timeout applies per upstream call. The run makes up to six # calls (models, two short probes after a max_completion_tokens retry, # three cache probes), so the request can stay open for six times it. diff --git a/routstr/proxy.py b/routstr/proxy.py index 9faa1acb..7c210e0c 100644 --- a/routstr/proxy.py +++ b/routstr/proxy.py @@ -53,9 +53,9 @@ from .upstream.ehbp import forward_ehbp_request, forward_ehbp_x_cashu_request from .upstream.helpers import init_upstreams from .upstream.model_paths import ( ModelPathSelector, - apply_model_path_pricing, decode_model_path, is_openrouter_base_url, + price_pinned_endpoint, public_model_id, public_provider_url, ) @@ -194,7 +194,9 @@ async def _price_pinned_endpoint( extra={"model": selector.model_id, "endpoint": selector.endpoint_tag}, ) return model_obj - return apply_model_path_pricing(model_obj, row, upstream.provider_fee, sats_to_usd) + return await price_pinned_endpoint( + session, model_obj, row, upstream.provider_fee, sats_to_usd + ) def get_model_instance(model_id: str) -> Model | None: diff --git a/routstr/upstream/model_paths.py b/routstr/upstream/model_paths.py index de0e7f81..581cf6e4 100644 --- a/routstr/upstream/model_paths.py +++ b/routstr/upstream/model_paths.py @@ -38,7 +38,7 @@ from ..core.logging import get_logger if TYPE_CHECKING: from sqlmodel.ext.asyncio.session import AsyncSession - from ..payment.models import Model + from ..payment.models import Model, Pricing from .base import BaseUpstreamProvider logger = get_logger(__name__) @@ -893,19 +893,26 @@ def apply_model_path_pricing( row: ModelPathRow, provider_fee: float, sats_to_usd: float, + floor: "Pricing | None" = None, ) -> "Model": """Return ``model`` priced from an exact endpoint path's own rates. Direct paths already use the provider model cache and therefore carry the same pricing as ``model``. OpenRouter endpoint rows instead contain raw, - endpoint-specific USD rates, which can differ from the model's default - listing; the proxy reserves and token-bills a pinned endpoint with them. + endpoint-specific USD rates and limits, which can differ from the model's + default listing; OpenRouter charges the endpoint that serves the request, + so the proxy reserves and token-bills a pinned endpoint with them, using + the same limits ``/v1/models/paths`` quotes its max cost from. + + ``floor`` is an operator price override: no rate is billed below it, so + pinning an endpoint cannot bypass the operator's pricing. """ if row.endpoint_tag is None: return model from ..payment.models import ( Pricing, + TopProvider, _calculate_usd_max_costs, _update_model_sats_pricing, backfill_cache_pricing, @@ -921,10 +928,27 @@ def apply_model_path_pricing( model.forwarded_model_id or row.model_id, Pricing.parse_obj(metadata["pricing"]), ) - pricing = Pricing.parse_obj( - {key: float(value) * provider_fee for key, value in pricing.dict().items()} - ) - priced = model.copy(update={"pricing": pricing, "sats_pricing": None}) + rates = { + key: float(value) * provider_fee + for key, value in pricing.dict().items() + if not key.startswith("max_") + } + if floor is not None: + rates = { + key: max(value, float(getattr(floor, key))) + for key, value in rates.items() + } + pricing = Pricing.parse_obj(rates) + update: dict[str, Any] = {"pricing": pricing, "sats_pricing": None} + context_length = metadata.get("context_length") + max_completion_tokens = metadata.get("max_completion_tokens") + if context_length or max_completion_tokens: + update["context_length"] = context_length or model.context_length + update["top_provider"] = TopProvider( + context_length=context_length, + max_completion_tokens=max_completion_tokens, + ) + priced = model.copy(update=update) ( pricing.max_prompt_cost, pricing.max_completion_cost, @@ -939,6 +963,37 @@ def apply_model_path_pricing( return model +async def price_pinned_endpoint( + session: AsyncSession, + model: "Model", + row: ModelPathRow, + provider_fee: float, + sats_to_usd: float, +) -> "Model": + """Price ``model`` for a request pinned to ``row``'s endpoint. + + An enabled operator override for the model on this provider is the + price floor; ``model`` already carries it, since overrides replace the + provider's model in routing. + """ + override = ( + await session.exec( + select(ModelRow).where( + ModelRow.id == model.id, + ModelRow.upstream_provider_id == row.upstream_provider_id, + ModelRow.enabled, + ) + ) + ).first() + return apply_model_path_pricing( + model, + row, + provider_fee, + sats_to_usd, + floor=model.pricing if override is not None else None, + ) + + def _serialize_path(row: ModelPathRow, provider_fee: float) -> dict[str, Any]: endpoint = None if row.endpoint_tag or row.endpoint_name: diff --git a/tests/integration/test_certify_endpoint.py b/tests/integration/test_certify_endpoint.py index 74b25b77..8d338850 100644 --- a/tests/integration/test_certify_endpoint.py +++ b/tests/integration/test_certify_endpoint.py @@ -409,15 +409,16 @@ async def test_certify_margin_bills_pinned_path_pricing( assert resp.status_code == 200, resp.text margin = _find_row(resp.json()["rows"], "cost.margin") - # The proxy bills a pinned endpoint at its own rates, not the model's - # (which would give 3, 356 and 43); the endpoint's cache-read rate falls - # short of what it reported charging on the cached call. + # The proxy bills a pinned endpoint at its own rates, with the operator's + # override as the floor per rate: the endpoint's prompt rate, the + # override's completion and cache-read rates. The model's rates alone + # would give 3, 356 and 43; the endpoint's alone 3, 289 and 15, which + # misses the cached call's reported cost of 26. assert [ (sample["upstream_msats_with_fee"], sample["configured_msats"]) for sample in margin["evidence"]["samples"] - ] == [(3, 3), (269, 289), (26, 15)] - assert margin["status"] == "fail" - assert "15 < 26" in margin["detail"] + ] == [(3, 3), (269, 289), (26, 27)] + assert margin["status"] == "ok" @pytest.mark.integration diff --git a/tests/unit/test_model_path_routing.py b/tests/unit/test_model_path_routing.py index 4cadacd7..765ed216 100644 --- a/tests/unit/test_model_path_routing.py +++ b/tests/unit/test_model_path_routing.py @@ -853,7 +853,7 @@ _SATS_USD = 0.001 _ENDPOINT_PRICING = {"prompt": 2e-6, "completion": 4e-6} -def _priced_model() -> Any: +def _priced_model(prompt: float = 1e-6, completion: float = 2e-6) -> Any: from routstr.payment.models import ( Architecture, Model, @@ -875,7 +875,7 @@ def _priced_model() -> Any: tokenizer="unknown", instruct_type=None, ), - pricing=Pricing(prompt=1e-6, completion=2e-6), + pricing=Pricing(prompt=prompt, completion=completion), ) ( model.pricing.max_prompt_cost, @@ -885,7 +885,11 @@ def _priced_model() -> Any: return _update_model_sats_pricing(model, _SATS_USD) -def _endpoint_row(model_id: str = MODEL_ID, endpoint_tag: str = "deepinfra/fp8") -> Any: +def _endpoint_row( + model_id: str = MODEL_ID, + endpoint_tag: str = "deepinfra/fp8", + **limits: int, +) -> Any: from routstr.core.db import ModelPathRow return ModelPathRow( @@ -894,14 +898,21 @@ def _endpoint_row(model_id: str = MODEL_ID, endpoint_tag: str = "deepinfra/fp8") provider_slug="openrouter", provider_type="openrouter", endpoint_tag=endpoint_tag, - model_metadata=json.dumps({"id": model_id, "pricing": _ENDPOINT_PRICING}), + model_metadata=json.dumps( + {"id": model_id, "pricing": _ENDPOINT_PRICING, **limits} + ), upstream_provider_id=1, ) -def _session_with_rows(rows: list[Any]) -> MagicMock: +def _session_with_rows(rows: list[Any], override: Any = None) -> MagicMock: + """Answers the path-row query with ``rows``, the override one with ``override``.""" session = MagicMock() - session.exec = AsyncMock(return_value=MagicMock(all=MagicMock(return_value=rows))) + session.exec = AsyncMock( + return_value=MagicMock( + all=MagicMock(return_value=rows), first=MagicMock(return_value=override) + ) + ) return session @@ -981,3 +992,53 @@ async def test_endpoint_pin_without_sats_price_keeps_model_pricing() -> None: ) assert priced is model session.exec.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_endpoint_pin_reserves_the_max_cost_paths_quotes() -> None: + """The reservation uses the endpoint's own context and completion limits, + the same ones ``/v1/models/paths`` quotes its max cost from.""" + from routstr.upstream.model_paths import _serialize_path + + row = _endpoint_row(context_length=32768, max_completion_tokens=8192) + with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): + priced = await proxy_module._price_pinned_endpoint( + _session_with_rows([row]), + _endpoint_selector(), + _priced_model(), + _openrouter_upstream(), + ) + quoted = _serialize_path(row, 1.0)["model"]["sats_pricing"]["max_cost"] + + assert priced.sats_pricing is not None and priced.top_provider is not None + assert priced.sats_pricing.max_cost == pytest.approx(quoted) + assert priced.top_provider.max_completion_tokens == 8192 + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("override", "expected"), + [ + # The operator's override is above the endpoint: it stays the price. + ((6e-6, 12e-6), (6e-6, 12e-6)), + # The endpoint costs more than the override: never bill below cost. + ((1e-6, 2e-6), (2e-6, 4e-6)), + # Each rate takes the higher of the two. + ((3e-6, 1e-6), (3e-6, 4e-6)), + ], +) +async def test_endpoint_pin_never_bills_below_an_operator_override( + override: tuple[float, float], expected: tuple[float, float] +) -> None: + from routstr.core.db import ModelRow + + model = _priced_model(*override) + with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): + priced = await proxy_module._price_pinned_endpoint( + _session_with_rows([_endpoint_row()], override=MagicMock(spec=ModelRow)), + _endpoint_selector(), + model, + _openrouter_upstream(), + ) + + assert (priced.pricing.prompt, priced.pricing.completion) == pytest.approx(expected) From 9068d681d7eb95725862b6b09fd84caf76a05f0d Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Sat, 3 Oct 2026 17:28:46 +0200 Subject: [PATCH 3/4] fix: compare effective cache rates when flooring pinned endpoints at an override --- routstr/upstream/model_paths.py | 21 +++++++++-- tests/unit/test_model_path_routing.py | 50 +++++++++++++++++++++++++-- 2 files changed, 66 insertions(+), 5 deletions(-) diff --git a/routstr/upstream/model_paths.py b/routstr/upstream/model_paths.py index 581cf6e4..a72fe807 100644 --- a/routstr/upstream/model_paths.py +++ b/routstr/upstream/model_paths.py @@ -934,9 +934,12 @@ def apply_model_path_pricing( if not key.startswith("max_") } if floor is not None: + floor_rates = _effective_cache_rates( + {key: float(getattr(floor, key)) for key in rates} + ) rates = { - key: max(value, float(getattr(floor, key))) - for key, value in rates.items() + key: max(value, floor_rates[key]) + for key, value in _effective_cache_rates(rates).items() } pricing = Pricing.parse_obj(rates) update: dict[str, Any] = {"pricing": pricing, "sats_pricing": None} @@ -963,6 +966,20 @@ def apply_model_path_pricing( return model +def _effective_cache_rates(rates: dict[str, float]) -> dict[str, float]: + """Spell out cache rates settlement reads as "bill at the prompt rate". + + A zero cache rate is billed at the prompt rate, so a per-rate ``max`` + must compare those prompt rates, not the zeros. + """ + return { + key: value + if value > 0 or key not in ("input_cache_read", "input_cache_write") + else rates["prompt"] + for key, value in rates.items() + } + + async def price_pinned_endpoint( session: AsyncSession, model: "Model", diff --git a/tests/unit/test_model_path_routing.py b/tests/unit/test_model_path_routing.py index 765ed216..0f12eef9 100644 --- a/tests/unit/test_model_path_routing.py +++ b/tests/unit/test_model_path_routing.py @@ -853,7 +853,9 @@ _SATS_USD = 0.001 _ENDPOINT_PRICING = {"prompt": 2e-6, "completion": 4e-6} -def _priced_model(prompt: float = 1e-6, completion: float = 2e-6) -> Any: +def _priced_model( + prompt: float = 1e-6, completion: float = 2e-6, cache_read: float = 0.0 +) -> Any: from routstr.payment.models import ( Architecture, Model, @@ -875,7 +877,9 @@ def _priced_model(prompt: float = 1e-6, completion: float = 2e-6) -> Any: tokenizer="unknown", instruct_type=None, ), - pricing=Pricing(prompt=prompt, completion=completion), + pricing=Pricing( + prompt=prompt, completion=completion, input_cache_read=cache_read + ), ) ( model.pricing.max_prompt_cost, @@ -888,6 +892,7 @@ def _priced_model(prompt: float = 1e-6, completion: float = 2e-6) -> Any: def _endpoint_row( model_id: str = MODEL_ID, endpoint_tag: str = "deepinfra/fp8", + pricing: dict[str, float] | None = None, **limits: int, ) -> Any: from routstr.core.db import ModelPathRow @@ -899,7 +904,7 @@ def _endpoint_row( provider_type="openrouter", endpoint_tag=endpoint_tag, model_metadata=json.dumps( - {"id": model_id, "pricing": _ENDPOINT_PRICING, **limits} + {"id": model_id, "pricing": pricing or _ENDPOINT_PRICING, **limits} ), upstream_provider_id=1, ) @@ -1042,3 +1047,42 @@ async def test_endpoint_pin_never_bills_below_an_operator_override( ) assert (priced.pricing.prompt, priced.pricing.completion) == pytest.approx(expected) + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + ("override", "endpoint", "expected_cache_read"), + [ + # The override has no cache rate, so its cache reads bill at its + # prompt rate; the endpoint's cheaper cache rate must not undercut it. + ( + (2e-6, 8e-6, 0.0), + {"prompt": 1e-6, "completion": 2e-6, "input_cache_read": 1e-7}, + 2e-6, + ), + # The endpoint has no cache rate, so it charges its prompt rate; the + # override's cheaper cache rate must not bill below that cost. + ((1e-6, 2e-6, 1e-8), {"prompt": 2e-6, "completion": 4e-6}, 2e-6), + ], + ids=["override-without-cache-rate", "endpoint-without-cache-rate"], +) +async def test_override_floor_compares_effective_cache_rates( + override: tuple[float, float, float], + endpoint: dict[str, float], + expected_cache_read: float, +) -> None: + """A zero cache rate is billed at the prompt rate, so the floor compares + those effective rates rather than the zeros.""" + from routstr.core.db import ModelRow + + with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): + priced = await proxy_module._price_pinned_endpoint( + _session_with_rows( + [_endpoint_row(pricing=endpoint)], override=MagicMock(spec=ModelRow) + ), + _endpoint_selector(), + _priced_model(*override), + _openrouter_upstream(), + ) + + assert priced.pricing.input_cache_read == pytest.approx(expected_cache_read) From 799d1ba41163ba768549e916899b170cf6c422c2 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Sat, 3 Oct 2026 18:37:42 +0200 Subject: [PATCH 4/4] refactor: drop the operator-override floor on pinned endpoints --- routstr/core/admin.py | 16 ++--- routstr/proxy.py | 6 +- routstr/upstream/model_paths.py | 68 ++----------------- tests/integration/test_certify_endpoint.py | 11 ++-- tests/unit/test_model_path_routing.py | 77 +--------------------- 5 files changed, 19 insertions(+), 159 deletions(-) diff --git a/routstr/core/admin.py b/routstr/core/admin.py index f6f24db6..a36aa7bf 100644 --- a/routstr/core/admin.py +++ b/routstr/core/admin.py @@ -1727,16 +1727,14 @@ async def certify_upstream_provider( # The proxy reserves and token-bills a pinned endpoint at that # endpoint's own rates, so the cost rows price it the same way. if selected_path is not None: - from ..upstream.model_paths import price_pinned_endpoint + from ..upstream.model_paths import apply_model_path_pricing - async with create_session() as session: - model_obj = await price_pinned_endpoint( - session, - model_obj, - selected_path, - provider.provider_fee, - sats_to_usd, - ) + model_obj = apply_model_path_pricing( + model_obj, + selected_path, + provider.provider_fee, + sats_to_usd, + ) # The timeout applies per upstream call. The run makes up to six # calls (models, two short probes after a max_completion_tokens retry, # three cache probes), so the request can stay open for six times it. diff --git a/routstr/proxy.py b/routstr/proxy.py index 7c210e0c..9faa1acb 100644 --- a/routstr/proxy.py +++ b/routstr/proxy.py @@ -53,9 +53,9 @@ from .upstream.ehbp import forward_ehbp_request, forward_ehbp_x_cashu_request from .upstream.helpers import init_upstreams from .upstream.model_paths import ( ModelPathSelector, + apply_model_path_pricing, decode_model_path, is_openrouter_base_url, - price_pinned_endpoint, public_model_id, public_provider_url, ) @@ -194,9 +194,7 @@ async def _price_pinned_endpoint( extra={"model": selector.model_id, "endpoint": selector.endpoint_tag}, ) return model_obj - return await price_pinned_endpoint( - session, model_obj, row, upstream.provider_fee, sats_to_usd - ) + return apply_model_path_pricing(model_obj, row, upstream.provider_fee, sats_to_usd) def get_model_instance(model_id: str) -> Model | None: diff --git a/routstr/upstream/model_paths.py b/routstr/upstream/model_paths.py index a72fe807..b69bef9e 100644 --- a/routstr/upstream/model_paths.py +++ b/routstr/upstream/model_paths.py @@ -38,7 +38,7 @@ from ..core.logging import get_logger if TYPE_CHECKING: from sqlmodel.ext.asyncio.session import AsyncSession - from ..payment.models import Model, Pricing + from ..payment.models import Model from .base import BaseUpstreamProvider logger = get_logger(__name__) @@ -893,7 +893,6 @@ def apply_model_path_pricing( row: ModelPathRow, provider_fee: float, sats_to_usd: float, - floor: "Pricing | None" = None, ) -> "Model": """Return ``model`` priced from an exact endpoint path's own rates. @@ -903,9 +902,6 @@ def apply_model_path_pricing( default listing; OpenRouter charges the endpoint that serves the request, so the proxy reserves and token-bills a pinned endpoint with them, using the same limits ``/v1/models/paths`` quotes its max cost from. - - ``floor`` is an operator price override: no rate is billed below it, so - pinning an endpoint cannot bypass the operator's pricing. """ if row.endpoint_tag is None: return model @@ -928,20 +924,9 @@ def apply_model_path_pricing( model.forwarded_model_id or row.model_id, Pricing.parse_obj(metadata["pricing"]), ) - rates = { - key: float(value) * provider_fee - for key, value in pricing.dict().items() - if not key.startswith("max_") - } - if floor is not None: - floor_rates = _effective_cache_rates( - {key: float(getattr(floor, key)) for key in rates} - ) - rates = { - key: max(value, floor_rates[key]) - for key, value in _effective_cache_rates(rates).items() - } - pricing = Pricing.parse_obj(rates) + pricing = Pricing.parse_obj( + {key: float(value) * provider_fee for key, value in pricing.dict().items()} + ) update: dict[str, Any] = {"pricing": pricing, "sats_pricing": None} context_length = metadata.get("context_length") max_completion_tokens = metadata.get("max_completion_tokens") @@ -966,51 +951,6 @@ def apply_model_path_pricing( return model -def _effective_cache_rates(rates: dict[str, float]) -> dict[str, float]: - """Spell out cache rates settlement reads as "bill at the prompt rate". - - A zero cache rate is billed at the prompt rate, so a per-rate ``max`` - must compare those prompt rates, not the zeros. - """ - return { - key: value - if value > 0 or key not in ("input_cache_read", "input_cache_write") - else rates["prompt"] - for key, value in rates.items() - } - - -async def price_pinned_endpoint( - session: AsyncSession, - model: "Model", - row: ModelPathRow, - provider_fee: float, - sats_to_usd: float, -) -> "Model": - """Price ``model`` for a request pinned to ``row``'s endpoint. - - An enabled operator override for the model on this provider is the - price floor; ``model`` already carries it, since overrides replace the - provider's model in routing. - """ - override = ( - await session.exec( - select(ModelRow).where( - ModelRow.id == model.id, - ModelRow.upstream_provider_id == row.upstream_provider_id, - ModelRow.enabled, - ) - ) - ).first() - return apply_model_path_pricing( - model, - row, - provider_fee, - sats_to_usd, - floor=model.pricing if override is not None else None, - ) - - def _serialize_path(row: ModelPathRow, provider_fee: float) -> dict[str, Any]: endpoint = None if row.endpoint_tag or row.endpoint_name: diff --git a/tests/integration/test_certify_endpoint.py b/tests/integration/test_certify_endpoint.py index 8d338850..065f888f 100644 --- a/tests/integration/test_certify_endpoint.py +++ b/tests/integration/test_certify_endpoint.py @@ -409,16 +409,13 @@ async def test_certify_margin_bills_pinned_path_pricing( assert resp.status_code == 200, resp.text margin = _find_row(resp.json()["rows"], "cost.margin") - # The proxy bills a pinned endpoint at its own rates, with the operator's - # override as the floor per rate: the endpoint's prompt rate, the - # override's completion and cache-read rates. The model's rates alone - # would give 3, 356 and 43; the endpoint's alone 3, 289 and 15, which - # misses the cached call's reported cost of 26. + # The proxy token-bills a pinned endpoint at its own rates, not the + # model's (3, 356, 43). Those rates miss the cached call's reported cost. assert [ (sample["upstream_msats_with_fee"], sample["configured_msats"]) for sample in margin["evidence"]["samples"] - ] == [(3, 3), (269, 289), (26, 27)] - assert margin["status"] == "ok" + ] == [(3, 3), (269, 289), (26, 15)] + assert margin["status"] == "fail" @pytest.mark.integration diff --git a/tests/unit/test_model_path_routing.py b/tests/unit/test_model_path_routing.py index 0f12eef9..a62b0eeb 100644 --- a/tests/unit/test_model_path_routing.py +++ b/tests/unit/test_model_path_routing.py @@ -910,14 +910,9 @@ def _endpoint_row( ) -def _session_with_rows(rows: list[Any], override: Any = None) -> MagicMock: - """Answers the path-row query with ``rows``, the override one with ``override``.""" +def _session_with_rows(rows: list[Any]) -> MagicMock: session = MagicMock() - session.exec = AsyncMock( - return_value=MagicMock( - all=MagicMock(return_value=rows), first=MagicMock(return_value=override) - ) - ) + session.exec = AsyncMock(return_value=MagicMock(all=MagicMock(return_value=rows))) return session @@ -1018,71 +1013,3 @@ async def test_endpoint_pin_reserves_the_max_cost_paths_quotes() -> None: assert priced.sats_pricing is not None and priced.top_provider is not None assert priced.sats_pricing.max_cost == pytest.approx(quoted) assert priced.top_provider.max_completion_tokens == 8192 - - -@pytest.mark.asyncio -@pytest.mark.parametrize( - ("override", "expected"), - [ - # The operator's override is above the endpoint: it stays the price. - ((6e-6, 12e-6), (6e-6, 12e-6)), - # The endpoint costs more than the override: never bill below cost. - ((1e-6, 2e-6), (2e-6, 4e-6)), - # Each rate takes the higher of the two. - ((3e-6, 1e-6), (3e-6, 4e-6)), - ], -) -async def test_endpoint_pin_never_bills_below_an_operator_override( - override: tuple[float, float], expected: tuple[float, float] -) -> None: - from routstr.core.db import ModelRow - - model = _priced_model(*override) - with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): - priced = await proxy_module._price_pinned_endpoint( - _session_with_rows([_endpoint_row()], override=MagicMock(spec=ModelRow)), - _endpoint_selector(), - model, - _openrouter_upstream(), - ) - - assert (priced.pricing.prompt, priced.pricing.completion) == pytest.approx(expected) - - -@pytest.mark.asyncio -@pytest.mark.parametrize( - ("override", "endpoint", "expected_cache_read"), - [ - # The override has no cache rate, so its cache reads bill at its - # prompt rate; the endpoint's cheaper cache rate must not undercut it. - ( - (2e-6, 8e-6, 0.0), - {"prompt": 1e-6, "completion": 2e-6, "input_cache_read": 1e-7}, - 2e-6, - ), - # The endpoint has no cache rate, so it charges its prompt rate; the - # override's cheaper cache rate must not bill below that cost. - ((1e-6, 2e-6, 1e-8), {"prompt": 2e-6, "completion": 4e-6}, 2e-6), - ], - ids=["override-without-cache-rate", "endpoint-without-cache-rate"], -) -async def test_override_floor_compares_effective_cache_rates( - override: tuple[float, float, float], - endpoint: dict[str, float], - expected_cache_read: float, -) -> None: - """A zero cache rate is billed at the prompt rate, so the floor compares - those effective rates rather than the zeros.""" - from routstr.core.db import ModelRow - - with patch("routstr.payment.price.SATS_USD_PRICE", _SATS_USD): - priced = await proxy_module._price_pinned_endpoint( - _session_with_rows( - [_endpoint_row(pricing=endpoint)], override=MagicMock(spec=ModelRow) - ), - _endpoint_selector(), - _priced_model(*override), - _openrouter_upstream(), - ) - - assert priced.pricing.input_cache_read == pytest.approx(expected_cache_read)