From 2baea4149e9a9f38f938c5cc9b1dc60df69d081c Mon Sep 17 00:00:00 2001 From: Jeroen Ubbink Date: Tue, 7 Jul 2026 12:25:09 +0200 Subject: [PATCH] fix(upstream): break OpenRouter bare-tail ties on combined token cost MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The bare-tail tie-break ranked candidates by prompt price alone, so two entries sharing a tail where one is cheaper on prompt but far dearer on completion could resolve to the entry that undercharges output-heavy traffic — contradicting the "highest-priced wins for money safety" promise. Rank by the combined prompt + completion per-token cost instead, so the choice stays deterministic and money-safe whichever way traffic leans. As before there are zero bare-tail collisions in the live feed, so this changes no resolved price today; it only governs the latent case. Co-Authored-By: Claude Opus 4.8 --- routstr/upstream/pricing_resolver.py | 22 ++++++++++------ tests/unit/test_upstream_generic.py | 38 ++++++++++++++++++++++++++++ 2 files changed, 52 insertions(+), 8 deletions(-) diff --git a/routstr/upstream/pricing_resolver.py b/routstr/upstream/pricing_resolver.py index f5c1cb32..3d009fdc 100644 --- a/routstr/upstream/pricing_resolver.py +++ b/routstr/upstream/pricing_resolver.py @@ -123,10 +123,12 @@ def _match_openrouter(model_id: str, feed: list[dict]) -> dict | None: Bare-tail matching (``deepseek-chat`` ↔ ``deepseek/deepseek-chat``) is a looser, lower-trust match — OpenRouter fans a model out across resellers — so an exact id match always wins first. When several entries share the bare - tail, the highest-priced one wins: the choice must be deterministic (not - feed-order-dependent) and money-safe, since undercharging is the hazard. - The live feed has no such collisions today; this only governs the latent - case. + tail, the one with the highest *combined* (prompt + completion) per-token + cost wins: the choice must be deterministic (not feed-order-dependent) and + money-safe whichever way traffic leans, since undercharging is the hazard. + Ranking on prompt alone could pick an entry that is cheap on input but dear + on output. The live feed has no such collisions today; this only governs + the latent case. """ bare = model_id.split("/", 1)[-1] exact = next((m for m in feed if m.get("id") == model_id), None) @@ -135,10 +137,14 @@ def _match_openrouter(model_id: str, feed: list[dict]) -> dict | None: matches = [m for m in feed if m.get("id", "").split("/", 1)[-1] == bare] if not matches: return None - return max( - matches, - key=lambda m: _as_float(m.get("pricing", {}).get("prompt")) or 0.0, - ) + + def _combined_cost(m: dict) -> float: + pricing = m.get("pricing", {}) + return (_as_float(pricing.get("prompt")) or 0.0) + ( + _as_float(pricing.get("completion")) or 0.0 + ) + + return max(matches, key=_combined_cost) def _from_openrouter(model_id: str, feed: list[dict]) -> ResolvedPricing | None: diff --git a/tests/unit/test_upstream_generic.py b/tests/unit/test_upstream_generic.py index 4ff6f589..39daff5c 100644 --- a/tests/unit/test_upstream_generic.py +++ b/tests/unit/test_upstream_generic.py @@ -404,6 +404,44 @@ async def test_openrouter_bare_tail_collision_picks_highest_price() -> None: assert model.pricing.completion == pytest.approx(1e-05) +@pytest.mark.asyncio +async def test_openrouter_bare_tail_tie_breaks_on_combined_cost() -> None: + """The bare-tail tie-break must weigh *both* rates, not prompt alone. + Given two colliding entries where one is cheaper on prompt but far dearer + on completion, ranking by prompt would pick the entry that undercharges + output-heavy traffic. Pick the highest *combined* per-token cost so the + money-safe choice holds whichever way the traffic leans.""" + payload = { + "data": [ + {"id": "yyy-phantom-model", "object": "model", "owned_by": "mystery"}, + ] + } + # dear-overall is listed first with the *lower* prompt, so a prompt-only max + # would wrongly pick the second (cheaper-overall) entry. + or_feed = AsyncMock( + return_value=[ + { + "id": "dearco/yyy-phantom-model", + "context_length": 8192, + "pricing": {"prompt": "0.000001", "completion": "0.000100"}, + }, + { + "id": "cheapco/yyy-phantom-model", + "context_length": 8192, + "pricing": {"prompt": "0.000009", "completion": "0.000002"}, + }, + ] + ) + + with _patch_models_endpoint(payload): + with patch("routstr.payment.models.async_fetch_openrouter_models", or_feed): + models = await GenericUpstreamProvider(base_url="http://x").fetch_models() + + model = _model_by_id(models, "yyy-phantom-model") + assert model.pricing.prompt == pytest.approx(1e-06) + assert model.pricing.completion == pytest.approx(1e-04) + + @pytest.mark.asyncio async def test_openrouter_feed_fetched_once_per_discovery() -> None: """Two models both missing litellm must share a single OpenRouter fetch —