feat(upstream): add native DeepSeek provider, retire V4 pricing shim

Add a first-class `deepseek` upstream provider that needs only an API key;
the base URL is fixed to https://api.deepseek.com. DEEPSEEK_API_KEY seeds it
on startup.

Models come from DeepSeek's own /models and are priced from a peak-rate
table in routstr/upstream/deepseek.py. A listed model the table misses is
imported disabled instead of taking a litellm/OpenRouter price, via a new
GenericUpstreamProvider.use_fallback_pricing switch (default True, so other
providers are unchanged). The node bills one flat price per model, so the
table holds the peak rate and never bills below DeepSeek's cost.

Thinking-mode reasoning_content is forwarded unchanged: DeepSeek requires it
on requests that carry tools and ignores it otherwise.

Remove the temporary DeepSeek V4 pricing shim and its startup call. The
pinned litellm 1.101.2 already ships every key it filled.
This commit is contained in:
9qeklajc
2026-09-30 13:35:01 +00:00
parent 8a06e6ab36
commit e589b37bc5
10 changed files with 369 additions and 91 deletions
+23
View File
@@ -48,6 +48,29 @@ Connect to your AI provider(s):
| **Upstream URL** | API endpoint (e.g., `https://api.openai.com/v1`) | | **Upstream URL** | API endpoint (e.g., `https://api.openai.com/v1`) |
| **API Key** | Your provider's API key | | **API Key** | Your provider's API key |
### DeepSeek
Choose **DeepSeek** as the provider type and paste an API key from
[platform.deepseek.com](https://platform.deepseek.com/api_keys); the base URL
is fixed to `https://api.deepseek.com`. Setting `DEEPSEEK_API_KEY` seeds the
provider on startup instead.
Models are listed from DeepSeek's own `/models` and priced from a rate table
in `routstr/upstream/deepseek.py`, not from litellm or OpenRouter:
- **Peak rates only.** DeepSeek charges half price off-peak, but the node bills
one flat price per model, so it bills the peak rate. Clients overpay
off-peak; the node never bills below cost. Time-of-day pricing is planned.
- **Unknown models import disabled.** A model DeepSeek lists that the table
does not price shows up disabled in the Admin Dashboard. Enable it with a
manual price, or add it to the table.
- **Cache hits** bill at DeepSeek's cache-hit rate (about 2% of the input
rate).
Thinking-mode `reasoning_content` is returned to clients unchanged in
responses, and forwarded unchanged when it appears in conversation history.
DeepSeek requires it on requests that carry `tools` and ignores it otherwise.
### PPQ Auto Top-up ### PPQ Auto Top-up
PPQ providers can automatically purchase more credits when their USD balance PPQ providers can automatically purchase more credits when their USD balance
-6
View File
@@ -34,7 +34,6 @@ from ..payment.price import update_prices_periodically
from ..proxy import initialize_upstreams, proxy_router, refresh_model_maps_periodically from ..proxy import initialize_upstreams, proxy_router, refresh_model_maps_periodically
from ..refund import periodic_refund_reconcile from ..refund import periodic_refund_reconcile
from ..upstream.auto_topup import periodic_auto_topup from ..upstream.auto_topup import periodic_auto_topup
from ..upstream.deepseek_v4_pricing_shim import register_deepseek_v4_pricing
from ..upstream.http_client import close_upstream_http_client from ..upstream.http_client import close_upstream_http_client
from ..upstream.litellm_routing import configure_litellm from ..upstream.litellm_routing import configure_litellm
from ..wallet import periodic_payout, periodic_refund_sweep, periodic_routstr_fee_payout from ..wallet import periodic_payout, periodic_refund_sweep, periodic_routstr_fee_payout
@@ -89,11 +88,6 @@ async def lifespan(_: FastAPI) -> AsyncGenerator[None, None]:
# debug logging) before any upstream provider dispatches a request. # debug logging) before any upstream provider dispatches a request.
configure_litellm() configure_litellm()
# TEMPORARY: backfill DeepSeek V4 pricing missing from litellm's cost
# map (BerriAI/litellm#30430). Remove this call and
# deepseek_v4_pricing_shim.py once litellm ships these models.
register_deepseek_v4_pricing()
# Run database migrations on startup # Run database migrations on startup
run_migrations() run_migrations()
+2
View File
@@ -1,6 +1,7 @@
from .anthropic import AnthropicUpstreamProvider from .anthropic import AnthropicUpstreamProvider
from .azure import AzureUpstreamProvider from .azure import AzureUpstreamProvider
from .base import BaseUpstreamProvider from .base import BaseUpstreamProvider
from .deepseek import DeepSeekUpstreamProvider
from .fireworks import FireworksUpstreamProvider from .fireworks import FireworksUpstreamProvider
from .gemini import GeminiUpstreamProvider from .gemini import GeminiUpstreamProvider
from .generic import GenericUpstreamProvider from .generic import GenericUpstreamProvider
@@ -19,6 +20,7 @@ from .xai import XAIUpstreamProvider
upstream_provider_classes: list[type[BaseUpstreamProvider]] = [ upstream_provider_classes: list[type[BaseUpstreamProvider]] = [
AnthropicUpstreamProvider, AnthropicUpstreamProvider,
AzureUpstreamProvider, AzureUpstreamProvider,
DeepSeekUpstreamProvider,
FireworksUpstreamProvider, FireworksUpstreamProvider,
GeminiUpstreamProvider, GeminiUpstreamProvider,
GenericUpstreamProvider, GenericUpstreamProvider,
+105
View File
@@ -0,0 +1,105 @@
"""First-class upstream for the DeepSeek API.
Pricing comes from ``_PEAK_RATES`` below, not from litellm or OpenRouter:
litellm's bundled ``deepseek-v4-flash`` entry is stale, the OpenRouter feed
carries resale prices below DeepSeek's own peak rate, and neither knows the
current ``deepseek-flash`` id. A model DeepSeek lists that the table does not
cover is imported disabled rather than priced from those sources.
DeepSeek bills peak hours at twice the off-peak rate. The node has one flat
price per model, so the table holds the PEAK rates: a client may overpay
off-peak but the node never bills below its own cost.
Rates: https://api-docs.deepseek.com/quick_start/pricing (checked 2026-09-30).
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from .base import BaseUpstreamProvider
from .generic import GenericUpstreamProvider
from .pricing_resolver import ResolvedPricing
if TYPE_CHECKING:
from ..core.db import UpstreamProviderRow
_CONTEXT_LENGTH = 1_000_000
_MAX_OUTPUT_TOKENS = 384_000
# USD per 1M tokens at DeepSeek's peak rate: (input cache miss, output, input
# cache hit). DeepSeek has no cache-write charge.
_FLASH = (0.30, 1.20, 0.006)
_PRO = (1.32, 3.96, 0.044)
_PEAK_RATES: dict[str, tuple[float, float, float]] = {
"deepseek-flash": _FLASH,
# Retired ids DeepSeek still accepts, served and billed as deepseek-flash.
"deepseek-v4-flash": _FLASH,
"deepseek-v4-flash-vision-exp": _FLASH,
"deepseek-v4-pro": _PRO,
}
# Pro is the only current model without vision support.
_TEXT_ONLY = {"deepseek-v4-pro"}
class DeepSeekUpstreamProvider(GenericUpstreamProvider):
"""Upstream provider specifically configured for the DeepSeek API."""
provider_type = "deepseek"
default_base_url = "https://api.deepseek.com"
platform_url = "https://platform.deepseek.com/api_keys"
litellm_provider_prefix = "deepseek/"
use_fallback_pricing = False
def __init__(self, api_key: str, provider_fee: float = 1.01):
super().__init__(
base_url=self.default_base_url,
api_key=api_key,
provider_fee=provider_fee,
upstream_name="DeepSeek",
)
@classmethod
def _build_from_row(
cls, provider_row: "UpstreamProviderRow"
) -> "DeepSeekUpstreamProvider":
return cls(api_key=provider_row.api_key, provider_fee=provider_row.provider_fee)
@classmethod
def get_provider_metadata(cls) -> dict[str, object]:
return {
"id": cls.provider_type,
"name": "DeepSeek",
"default_base_url": cls.default_base_url,
"fixed_base_url": True,
"platform_url": cls.platform_url,
}
def _apply_provider_field(self, response_json: object) -> None:
# A first-party upstream: stamp "deepseek", not Generic's hostname.
BaseUpstreamProvider._apply_provider_field(self, response_json)
def transform_model_name(self, model_id: str) -> str:
"""Strip the 'deepseek/' prefix for DeepSeek API compatibility."""
return model_id.removeprefix("deepseek/")
def _native_pricing(
self, model_id: str, model_spec: dict
) -> ResolvedPricing | None:
"""Price ``model_id`` from the peak-rate table; ``None`` if absent."""
rates = _PEAK_RATES.get(model_id)
if rates is None:
return None
input_usd, output_usd, cache_hit_usd = rates
input_modalities = ["text"] if model_id in _TEXT_ONLY else ["text", "image"]
return ResolvedPricing(
prompt=input_usd / 1_000_000,
completion=output_usd / 1_000_000,
context_length=_CONTEXT_LENGTH,
source="native",
max_completion_tokens=_MAX_OUTPUT_TOKENS,
input_cache_read=cache_hit_usd / 1_000_000,
input_modalities=input_modalities,
)
@@ -1,73 +0,0 @@
"""TEMPORARY: local DeepSeek V4 pricing shim.
litellm's bundled cost map does not yet ship ``deepseek-v4-flash`` /
``deepseek-v4-pro``. Without an entry, ``backfill_cache_pricing`` cannot find a
``cache_read_input_token_cost`` and cache reads fall back to the full input
rate — a large overcharge on cache hits (DeepSeek V4 hits are ~0.008-0.02x
input, i.e. cached tokens cost 50-120x less than regular input).
This module injects the missing entries into ``litellm.model_cost`` at startup
so the existing backfill path resolves them. Rates mirror the canonical
``deepseek`` provider entries now in litellm's ``model_prices`` map
(``input_cost_per_token`` is the cache-*miss* rate;
``cache_read_input_token_cost`` is the cache-*hit* rate), sourced from
https://api-docs.deepseek.com/quick_start/pricing via
https://github.com/BerriAI/litellm/pull/26380 (issue
https://github.com/BerriAI/litellm/issues/30430).
=== REMOVAL (once litellm ships these models) ===
Delete this file and the single ``register_deepseek_v4_pricing()`` call in
``routstr/core/main.py``. Nothing else depends on it. Entries are only added
when absent, so a stale shim is harmless after upstream lands — but remove it.
"""
import litellm
from ..core import get_logger
logger = get_logger(__name__)
# USD per token. Mirrors the canonical ``deepseek`` provider entries in
# litellm's model_prices map (source: DeepSeek API pricing docs). Keep these in
# sync with ``litellm.model_cost["deepseek/deepseek-v4-*"]``.
_DEEPSEEK_V4_RATES: dict[str, dict[str, float]] = {
"deepseek-v4-flash": {
"input_cost_per_token": 1.4e-07,
"output_cost_per_token": 2.8e-07,
"cache_read_input_token_cost": 2.8e-09,
"cache_creation_input_token_cost": 0.0,
"input_cost_per_token_cache_hit": 2.8e-09,
},
"deepseek-v4-pro": {
"input_cost_per_token": 4.35e-07,
"output_cost_per_token": 8.7e-07,
"cache_read_input_token_cost": 3.625e-09,
"cache_creation_input_token_cost": 0.0,
"input_cost_per_token_cache_hit": 3.625e-09,
},
}
def register_deepseek_v4_pricing() -> None:
"""Inject DeepSeek V4 pricing into ``litellm.model_cost`` if absent.
Idempotent and non-destructive: a key already present in the cost map
(e.g. once litellm ships it) is left untouched. Registers both the bare
(``deepseek-v4-flash``) and prefixed (``deepseek/deepseek-v4-flash``)
spellings since ``backfill_cache_pricing`` tries both.
"""
added = []
for bare, rates in _DEEPSEEK_V4_RATES.items():
for key in (bare, f"deepseek/{bare}"):
if key in litellm.model_cost:
continue
entry: dict[str, object] = dict(rates)
entry["litellm_provider"] = "deepseek"
entry["mode"] = "chat"
litellm.model_cost[key] = entry
added.append(key)
if added:
logger.info(
"Registered temporary DeepSeek V4 pricing shim",
extra={"models": added},
)
+6 -2
View File
@@ -28,7 +28,11 @@ class GenericUpstreamProvider(BaseUpstreamProvider):
provider_type = "generic" provider_type = "generic"
default_base_url = "http://localhost:8888" default_base_url = "http://localhost:8888"
platform_url = None platform_url: str | None = None
# Subclasses that own an authoritative price table set this False so a model
# the table misses imports disabled instead of taking a litellm/OpenRouter
# price that may undercut the upstream's own rate.
use_fallback_pricing = True
def __init__( def __init__(
self, self,
@@ -162,7 +166,7 @@ class GenericUpstreamProvider(BaseUpstreamProvider):
model_spec = model_data.get("model_spec", {}) model_spec = model_data.get("model_spec", {})
resolved = self._native_pricing(model_id, model_spec) resolved = self._native_pricing(model_id, model_spec)
if resolved is None: if resolved is None and self.use_fallback_pricing:
resolved = await resolver.resolve(model_id) resolved = await resolver.resolve(model_id)
if resolved is None: if resolved is None:
+1
View File
@@ -272,6 +272,7 @@ async def _seed_providers_from_settings(
("PERPLEXITY_API_KEY", "perplexity", None, None), ("PERPLEXITY_API_KEY", "perplexity", None, None),
("FIREWORKS_API_KEY", "fireworks", None, None), ("FIREWORKS_API_KEY", "fireworks", None, None),
("XAI_API_KEY", "xai", None, None), ("XAI_API_KEY", "xai", None, None),
("DEEPSEEK_API_KEY", "deepseek", None, None),
("TINFOIL_API_KEY", "tinfoil", None, None), ("TINFOIL_API_KEY", "tinfoil", None, None),
("TYPESAFE_API_KEY", "typesafe", None, None), ("TYPESAFE_API_KEY", "typesafe", None, None),
] ]
@@ -470,7 +470,6 @@ async def test_startup_runs_bootstrap_before_settings_initialize(
return None return None
monkeypatch.setattr(main, "configure_litellm", lambda: None) monkeypatch.setattr(main, "configure_litellm", lambda: None)
monkeypatch.setattr(main, "register_deepseek_v4_pricing", lambda: None)
monkeypatch.setattr(main, "run_migrations", lambda: None) monkeypatch.setattr(main, "run_migrations", lambda: None)
monkeypatch.setattr(main, "init_db", noop_init_db) monkeypatch.setattr(main, "init_db", noop_init_db)
monkeypatch.setattr(main, "create_session", fake_create_session) monkeypatch.setattr(main, "create_session", fake_create_session)
-9
View File
@@ -31,15 +31,6 @@ from routstr.payment.models import (
backfill_cache_pricing, backfill_cache_pricing,
) )
from routstr.upstream import GenericUpstreamProvider from routstr.upstream import GenericUpstreamProvider
from routstr.upstream.deepseek_v4_pricing_shim import register_deepseek_v4_pricing
@pytest.fixture(autouse=True)
def _deepseek_v4_pricing() -> None:
# litellm's bundled cost map lacks the DeepSeek V4 entries (they only
# appear when its remote map is reachable); production injects them at
# startup via this same shim.
register_deepseek_v4_pricing()
def _make_model(model_id: str, pricing: Pricing) -> Model: def _make_model(model_id: str, pricing: Pricing) -> Model:
+232
View File
@@ -0,0 +1,232 @@
"""Unit tests for ``DeepSeekUpstreamProvider``.
DeepSeek is priced from the provider's own peak-rate table, never from litellm
or OpenRouter: litellm's ``deepseek-v4-flash`` entry is stale and OpenRouter
resells below DeepSeek's peak rate, so either would bill under cost. These
tests pin the table prices (including the cache-hit rate), that a model the
table misses imports disabled without consulting the fallback chain, and that
``reasoning_content`` in history reaches DeepSeek untouched — thinking mode
with ``tools`` answers 400 when it is stripped.
"""
from __future__ import annotations
import json
from typing import Any
from unittest.mock import AsyncMock, Mock, patch
import pytest
from routstr.upstream import upstream_provider_classes
from routstr.upstream.deepseek import DeepSeekUpstreamProvider
class _FakeResponse:
def __init__(self, payload: dict[str, Any]) -> None:
self._payload = payload
def raise_for_status(self) -> None:
return None
def json(self) -> dict[str, Any]:
return self._payload
class _FakeAsyncClient:
def __init__(self, payload: dict[str, Any], calls: list[dict[str, Any]]) -> None:
self._payload = payload
self._calls = calls
async def __aenter__(self) -> "_FakeAsyncClient":
return self
async def __aexit__(self, *exc: object) -> bool:
return False
async def get(
self, url: str, headers: dict[str, str] | None = None
) -> _FakeResponse:
self._calls.append({"url": url, "headers": headers})
return _FakeResponse(self._payload)
# Shape of DeepSeek's ``GET /models``: bare ids, no pricing.
CATALOG: dict[str, Any] = {
"object": "list",
"data": [
{"id": "deepseek-flash", "object": "model", "owned_by": "deepseek"},
{"id": "deepseek-v4-pro", "object": "model", "owned_by": "deepseek"},
{"id": "deepseek-v4-flash", "object": "model", "owned_by": "deepseek"},
{"id": "deepseek-chat", "object": "model", "owned_by": "deepseek"},
],
}
async def _fetch(
catalog: dict[str, Any] = CATALOG,
) -> tuple[dict[str, Any], list[dict[str, Any]], AsyncMock]:
calls: list[dict[str, Any]] = []
fallback = AsyncMock(return_value=None)
provider = DeepSeekUpstreamProvider(api_key="sk-test")
with (
patch(
"routstr.upstream.generic.httpx.AsyncClient",
lambda *args, **kwargs: _FakeAsyncClient(catalog, calls),
),
patch("routstr.upstream.generic.FallbackPricingResolver.resolve", fallback),
):
models = await provider.fetch_models()
return {m.id: m for m in models}, calls, fallback
def test_metadata_and_registration() -> None:
assert DeepSeekUpstreamProvider in upstream_provider_classes
assert DeepSeekUpstreamProvider.get_provider_metadata() == {
"id": "deepseek",
"name": "DeepSeek",
"default_base_url": "https://api.deepseek.com",
"fixed_base_url": True,
"platform_url": "https://platform.deepseek.com/api_keys",
}
def test_build_from_row_ignores_row_base_url() -> None:
row = Mock(
api_key="sk-row", provider_fee=1.05, base_url="https://elsewhere.example"
)
provider = DeepSeekUpstreamProvider._build_from_row(row)
assert provider.api_key == "sk-row"
assert provider.provider_fee == 1.05
assert provider.base_url == "https://api.deepseek.com"
def test_litellm_prefix_is_deepseek() -> None:
provider = DeepSeekUpstreamProvider(api_key="sk-test")
assert provider.get_litellm_provider_prefix() == "deepseek/"
@pytest.mark.parametrize(
"model_id,expected",
[
("deepseek/deepseek-v4-flash", "deepseek-v4-flash"),
("deepseek-v4-flash", "deepseek-v4-flash"),
("deepseek/deepseek-flash", "deepseek-flash"),
],
)
def test_transform_model_name(model_id: str, expected: str) -> None:
provider = DeepSeekUpstreamProvider(api_key="sk-test")
assert provider.transform_model_name(model_id) == expected
def test_provider_field_names_deepseek_not_host() -> None:
provider = DeepSeekUpstreamProvider(api_key="sk-test")
payload: dict[str, Any] = {"id": "chatcmpl-1"}
provider._apply_provider_field(payload)
assert payload["provider"] == "deepseek"
@pytest.mark.asyncio
async def test_fetch_models_calls_deepseek_models_endpoint_with_key() -> None:
_, calls, _ = await _fetch()
assert calls == [
{
"url": "https://api.deepseek.com/models",
"headers": {"Authorization": "Bearer sk-test"},
}
]
@pytest.mark.asyncio
@pytest.mark.parametrize(
"model_id,prompt,completion,cache_read",
[
("deepseek-flash", 0.30, 1.20, 0.006),
# Retired alias DeepSeek serves and bills as deepseek-flash.
("deepseek-v4-flash", 0.30, 1.20, 0.006),
("deepseek-v4-pro", 1.32, 3.96, 0.044),
],
)
async def test_table_models_priced_at_peak_rate(
model_id: str, prompt: float, completion: float, cache_read: float
) -> None:
models, _, _ = await _fetch()
model = models[model_id]
assert model.enabled is True
assert model.pricing.prompt == pytest.approx(prompt / 1_000_000)
assert model.pricing.completion == pytest.approx(completion / 1_000_000)
assert model.pricing.input_cache_read == pytest.approx(cache_read / 1_000_000)
assert model.context_length == 1_000_000
@pytest.mark.asyncio
async def test_vision_follows_the_model() -> None:
models, _, _ = await _fetch()
assert "image" in models["deepseek-flash"].architecture.input_modalities
assert models["deepseek-v4-pro"].architecture.input_modalities == ["text"]
@pytest.mark.asyncio
async def test_unlisted_model_imports_disabled_without_fallback() -> None:
"""litellm prices ``deepseek-chat``; the provider must not take that price."""
models, _, fallback = await _fetch()
model = models["deepseek-chat"]
assert model.enabled is False
assert model.pricing.prompt == 0.0
assert model.pricing.completion == 0.0
fallback.assert_not_awaited()
@pytest.mark.asyncio
async def test_cache_rate_survives_fee_and_is_not_replaced_by_litellm() -> None:
"""litellm's stale ``deepseek-v4-flash`` cache rate (1.4e-08 in the bundled
map) must not replace the table's; backfill only fills an absent rate. The
fee applies to the cache rate like every other component.
The litellm entry is pinned here because the remote cost map already
carries the table's rate, which would let an overwrite go unnoticed."""
models, _, _ = await _fetch()
provider = DeepSeekUpstreamProvider(api_key="sk-test", provider_fee=1.05)
stale = {"cache_read_input_token_cost": 1.4e-08}
with patch("routstr.payment.models.litellm_cost_entry", return_value=stale):
priced = provider._apply_provider_fee_to_model(models["deepseek-v4-flash"])
assert priced.pricing.input_cache_read == pytest.approx(0.006e-6 * 1.05)
assert priced.pricing.prompt == pytest.approx(0.30e-6 * 1.05)
# A cache hit costs 2% of a miss, not the full input rate.
assert priced.pricing.input_cache_read / priced.pricing.prompt == pytest.approx(
0.02
)
@pytest.mark.asyncio
async def test_reasoning_content_in_history_reaches_upstream() -> None:
models, _, _ = await _fetch()
provider = DeepSeekUpstreamProvider(api_key="sk-test")
messages = [
{"role": "user", "content": "weather in Paris?"},
{
"role": "assistant",
"content": "",
"reasoning_content": "Need the weather tool.",
"tool_calls": [
{
"id": "call_1",
"type": "function",
"function": {"name": "get_weather", "arguments": "{}"},
}
],
},
{"role": "tool", "tool_call_id": "call_1", "content": "18C"},
]
body = json.dumps(
{
"model": "deepseek/deepseek-flash",
"messages": messages,
"tools": [{"type": "function", "function": {"name": "get_weather"}}],
}
).encode()
out = provider.prepare_request_body(body, models["deepseek-flash"])
assert out is not None
sent = json.loads(out)
assert sent["model"] == "deepseek-flash"
assert sent["messages"] == messages