diff --git a/docs/provider/configuration.md b/docs/provider/configuration.md index 6a47854d..d2695fd9 100644 --- a/docs/provider/configuration.md +++ b/docs/provider/configuration.md @@ -48,6 +48,29 @@ Connect to your AI provider(s): | **Upstream URL** | API endpoint (e.g., `https://api.openai.com/v1`) | | **API Key** | Your provider's API key | +### DeepSeek + +Choose **DeepSeek** as the provider type and paste an API key from +[platform.deepseek.com](https://platform.deepseek.com/api_keys); the base URL +is fixed to `https://api.deepseek.com`. Setting `DEEPSEEK_API_KEY` seeds the +provider on startup instead. + +Models are listed from DeepSeek's own `/models` and priced from a rate table +in `routstr/upstream/deepseek.py`, not from litellm or OpenRouter: + +- **Peak rates only.** DeepSeek charges half price off-peak, but the node bills + one flat price per model, so it bills the peak rate. Clients overpay + off-peak; the node never bills below cost. Time-of-day pricing is planned. +- **Unknown models import disabled.** A model DeepSeek lists that the table + does not price shows up disabled in the Admin Dashboard. Enable it with a + manual price, or add it to the table. +- **Cache hits** bill at DeepSeek's cache-hit rate (about 2% of the input + rate). + +Thinking-mode `reasoning_content` is returned to clients unchanged in +responses, and forwarded unchanged when it appears in conversation history. +DeepSeek requires it on requests that carry `tools` and ignores it otherwise. + ### PPQ Auto Top-up PPQ providers can automatically purchase more credits when their USD balance diff --git a/routstr/core/main.py b/routstr/core/main.py index 584fa736..424025f0 100644 --- a/routstr/core/main.py +++ b/routstr/core/main.py @@ -34,7 +34,6 @@ from ..payment.price import update_prices_periodically from ..proxy import initialize_upstreams, proxy_router, refresh_model_maps_periodically from ..refund import periodic_refund_reconcile from ..upstream.auto_topup import periodic_auto_topup -from ..upstream.deepseek_v4_pricing_shim import register_deepseek_v4_pricing from ..upstream.http_client import close_upstream_http_client from ..upstream.litellm_routing import configure_litellm from ..wallet import periodic_payout, periodic_refund_sweep, periodic_routstr_fee_payout @@ -89,11 +88,6 @@ async def lifespan(_: FastAPI) -> AsyncGenerator[None, None]: # debug logging) before any upstream provider dispatches a request. configure_litellm() - # TEMPORARY: backfill DeepSeek V4 pricing missing from litellm's cost - # map (BerriAI/litellm#30430). Remove this call and - # deepseek_v4_pricing_shim.py once litellm ships these models. - register_deepseek_v4_pricing() - # Run database migrations on startup run_migrations() diff --git a/routstr/upstream/__init__.py b/routstr/upstream/__init__.py index 85d094e8..c57e0c09 100644 --- a/routstr/upstream/__init__.py +++ b/routstr/upstream/__init__.py @@ -1,6 +1,7 @@ from .anthropic import AnthropicUpstreamProvider from .azure import AzureUpstreamProvider from .base import BaseUpstreamProvider +from .deepseek import DeepSeekUpstreamProvider from .fireworks import FireworksUpstreamProvider from .gemini import GeminiUpstreamProvider from .generic import GenericUpstreamProvider @@ -19,6 +20,7 @@ from .xai import XAIUpstreamProvider upstream_provider_classes: list[type[BaseUpstreamProvider]] = [ AnthropicUpstreamProvider, AzureUpstreamProvider, + DeepSeekUpstreamProvider, FireworksUpstreamProvider, GeminiUpstreamProvider, GenericUpstreamProvider, diff --git a/routstr/upstream/deepseek.py b/routstr/upstream/deepseek.py new file mode 100644 index 00000000..01e5e334 --- /dev/null +++ b/routstr/upstream/deepseek.py @@ -0,0 +1,105 @@ +"""First-class upstream for the DeepSeek API. + +Pricing comes from ``_PEAK_RATES`` below, not from litellm or OpenRouter: +litellm's bundled ``deepseek-v4-flash`` entry is stale, the OpenRouter feed +carries resale prices below DeepSeek's own peak rate, and neither knows the +current ``deepseek-flash`` id. A model DeepSeek lists that the table does not +cover is imported disabled rather than priced from those sources. + +DeepSeek bills peak hours at twice the off-peak rate. The node has one flat +price per model, so the table holds the PEAK rates: a client may overpay +off-peak but the node never bills below its own cost. + +Rates: https://api-docs.deepseek.com/quick_start/pricing (checked 2026-09-30). +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +from .base import BaseUpstreamProvider +from .generic import GenericUpstreamProvider +from .pricing_resolver import ResolvedPricing + +if TYPE_CHECKING: + from ..core.db import UpstreamProviderRow + +_CONTEXT_LENGTH = 1_000_000 +_MAX_OUTPUT_TOKENS = 384_000 + +# USD per 1M tokens at DeepSeek's peak rate: (input cache miss, output, input +# cache hit). DeepSeek has no cache-write charge. +_FLASH = (0.30, 1.20, 0.006) +_PRO = (1.32, 3.96, 0.044) + +_PEAK_RATES: dict[str, tuple[float, float, float]] = { + "deepseek-flash": _FLASH, + # Retired ids DeepSeek still accepts, served and billed as deepseek-flash. + "deepseek-v4-flash": _FLASH, + "deepseek-v4-flash-vision-exp": _FLASH, + "deepseek-v4-pro": _PRO, +} + +# Pro is the only current model without vision support. +_TEXT_ONLY = {"deepseek-v4-pro"} + + +class DeepSeekUpstreamProvider(GenericUpstreamProvider): + """Upstream provider specifically configured for the DeepSeek API.""" + + provider_type = "deepseek" + default_base_url = "https://api.deepseek.com" + platform_url = "https://platform.deepseek.com/api_keys" + litellm_provider_prefix = "deepseek/" + use_fallback_pricing = False + + def __init__(self, api_key: str, provider_fee: float = 1.01): + super().__init__( + base_url=self.default_base_url, + api_key=api_key, + provider_fee=provider_fee, + upstream_name="DeepSeek", + ) + + @classmethod + def _build_from_row( + cls, provider_row: "UpstreamProviderRow" + ) -> "DeepSeekUpstreamProvider": + return cls(api_key=provider_row.api_key, provider_fee=provider_row.provider_fee) + + @classmethod + def get_provider_metadata(cls) -> dict[str, object]: + return { + "id": cls.provider_type, + "name": "DeepSeek", + "default_base_url": cls.default_base_url, + "fixed_base_url": True, + "platform_url": cls.platform_url, + } + + def _apply_provider_field(self, response_json: object) -> None: + # A first-party upstream: stamp "deepseek", not Generic's hostname. + BaseUpstreamProvider._apply_provider_field(self, response_json) + + def transform_model_name(self, model_id: str) -> str: + """Strip the 'deepseek/' prefix for DeepSeek API compatibility.""" + return model_id.removeprefix("deepseek/") + + def _native_pricing( + self, model_id: str, model_spec: dict + ) -> ResolvedPricing | None: + """Price ``model_id`` from the peak-rate table; ``None`` if absent.""" + rates = _PEAK_RATES.get(model_id) + if rates is None: + return None + input_usd, output_usd, cache_hit_usd = rates + input_modalities = ["text"] if model_id in _TEXT_ONLY else ["text", "image"] + return ResolvedPricing( + prompt=input_usd / 1_000_000, + completion=output_usd / 1_000_000, + context_length=_CONTEXT_LENGTH, + source="native", + max_completion_tokens=_MAX_OUTPUT_TOKENS, + input_cache_read=cache_hit_usd / 1_000_000, + input_modalities=input_modalities, + ) diff --git a/routstr/upstream/deepseek_v4_pricing_shim.py b/routstr/upstream/deepseek_v4_pricing_shim.py deleted file mode 100644 index ba0c392d..00000000 --- a/routstr/upstream/deepseek_v4_pricing_shim.py +++ /dev/null @@ -1,73 +0,0 @@ -"""TEMPORARY: local DeepSeek V4 pricing shim. - -litellm's bundled cost map does not yet ship ``deepseek-v4-flash`` / -``deepseek-v4-pro``. Without an entry, ``backfill_cache_pricing`` cannot find a -``cache_read_input_token_cost`` and cache reads fall back to the full input -rate — a large overcharge on cache hits (DeepSeek V4 hits are ~0.008-0.02x -input, i.e. cached tokens cost 50-120x less than regular input). - -This module injects the missing entries into ``litellm.model_cost`` at startup -so the existing backfill path resolves them. Rates mirror the canonical -``deepseek`` provider entries now in litellm's ``model_prices`` map -(``input_cost_per_token`` is the cache-*miss* rate; -``cache_read_input_token_cost`` is the cache-*hit* rate), sourced from -https://api-docs.deepseek.com/quick_start/pricing via -https://github.com/BerriAI/litellm/pull/26380 (issue -https://github.com/BerriAI/litellm/issues/30430). - -=== REMOVAL (once litellm ships these models) === -Delete this file and the single ``register_deepseek_v4_pricing()`` call in -``routstr/core/main.py``. Nothing else depends on it. Entries are only added -when absent, so a stale shim is harmless after upstream lands — but remove it. -""" - -import litellm - -from ..core import get_logger - -logger = get_logger(__name__) - -# USD per token. Mirrors the canonical ``deepseek`` provider entries in -# litellm's model_prices map (source: DeepSeek API pricing docs). Keep these in -# sync with ``litellm.model_cost["deepseek/deepseek-v4-*"]``. -_DEEPSEEK_V4_RATES: dict[str, dict[str, float]] = { - "deepseek-v4-flash": { - "input_cost_per_token": 1.4e-07, - "output_cost_per_token": 2.8e-07, - "cache_read_input_token_cost": 2.8e-09, - "cache_creation_input_token_cost": 0.0, - "input_cost_per_token_cache_hit": 2.8e-09, - }, - "deepseek-v4-pro": { - "input_cost_per_token": 4.35e-07, - "output_cost_per_token": 8.7e-07, - "cache_read_input_token_cost": 3.625e-09, - "cache_creation_input_token_cost": 0.0, - "input_cost_per_token_cache_hit": 3.625e-09, - }, -} - - -def register_deepseek_v4_pricing() -> None: - """Inject DeepSeek V4 pricing into ``litellm.model_cost`` if absent. - - Idempotent and non-destructive: a key already present in the cost map - (e.g. once litellm ships it) is left untouched. Registers both the bare - (``deepseek-v4-flash``) and prefixed (``deepseek/deepseek-v4-flash``) - spellings since ``backfill_cache_pricing`` tries both. - """ - added = [] - for bare, rates in _DEEPSEEK_V4_RATES.items(): - for key in (bare, f"deepseek/{bare}"): - if key in litellm.model_cost: - continue - entry: dict[str, object] = dict(rates) - entry["litellm_provider"] = "deepseek" - entry["mode"] = "chat" - litellm.model_cost[key] = entry - added.append(key) - if added: - logger.info( - "Registered temporary DeepSeek V4 pricing shim", - extra={"models": added}, - ) diff --git a/routstr/upstream/generic.py b/routstr/upstream/generic.py index c9edf109..1032e85d 100644 --- a/routstr/upstream/generic.py +++ b/routstr/upstream/generic.py @@ -28,7 +28,11 @@ class GenericUpstreamProvider(BaseUpstreamProvider): provider_type = "generic" default_base_url = "http://localhost:8888" - platform_url = None + platform_url: str | None = None + # Subclasses that own an authoritative price table set this False so a model + # the table misses imports disabled instead of taking a litellm/OpenRouter + # price that may undercut the upstream's own rate. + use_fallback_pricing = True def __init__( self, @@ -162,7 +166,7 @@ class GenericUpstreamProvider(BaseUpstreamProvider): model_spec = model_data.get("model_spec", {}) resolved = self._native_pricing(model_id, model_spec) - if resolved is None: + if resolved is None and self.use_fallback_pricing: resolved = await resolver.resolve(model_id) if resolved is None: diff --git a/routstr/upstream/helpers.py b/routstr/upstream/helpers.py index dc544b3a..c94733bc 100644 --- a/routstr/upstream/helpers.py +++ b/routstr/upstream/helpers.py @@ -272,6 +272,7 @@ async def _seed_providers_from_settings( ("PERPLEXITY_API_KEY", "perplexity", None, None), ("FIREWORKS_API_KEY", "fireworks", None, None), ("XAI_API_KEY", "xai", None, None), + ("DEEPSEEK_API_KEY", "deepseek", None, None), ("TINFOIL_API_KEY", "tinfoil", None, None), ("TYPESAFE_API_KEY", "typesafe", None, None), ] diff --git a/tests/integration/test_secret_bootstrap.py b/tests/integration/test_secret_bootstrap.py index 12f0d9a1..6dcdb990 100644 --- a/tests/integration/test_secret_bootstrap.py +++ b/tests/integration/test_secret_bootstrap.py @@ -470,7 +470,6 @@ async def test_startup_runs_bootstrap_before_settings_initialize( return None monkeypatch.setattr(main, "configure_litellm", lambda: None) - monkeypatch.setattr(main, "register_deepseek_v4_pricing", lambda: None) monkeypatch.setattr(main, "run_migrations", lambda: None) monkeypatch.setattr(main, "init_db", noop_init_db) monkeypatch.setattr(main, "create_session", fake_create_session) diff --git a/tests/unit/test_cache_pricing.py b/tests/unit/test_cache_pricing.py index cc4a4dbc..cfb7e9a6 100644 --- a/tests/unit/test_cache_pricing.py +++ b/tests/unit/test_cache_pricing.py @@ -31,15 +31,6 @@ from routstr.payment.models import ( backfill_cache_pricing, ) from routstr.upstream import GenericUpstreamProvider -from routstr.upstream.deepseek_v4_pricing_shim import register_deepseek_v4_pricing - - -@pytest.fixture(autouse=True) -def _deepseek_v4_pricing() -> None: - # litellm's bundled cost map lacks the DeepSeek V4 entries (they only - # appear when its remote map is reachable); production injects them at - # startup via this same shim. - register_deepseek_v4_pricing() def _make_model(model_id: str, pricing: Pricing) -> Model: diff --git a/tests/unit/test_upstream_deepseek.py b/tests/unit/test_upstream_deepseek.py new file mode 100644 index 00000000..f310f7ac --- /dev/null +++ b/tests/unit/test_upstream_deepseek.py @@ -0,0 +1,232 @@ +"""Unit tests for ``DeepSeekUpstreamProvider``. + +DeepSeek is priced from the provider's own peak-rate table, never from litellm +or OpenRouter: litellm's ``deepseek-v4-flash`` entry is stale and OpenRouter +resells below DeepSeek's peak rate, so either would bill under cost. These +tests pin the table prices (including the cache-hit rate), that a model the +table misses imports disabled without consulting the fallback chain, and that +``reasoning_content`` in history reaches DeepSeek untouched — thinking mode +with ``tools`` answers 400 when it is stripped. +""" + +from __future__ import annotations + +import json +from typing import Any +from unittest.mock import AsyncMock, Mock, patch + +import pytest + +from routstr.upstream import upstream_provider_classes +from routstr.upstream.deepseek import DeepSeekUpstreamProvider + + +class _FakeResponse: + def __init__(self, payload: dict[str, Any]) -> None: + self._payload = payload + + def raise_for_status(self) -> None: + return None + + def json(self) -> dict[str, Any]: + return self._payload + + +class _FakeAsyncClient: + def __init__(self, payload: dict[str, Any], calls: list[dict[str, Any]]) -> None: + self._payload = payload + self._calls = calls + + async def __aenter__(self) -> "_FakeAsyncClient": + return self + + async def __aexit__(self, *exc: object) -> bool: + return False + + async def get( + self, url: str, headers: dict[str, str] | None = None + ) -> _FakeResponse: + self._calls.append({"url": url, "headers": headers}) + return _FakeResponse(self._payload) + + +# Shape of DeepSeek's ``GET /models``: bare ids, no pricing. +CATALOG: dict[str, Any] = { + "object": "list", + "data": [ + {"id": "deepseek-flash", "object": "model", "owned_by": "deepseek"}, + {"id": "deepseek-v4-pro", "object": "model", "owned_by": "deepseek"}, + {"id": "deepseek-v4-flash", "object": "model", "owned_by": "deepseek"}, + {"id": "deepseek-chat", "object": "model", "owned_by": "deepseek"}, + ], +} + + +async def _fetch( + catalog: dict[str, Any] = CATALOG, +) -> tuple[dict[str, Any], list[dict[str, Any]], AsyncMock]: + calls: list[dict[str, Any]] = [] + fallback = AsyncMock(return_value=None) + provider = DeepSeekUpstreamProvider(api_key="sk-test") + with ( + patch( + "routstr.upstream.generic.httpx.AsyncClient", + lambda *args, **kwargs: _FakeAsyncClient(catalog, calls), + ), + patch("routstr.upstream.generic.FallbackPricingResolver.resolve", fallback), + ): + models = await provider.fetch_models() + return {m.id: m for m in models}, calls, fallback + + +def test_metadata_and_registration() -> None: + assert DeepSeekUpstreamProvider in upstream_provider_classes + assert DeepSeekUpstreamProvider.get_provider_metadata() == { + "id": "deepseek", + "name": "DeepSeek", + "default_base_url": "https://api.deepseek.com", + "fixed_base_url": True, + "platform_url": "https://platform.deepseek.com/api_keys", + } + + +def test_build_from_row_ignores_row_base_url() -> None: + row = Mock( + api_key="sk-row", provider_fee=1.05, base_url="https://elsewhere.example" + ) + provider = DeepSeekUpstreamProvider._build_from_row(row) + assert provider.api_key == "sk-row" + assert provider.provider_fee == 1.05 + assert provider.base_url == "https://api.deepseek.com" + + +def test_litellm_prefix_is_deepseek() -> None: + provider = DeepSeekUpstreamProvider(api_key="sk-test") + assert provider.get_litellm_provider_prefix() == "deepseek/" + + +@pytest.mark.parametrize( + "model_id,expected", + [ + ("deepseek/deepseek-v4-flash", "deepseek-v4-flash"), + ("deepseek-v4-flash", "deepseek-v4-flash"), + ("deepseek/deepseek-flash", "deepseek-flash"), + ], +) +def test_transform_model_name(model_id: str, expected: str) -> None: + provider = DeepSeekUpstreamProvider(api_key="sk-test") + assert provider.transform_model_name(model_id) == expected + + +def test_provider_field_names_deepseek_not_host() -> None: + provider = DeepSeekUpstreamProvider(api_key="sk-test") + payload: dict[str, Any] = {"id": "chatcmpl-1"} + provider._apply_provider_field(payload) + assert payload["provider"] == "deepseek" + + +@pytest.mark.asyncio +async def test_fetch_models_calls_deepseek_models_endpoint_with_key() -> None: + _, calls, _ = await _fetch() + assert calls == [ + { + "url": "https://api.deepseek.com/models", + "headers": {"Authorization": "Bearer sk-test"}, + } + ] + + +@pytest.mark.asyncio +@pytest.mark.parametrize( + "model_id,prompt,completion,cache_read", + [ + ("deepseek-flash", 0.30, 1.20, 0.006), + # Retired alias DeepSeek serves and bills as deepseek-flash. + ("deepseek-v4-flash", 0.30, 1.20, 0.006), + ("deepseek-v4-pro", 1.32, 3.96, 0.044), + ], +) +async def test_table_models_priced_at_peak_rate( + model_id: str, prompt: float, completion: float, cache_read: float +) -> None: + models, _, _ = await _fetch() + model = models[model_id] + assert model.enabled is True + assert model.pricing.prompt == pytest.approx(prompt / 1_000_000) + assert model.pricing.completion == pytest.approx(completion / 1_000_000) + assert model.pricing.input_cache_read == pytest.approx(cache_read / 1_000_000) + assert model.context_length == 1_000_000 + + +@pytest.mark.asyncio +async def test_vision_follows_the_model() -> None: + models, _, _ = await _fetch() + assert "image" in models["deepseek-flash"].architecture.input_modalities + assert models["deepseek-v4-pro"].architecture.input_modalities == ["text"] + + +@pytest.mark.asyncio +async def test_unlisted_model_imports_disabled_without_fallback() -> None: + """litellm prices ``deepseek-chat``; the provider must not take that price.""" + models, _, fallback = await _fetch() + model = models["deepseek-chat"] + assert model.enabled is False + assert model.pricing.prompt == 0.0 + assert model.pricing.completion == 0.0 + fallback.assert_not_awaited() + + +@pytest.mark.asyncio +async def test_cache_rate_survives_fee_and_is_not_replaced_by_litellm() -> None: + """litellm's stale ``deepseek-v4-flash`` cache rate (1.4e-08 in the bundled + map) must not replace the table's; backfill only fills an absent rate. The + fee applies to the cache rate like every other component. + + The litellm entry is pinned here because the remote cost map already + carries the table's rate, which would let an overwrite go unnoticed.""" + models, _, _ = await _fetch() + provider = DeepSeekUpstreamProvider(api_key="sk-test", provider_fee=1.05) + stale = {"cache_read_input_token_cost": 1.4e-08} + with patch("routstr.payment.models.litellm_cost_entry", return_value=stale): + priced = provider._apply_provider_fee_to_model(models["deepseek-v4-flash"]) + assert priced.pricing.input_cache_read == pytest.approx(0.006e-6 * 1.05) + assert priced.pricing.prompt == pytest.approx(0.30e-6 * 1.05) + # A cache hit costs 2% of a miss, not the full input rate. + assert priced.pricing.input_cache_read / priced.pricing.prompt == pytest.approx( + 0.02 + ) + + +@pytest.mark.asyncio +async def test_reasoning_content_in_history_reaches_upstream() -> None: + models, _, _ = await _fetch() + provider = DeepSeekUpstreamProvider(api_key="sk-test") + messages = [ + {"role": "user", "content": "weather in Paris?"}, + { + "role": "assistant", + "content": "", + "reasoning_content": "Need the weather tool.", + "tool_calls": [ + { + "id": "call_1", + "type": "function", + "function": {"name": "get_weather", "arguments": "{}"}, + } + ], + }, + {"role": "tool", "tool_call_id": "call_1", "content": "18C"}, + ] + body = json.dumps( + { + "model": "deepseek/deepseek-flash", + "messages": messages, + "tools": [{"type": "function", "function": {"name": "get_weather"}}], + } + ).encode() + out = provider.prepare_request_body(body, models["deepseek-flash"]) + + assert out is not None + sent = json.loads(out) + assert sent["model"] == "deepseek-flash" + assert sent["messages"] == messages