fix: add backoff so litellm's deepseek /v1/messages stream works on prod installs

This commit is contained in:
9qeklajc
2026-10-01 11:49:59 +02:00
parent e589b37bc5
commit 6492261e49
5 changed files with 82 additions and 4 deletions
+1 -1
View File
@@ -65,7 +65,7 @@ in `routstr/upstream/deepseek.py`, not from litellm or OpenRouter:
does not price shows up disabled in the Admin Dashboard. Enable it with a does not price shows up disabled in the Admin Dashboard. Enable it with a
manual price, or add it to the table. manual price, or add it to the table.
- **Cache hits** bill at DeepSeek's cache-hit rate (about 2% of the input - **Cache hits** bill at DeepSeek's cache-hit rate (about 2% of the input
rate). rate on flash, about 3% on pro).
Thinking-mode `reasoning_content` is returned to clients unchanged in Thinking-mode `reasoning_content` is returned to clients unchanged in
responses, and forwarded unchanged when it appears in conversation history. responses, and forwarded unchanged when it appears in conversation history.
+1
View File
@@ -22,6 +22,7 @@ dependencies = [
"pillow>=10", "pillow>=10",
"openai>=1.98.0", "openai>=1.98.0",
"litellm>=1.101.2,<1.102", "litellm>=1.101.2,<1.102",
"backoff>=2.2", # litellm's native Anthropic-messages streaming (e.g. deepseek/) imports litellm.proxy, which needs it
"orjson>=3.10", "orjson>=3.10",
] ]
+4 -3
View File
@@ -1,9 +1,10 @@
"""First-class upstream for the DeepSeek API. """First-class upstream for the DeepSeek API.
Pricing comes from ``_PEAK_RATES`` below, not from litellm or OpenRouter: Pricing comes from ``_PEAK_RATES`` below, not from litellm or OpenRouter:
litellm's bundled ``deepseek-v4-flash`` entry is stale, the OpenRouter feed litellm's bundled ``deepseek-v4-flash`` entry is stale (input, output and cache
carries resale prices below DeepSeek's own peak rate, and neither knows the rates alike), the OpenRouter feed carries resale prices below DeepSeek's own
current ``deepseek-flash`` id. A model DeepSeek lists that the table does not peak rate, and neither the bundled map nor OpenRouter knows the current
``deepseek-flash`` id. A model DeepSeek lists that the table does not
cover is imported disabled rather than priced from those sources. cover is imported disabled rather than priced from those sources.
DeepSeek bills peak hours at twice the off-peak rate. The node has one flat DeepSeek bills peak hours at twice the off-peak rate. The node has one flat
+65
View File
@@ -12,9 +12,13 @@ with ``tools`` answers 400 when it is stripped.
from __future__ import annotations from __future__ import annotations
import json import json
import threading
from collections.abc import Iterator
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from typing import Any from typing import Any
from unittest.mock import AsyncMock, Mock, patch from unittest.mock import AsyncMock, Mock, patch
import litellm
import pytest import pytest
from routstr.upstream import upstream_provider_classes from routstr.upstream import upstream_provider_classes
@@ -230,3 +234,64 @@ async def test_reasoning_content_in_history_reaches_upstream() -> None:
sent = json.loads(out) sent = json.loads(out)
assert sent["model"] == "deepseek-flash" assert sent["model"] == "deepseek-flash"
assert sent["messages"] == messages assert sent["messages"] == messages
_ANTHROPIC_SSE = (
b"event: message_start\n"
b'data: {"type":"message_start","message":{"id":"msg_1","type":"message",'
b'"role":"assistant","model":"deepseek-flash","content":[],'
b'"stop_reason":null,"usage":{"input_tokens":3,"output_tokens":0}}}\n\n'
b"event: message_stop\n"
b'data: {"type":"message_stop"}\n\n'
)
@pytest.fixture
def anthropic_stub() -> Iterator[tuple[str, list[tuple[str, dict[str, Any]]]]]:
"""Loopback stand-in for DeepSeek's Anthropic-format endpoint."""
seen: list[tuple[str, dict[str, Any]]] = []
class Handler(BaseHTTPRequestHandler):
def do_POST(self) -> None:
length = int(self.headers["Content-Length"])
seen.append((self.path, json.loads(self.rfile.read(length))))
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Content-Length", str(len(_ANTHROPIC_SSE)))
self.end_headers()
self.wfile.write(_ANTHROPIC_SSE)
def log_message(self, *args: Any) -> None:
return None
server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
try:
yield f"http://127.0.0.1:{server.server_address[1]}", seen
finally:
server.shutdown()
server.server_close()
@pytest.mark.asyncio
async def test_messages_stream_reaches_deepseek_anthropic_endpoint(
anthropic_stub: tuple[str, list[tuple[str, dict[str, Any]]]],
) -> None:
# litellm sends deepseek/ Messages calls to DeepSeek's /anthropic endpoint;
# its stream iterator imports litellm.proxy, which needs ``backoff``.
api_base, seen = anthropic_stub
stream = await litellm.anthropic.messages.acreate(
model=DeepSeekUpstreamProvider.litellm_provider_prefix + "deepseek-flash",
messages=[{"role": "user", "content": "hi"}],
max_tokens=8,
stream=True,
api_key="sk-test",
api_base=api_base,
)
chunks = [chunk async for chunk in stream] # type: ignore[union-attr]
assert b"message_stop" in b"".join(chunks)
assert len(seen) == 1
assert seen[0][0] == "/anthropic/v1/messages"
assert seen[0][1]["model"] == "deepseek-flash"
Generated
+11
View File
@@ -282,6 +282,15 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/77/06/bb80f5f86020c4551da315d78b3ab75e8228f89f0162f2c3a819e407941a/attrs-25.3.0-py3-none-any.whl", hash = "sha256:427318ce031701fea540783410126f03899a97ffc6f61596ad581ac2e40e3bc3", size = 63815, upload-time = "2025-03-13T11:10:21.14Z" }, { url = "https://files.pythonhosted.org/packages/77/06/bb80f5f86020c4551da315d78b3ab75e8228f89f0162f2c3a819e407941a/attrs-25.3.0-py3-none-any.whl", hash = "sha256:427318ce031701fea540783410126f03899a97ffc6f61596ad581ac2e40e3bc3", size = 63815, upload-time = "2025-03-13T11:10:21.14Z" },
] ]
[[package]]
name = "backoff"
version = "2.2.1"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/47/d7/5bbeb12c44d7c4f2fb5b56abce497eb5ed9f34d85701de869acedd602619/backoff-2.2.1.tar.gz", hash = "sha256:03f829f5bb1923180821643f8753b0502c3b682293992485b0eef2807afa5cba", size = 17001, upload-time = "2022-10-05T19:19:32.061Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/df/73/b6e24bd22e6720ca8ee9a85a0c4a2971af8497d8f3193fa05390cbd46e09/backoff-2.2.1-py3-none-any.whl", hash = "sha256:63579f9a0628e06278f7e47b7d7d5b6ce20dc65c5e96a6f3ca99a6adca0396e8", size = 15148, upload-time = "2022-10-05T19:19:30.546Z" },
]
[[package]] [[package]]
name = "base58" name = "base58"
version = "2.1.1" version = "2.1.1"
@@ -2711,6 +2720,7 @@ source = { editable = "." }
dependencies = [ dependencies = [
{ name = "aiosqlite" }, { name = "aiosqlite" },
{ name = "alembic" }, { name = "alembic" },
{ name = "backoff" },
{ name = "cashu" }, { name = "cashu" },
{ name = "fastapi", extra = ["standard-no-fastapi-cloud-cli"] }, { name = "fastapi", extra = ["standard-no-fastapi-cloud-cli"] },
{ name = "greenlet" }, { name = "greenlet" },
@@ -2747,6 +2757,7 @@ dev = [
requires-dist = [ requires-dist = [
{ name = "aiosqlite", specifier = ">=0.20" }, { name = "aiosqlite", specifier = ">=0.20" },
{ name = "alembic", specifier = ">=1.13" }, { name = "alembic", specifier = ">=1.13" },
{ name = "backoff", specifier = ">=2.2" },
{ name = "cashu", specifier = ">=0.20" }, { name = "cashu", specifier = ">=0.20" },
{ name = "fastapi", extras = ["standard-no-fastapi-cloud-cli"], specifier = ">=0.141" }, { name = "fastapi", extras = ["standard-no-fastapi-cloud-cli"], specifier = ">=0.141" },
{ name = "greenlet", specifier = ">=3.2.1" }, { name = "greenlet", specifier = ">=3.2.1" },