From 0d8dd35d560fcc4747fd77d7b2eddf64cdab7882 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Sat, 26 Sep 2026 02:31:23 +0200 Subject: [PATCH] clean up --- routstr/upstream/base.py | 9 +- routstr/upstream/messages_dispatch.py | 5 +- routstr/upstream/venice.py | 96 ++++++------------- .../test_venice_web_search_wire.py | 12 +-- tests/unit/test_upstream_venice.py | 11 +-- tests/unit/test_venice_web_search.py | 16 +--- 6 files changed, 40 insertions(+), 109 deletions(-) diff --git a/routstr/upstream/base.py b/routstr/upstream/base.py index 3f9d7bde..62d4916f 100644 --- a/routstr/upstream/base.py +++ b/routstr/upstream/base.py @@ -2507,13 +2507,8 @@ class BaseUpstreamProvider: return await messages_dispatch.aggregate_anthropic_events_to_message(iterator) def adapt_messages_request(self, body: dict, model_obj: Model) -> str: - """Rewrite an allowlisted /v1/messages body for this upstream. - - Returns a suffix appended to the upstream model name, empty when the - provider needs none. Subclasses override this to express an Anthropic - feature the upstream spells differently; the base forwards the body - untouched. - """ + """Rewrite a /v1/messages body in place for this upstream and return a + suffix for the upstream model name.""" return "" async def _dispatch_anthropic_messages( diff --git a/routstr/upstream/messages_dispatch.py b/routstr/upstream/messages_dispatch.py index 488129a8..b28ff7fe 100644 --- a/routstr/upstream/messages_dispatch.py +++ b/routstr/upstream/messages_dispatch.py @@ -467,10 +467,7 @@ async def dispatch_anthropic_messages( Shared by the bearer-key and x-cashu paths. Raises :class:`UpstreamError` on bad input or upstream failure. - ``adapt_request`` is the provider's last word on the allowlisted body: it - may rewrite it in place and returns a suffix for the upstream model name, - which is how a provider expresses a feature litellm would otherwise - translate into a parameter the upstream rejects. + ``adapt_request`` may rewrite the body and returns a model-name suffix. """ if not request_body: raise UpstreamError("Missing request body for /v1/messages", status_code=400) diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index 379a4a66..9cd96f84 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -14,16 +14,12 @@ if TYPE_CHECKING: logger = get_logger(__name__) -# ``GET /models`` defaults to ``type=text``, which is why a Venice account -# configured as a generic upstream never sees the rest of its catalog. +# Venice's ``/models`` returns only text models unless asked for all. _MODELS_TYPE_PARAM = "all" -# Families this proxy can both route and price. Image, audio, music and video -# are billed per clip or per second and return no usage object to settle -# against, so exposing them would hand out unpriced inference. +# Other families bill per clip or second and return no usage to settle. _SUPPORTED_TYPES = frozenset({"text", "embedding"}) -# Venice prices text in USD per million tokens; Routstr prices per token. _USD_PER_MILLION = 1_000_000.0 _ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = { @@ -31,33 +27,19 @@ _ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = { "embedding": ("text->embedding", ["text"], ["embedding"]), } -# Venice runs search itself and reports it back through ``venice_parameters``; -# it has no Anthropic-shaped server tool and rejects the ``web_search_options`` -# that litellm's Anthropic adapter derives from one. ``auto`` matches Anthropic -# semantics, where declaring the tool leaves the decision to the model. -# Citations are asked for because litellm's Anthropic response translation -# carries no ``venice_parameters``, so the inline ``^n^`` markers Venice writes -# into the text are the only way a caller sees that sources were used. +# ``auto`` leaves the search decision to the model, as Anthropic does. +# Citations are the caller's only sign of a search: litellm drops +# ``venice_parameters`` from the response, leaving the inline ``^n^`` markers. _WEB_SEARCH_SUFFIX = ":enable_web_search=auto&enable_web_citations=true" -# Anthropic web-search constraints with no Venice equivalent. Honouring the -# request means enforcing them, so a request that sets one is refused rather -# than answered by a search that ignored it. ``max_uses`` is absent on purpose: -# ``auto`` runs at most one search per request, so any cap of 1 or more is -# already met, while domain filters and location would be silently ignored. -# Only ``max_uses: 0``, a request for no search at all, cannot be honoured. +# Refused rather than silently ignored, since Venice cannot enforce them. _UNENFORCEABLE_WEB_SEARCH_KEYS = frozenset( {"allowed_domains", "blocked_domains", "user_location"} ) def _is_web_search_tool(tool: Any) -> bool: - """An Anthropic server-side web-search tool, by either of its markers. - - Matches litellm's own detection (``litellm/llms/anthropic/ - experimental_pass_through/adapters/transformation.py``), so every tool it - would turn into ``web_search_options`` is caught here first. - """ + """Mirror litellm's detection, so every tool it would rewrite is caught.""" if not isinstance(tool, dict): return False tool_type = tool.get("type") @@ -75,13 +57,19 @@ def _usd(entry: Any) -> float | None: return None -class VeniceUpstreamProvider(BaseUpstreamProvider): - """Upstream provider for the Venice.ai API. +def _is_unenforceable(key: str, value: Any) -> bool: + if key in _UNENFORCEABLE_WEB_SEARCH_KEYS: + return value is not None and value != [] + if key == "max_uses": + # ``auto`` searches at most once, so only an integer cap of 1+ is met. + is_count = isinstance(value, int) and not isinstance(value, bool) + return value is not None and not (is_count and value >= 1) + return False - Venice publishes a complete price book on its own catalog, so models are - built from that rather than matched against OpenRouter, which has never - heard of most of Venice's catalog. - """ + +class VeniceUpstreamProvider(BaseUpstreamProvider): + """Venice.ai upstream, priced from Venice's own catalog since OpenRouter + lists little of it.""" provider_type = "venice" default_base_url = "https://api.venice.ai/api/v1" @@ -115,13 +103,10 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): return model_id.removeprefix("venice/") def adapt_messages_request(self, body: dict, model_obj: Model) -> str: - """Trade an Anthropic web-search tool for Venice's own search switch. + """Swap an Anthropic web-search tool for Venice's model-name suffix. - Left in the body, litellm's Anthropic adapter rewrites the tool into a - top-level ``web_search_options``, which Venice answers with a 400. The - tool is lifted out here and the same intent re-expressed as a model - feature suffix, the one form of ``venice_parameters`` that survives - that adapter. + litellm would turn the tool into ``web_search_options``, which Venice + rejects with a 400. """ tools = body.get("tools") if not isinstance(tools, list): @@ -130,28 +115,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): if not search_tools: return "" - # A key carrying null or an empty list states no constraint, so it is - # read as absent rather than refused. ``auto`` runs at most one search, - # so only an integer ``max_uses`` of one or more is known to be met. unenforceable = sorted( { key for tool in search_tools for key, value in tool.items() - if ( - key in _UNENFORCEABLE_WEB_SEARCH_KEYS - and value is not None - and value != [] - ) - or ( - key == "max_uses" - and value is not None - and not ( - isinstance(value, int) - and not isinstance(value, bool) - and value >= 1 - ) - ) + if _is_unenforceable(key, value) } ) if unenforceable: @@ -175,15 +144,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): remaining = [tool for tool in tools if not _is_web_search_tool(tool)] if remaining: - # A caller's ``tool_choice: any`` is kept and litellm maps it to - # OpenAI ``required``, so one of the remaining function tools must - # now be called where Anthropic would have let a search satisfy it. - # Deliberate: OpenRouter never rewrites tool_choice for web search - # either, and guessing an alternative would change caller intent. + # ``tool_choice: any`` now requires a function tool; kept as is, + # like OpenRouter, rather than guessing the caller's intent. body["tools"] = remaining else: body.pop("tools", None) - # tool_choice without tools is rejected by OpenAI-shaped upstreams. + # OpenAI-shaped upstreams reject tool_choice without tools. body.pop("tool_choice", None) return _WEB_SEARCH_SUFFIX @@ -290,14 +256,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): if not isinstance(raw, dict): return None - # The ``extended`` tier some models charge past a context threshold is - # ignored: billing it would overcharge every request staying under it. + # The long-context ``extended`` tier is ignored; billing it would + # overcharge every shorter request. input_usd = _usd(raw.get("input")) output_usd = _usd(raw.get("output")) - # Embeddings produce no completion tokens, so only they may omit an - # output price. Anywhere else a missing or all-zero price would serve - # completions free and a negative one would credit the caller, the - # same guards ``generic.py`` applies to this price book. + # Only embeddings may omit an output price. Free or negative prices + # are dropped, as in ``generic.py``. if output_usd is None and model_type == "embedding": output_usd = 0.0 if input_usd is None or output_usd is None: diff --git a/tests/integration/test_venice_web_search_wire.py b/tests/integration/test_venice_web_search_wire.py index 2636af1f..4fc8811a 100644 --- a/tests/integration/test_venice_web_search_wire.py +++ b/tests/integration/test_venice_web_search_wire.py @@ -1,11 +1,5 @@ -"""What Routstr actually puts on the wire for a Venice web-search request. - -The unit tests stop at the kwargs handed to litellm. Everything that produced -the reported ``400 Unrecognized key(s) in object: 'web_search_options'`` -happened *after* that point, inside litellm's Anthropic adapter, so this test -runs the whole dispatch against a loopback OpenAI-compatible server and reads -the bytes Venice would have received. -""" +"""The bytes Routstr sends Venice for a web-search request, captured past +litellm's Anthropic adapter where the ``web_search_options`` 400 arose.""" from __future__ import annotations @@ -122,10 +116,8 @@ async def test_web_search_request_reaches_venice_in_its_own_shape( body = captured["body"] assert captured["path"] == "/v1/chat/completions" - # The reported 400, at the only place it could be observed. assert "web_search_options" not in body assert body["model"] == ( "deepseek-v4-flash-0731:enable_web_search=auto&enable_web_citations=true" ) - # The function tool still travels, in OpenAI's shape. assert [tool["function"]["name"] for tool in body["tools"]] == ["lookup"] diff --git a/tests/unit/test_upstream_venice.py b/tests/unit/test_upstream_venice.py index 742c38ae..b0fa5dc1 100644 --- a/tests/unit/test_upstream_venice.py +++ b/tests/unit/test_upstream_venice.py @@ -1,10 +1,4 @@ -"""Unit tests for ``VeniceUpstreamProvider.fetch_models``. - -Venice answers ``/models`` with only its text catalog unless ``type`` is -passed, which is why the same account configured as a generic upstream sees a -different catalog. These tests pin that query parameter, the per-token pricing -shape, and the families dropped as unpriceable. -""" +"""Unit tests for ``VeniceUpstreamProvider.fetch_models``.""" from __future__ import annotations @@ -174,8 +168,7 @@ def test_embedding_models_are_listed() -> None: def test_families_billed_per_clip_are_dropped() -> None: - """Image, audio and video return no usage to settle against, so listing - them here would hand out inference this provider cannot price.""" + """Image, audio and video return no usage to settle against.""" models, _ = _fetch() ids = {m.id for m in models} assert "venice-sd35" not in ids diff --git a/tests/unit/test_venice_web_search.py b/tests/unit/test_venice_web_search.py index a836e956..c968cd7e 100644 --- a/tests/unit/test_venice_web_search.py +++ b/tests/unit/test_venice_web_search.py @@ -1,10 +1,7 @@ """Venice web search over ``/v1/messages``. -litellm's Anthropic adapter rewrites an Anthropic server-side web-search tool -into a top-level ``web_search_options``, which Venice rejects with -``400 Unrecognized key(s) in object: 'web_search_options'``. These tests pin -the trade: the tool is lifted out of the body and the same intent re-expressed -as a Venice model feature suffix. +Venice rejects the ``web_search_options`` litellm derives from an Anthropic +web-search tool, so the tool is swapped for a model-name suffix. """ from __future__ import annotations @@ -203,8 +200,7 @@ def test_web_search_only_request_drops_tool_choice() -> None: @pytest.mark.asyncio async def test_claude_code_web_search_tool_is_accepted() -> None: - """Claude Code always sends ``max_uses: 8``; Venice's single ``auto`` - search already stays under any cap of one or more.""" + """Claude Code always sends ``max_uses: 8``.""" provider = VeniceUpstreamProvider(api_key="sk-test") tool = { "type": "web_search_20250305", @@ -233,8 +229,6 @@ def test_max_uses_of_one_or_absent_is_accepted(max_uses: Any) -> None: @pytest.mark.parametrize("max_uses", [0, -1, 1.5, True, "0", "8"]) def test_max_uses_other_than_a_positive_integer_is_refused(max_uses: Any) -> None: - """``auto`` may still search, so a cap below one cannot be met, and a - malformed cap cannot be shown to be met.""" provider = VeniceUpstreamProvider(api_key="sk-test") tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": max_uses} @@ -256,8 +250,6 @@ def test_tool_named_web_search_without_the_type_marker_is_caught() -> None: def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() -> None: - """The fix at its cause: run the real litellm translation over the body - this provider produces and assert the rejected key is never derived.""" from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( # noqa: E501 LiteLLMAnthropicMessagesAdapter, ) @@ -267,8 +259,6 @@ def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() - body = _body(tools=[WEB_SEARCH_TOOL, FUNCTION_TOOL]) def translate(request: dict[str, Any]) -> dict: - # litellm types the request as a TypedDict; these bodies are built - # from client JSON, so they are plain dicts at this seam. translated, _ = adapter.translate_anthropic_to_openai(request) # type: ignore[arg-type] return dict(translated)