This commit is contained in:
9qeklajc
2026-09-26 02:31:23 +02:00
parent 6f2c10bfd0
commit 0d8dd35d56
6 changed files with 40 additions and 109 deletions
+2 -7
View File
@@ -2507,13 +2507,8 @@ class BaseUpstreamProvider:
return await messages_dispatch.aggregate_anthropic_events_to_message(iterator)
def adapt_messages_request(self, body: dict, model_obj: Model) -> str:
"""Rewrite an allowlisted /v1/messages body for this upstream.
Returns a suffix appended to the upstream model name, empty when the
provider needs none. Subclasses override this to express an Anthropic
feature the upstream spells differently; the base forwards the body
untouched.
"""
"""Rewrite a /v1/messages body in place for this upstream and return a
suffix for the upstream model name."""
return ""
async def _dispatch_anthropic_messages(
+1 -4
View File
@@ -467,10 +467,7 @@ async def dispatch_anthropic_messages(
Shared by the bearer-key and x-cashu paths. Raises :class:`UpstreamError`
on bad input or upstream failure.
``adapt_request`` is the provider's last word on the allowlisted body: it
may rewrite it in place and returns a suffix for the upstream model name,
which is how a provider expresses a feature litellm would otherwise
translate into a parameter the upstream rejects.
``adapt_request`` may rewrite the body and returns a model-name suffix.
"""
if not request_body:
raise UpstreamError("Missing request body for /v1/messages", status_code=400)
+30 -66
View File
@@ -14,16 +14,12 @@ if TYPE_CHECKING:
logger = get_logger(__name__)
# ``GET /models`` defaults to ``type=text``, which is why a Venice account
# configured as a generic upstream never sees the rest of its catalog.
# Venice's ``/models`` returns only text models unless asked for all.
_MODELS_TYPE_PARAM = "all"
# Families this proxy can both route and price. Image, audio, music and video
# are billed per clip or per second and return no usage object to settle
# against, so exposing them would hand out unpriced inference.
# Other families bill per clip or second and return no usage to settle.
_SUPPORTED_TYPES = frozenset({"text", "embedding"})
# Venice prices text in USD per million tokens; Routstr prices per token.
_USD_PER_MILLION = 1_000_000.0
_ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = {
@@ -31,33 +27,19 @@ _ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = {
"embedding": ("text->embedding", ["text"], ["embedding"]),
}
# Venice runs search itself and reports it back through ``venice_parameters``;
# it has no Anthropic-shaped server tool and rejects the ``web_search_options``
# that litellm's Anthropic adapter derives from one. ``auto`` matches Anthropic
# semantics, where declaring the tool leaves the decision to the model.
# Citations are asked for because litellm's Anthropic response translation
# carries no ``venice_parameters``, so the inline ``^n^`` markers Venice writes
# into the text are the only way a caller sees that sources were used.
# ``auto`` leaves the search decision to the model, as Anthropic does.
# Citations are the caller's only sign of a search: litellm drops
# ``venice_parameters`` from the response, leaving the inline ``^n^`` markers.
_WEB_SEARCH_SUFFIX = ":enable_web_search=auto&enable_web_citations=true"
# Anthropic web-search constraints with no Venice equivalent. Honouring the
# request means enforcing them, so a request that sets one is refused rather
# than answered by a search that ignored it. ``max_uses`` is absent on purpose:
# ``auto`` runs at most one search per request, so any cap of 1 or more is
# already met, while domain filters and location would be silently ignored.
# Only ``max_uses: 0``, a request for no search at all, cannot be honoured.
# Refused rather than silently ignored, since Venice cannot enforce them.
_UNENFORCEABLE_WEB_SEARCH_KEYS = frozenset(
{"allowed_domains", "blocked_domains", "user_location"}
)
def _is_web_search_tool(tool: Any) -> bool:
"""An Anthropic server-side web-search tool, by either of its markers.
Matches litellm's own detection (``litellm/llms/anthropic/
experimental_pass_through/adapters/transformation.py``), so every tool it
would turn into ``web_search_options`` is caught here first.
"""
"""Mirror litellm's detection, so every tool it would rewrite is caught."""
if not isinstance(tool, dict):
return False
tool_type = tool.get("type")
@@ -75,13 +57,19 @@ def _usd(entry: Any) -> float | None:
return None
class VeniceUpstreamProvider(BaseUpstreamProvider):
"""Upstream provider for the Venice.ai API.
def _is_unenforceable(key: str, value: Any) -> bool:
if key in _UNENFORCEABLE_WEB_SEARCH_KEYS:
return value is not None and value != []
if key == "max_uses":
# ``auto`` searches at most once, so only an integer cap of 1+ is met.
is_count = isinstance(value, int) and not isinstance(value, bool)
return value is not None and not (is_count and value >= 1)
return False
Venice publishes a complete price book on its own catalog, so models are
built from that rather than matched against OpenRouter, which has never
heard of most of Venice's catalog.
"""
class VeniceUpstreamProvider(BaseUpstreamProvider):
"""Venice.ai upstream, priced from Venice's own catalog since OpenRouter
lists little of it."""
provider_type = "venice"
default_base_url = "https://api.venice.ai/api/v1"
@@ -115,13 +103,10 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
return model_id.removeprefix("venice/")
def adapt_messages_request(self, body: dict, model_obj: Model) -> str:
"""Trade an Anthropic web-search tool for Venice's own search switch.
"""Swap an Anthropic web-search tool for Venice's model-name suffix.
Left in the body, litellm's Anthropic adapter rewrites the tool into a
top-level ``web_search_options``, which Venice answers with a 400. The
tool is lifted out here and the same intent re-expressed as a model
feature suffix, the one form of ``venice_parameters`` that survives
that adapter.
litellm would turn the tool into ``web_search_options``, which Venice
rejects with a 400.
"""
tools = body.get("tools")
if not isinstance(tools, list):
@@ -130,28 +115,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
if not search_tools:
return ""
# A key carrying null or an empty list states no constraint, so it is
# read as absent rather than refused. ``auto`` runs at most one search,
# so only an integer ``max_uses`` of one or more is known to be met.
unenforceable = sorted(
{
key
for tool in search_tools
for key, value in tool.items()
if (
key in _UNENFORCEABLE_WEB_SEARCH_KEYS
and value is not None
and value != []
)
or (
key == "max_uses"
and value is not None
and not (
isinstance(value, int)
and not isinstance(value, bool)
and value >= 1
)
)
if _is_unenforceable(key, value)
}
)
if unenforceable:
@@ -175,15 +144,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
remaining = [tool for tool in tools if not _is_web_search_tool(tool)]
if remaining:
# A caller's ``tool_choice: any`` is kept and litellm maps it to
# OpenAI ``required``, so one of the remaining function tools must
# now be called where Anthropic would have let a search satisfy it.
# Deliberate: OpenRouter never rewrites tool_choice for web search
# either, and guessing an alternative would change caller intent.
# ``tool_choice: any`` now requires a function tool; kept as is,
# like OpenRouter, rather than guessing the caller's intent.
body["tools"] = remaining
else:
body.pop("tools", None)
# tool_choice without tools is rejected by OpenAI-shaped upstreams.
# OpenAI-shaped upstreams reject tool_choice without tools.
body.pop("tool_choice", None)
return _WEB_SEARCH_SUFFIX
@@ -290,14 +256,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
if not isinstance(raw, dict):
return None
# The ``extended`` tier some models charge past a context threshold is
# ignored: billing it would overcharge every request staying under it.
# The long-context ``extended`` tier is ignored; billing it would
# overcharge every shorter request.
input_usd = _usd(raw.get("input"))
output_usd = _usd(raw.get("output"))
# Embeddings produce no completion tokens, so only they may omit an
# output price. Anywhere else a missing or all-zero price would serve
# completions free and a negative one would credit the caller, the
# same guards ``generic.py`` applies to this price book.
# Only embeddings may omit an output price. Free or negative prices
# are dropped, as in ``generic.py``.
if output_usd is None and model_type == "embedding":
output_usd = 0.0
if input_usd is None or output_usd is None:
@@ -1,11 +1,5 @@
"""What Routstr actually puts on the wire for a Venice web-search request.
The unit tests stop at the kwargs handed to litellm. Everything that produced
the reported ``400 Unrecognized key(s) in object: 'web_search_options'``
happened *after* that point, inside litellm's Anthropic adapter, so this test
runs the whole dispatch against a loopback OpenAI-compatible server and reads
the bytes Venice would have received.
"""
"""The bytes Routstr sends Venice for a web-search request, captured past
litellm's Anthropic adapter where the ``web_search_options`` 400 arose."""
from __future__ import annotations
@@ -122,10 +116,8 @@ async def test_web_search_request_reaches_venice_in_its_own_shape(
body = captured["body"]
assert captured["path"] == "/v1/chat/completions"
# The reported 400, at the only place it could be observed.
assert "web_search_options" not in body
assert body["model"] == (
"deepseek-v4-flash-0731:enable_web_search=auto&enable_web_citations=true"
)
# The function tool still travels, in OpenAI's shape.
assert [tool["function"]["name"] for tool in body["tools"]] == ["lookup"]
+2 -9
View File
@@ -1,10 +1,4 @@
"""Unit tests for ``VeniceUpstreamProvider.fetch_models``.
Venice answers ``/models`` with only its text catalog unless ``type`` is
passed, which is why the same account configured as a generic upstream sees a
different catalog. These tests pin that query parameter, the per-token pricing
shape, and the families dropped as unpriceable.
"""
"""Unit tests for ``VeniceUpstreamProvider.fetch_models``."""
from __future__ import annotations
@@ -174,8 +168,7 @@ def test_embedding_models_are_listed() -> None:
def test_families_billed_per_clip_are_dropped() -> None:
"""Image, audio and video return no usage to settle against, so listing
them here would hand out inference this provider cannot price."""
"""Image, audio and video return no usage to settle against."""
models, _ = _fetch()
ids = {m.id for m in models}
assert "venice-sd35" not in ids
+3 -13
View File
@@ -1,10 +1,7 @@
"""Venice web search over ``/v1/messages``.
litellm's Anthropic adapter rewrites an Anthropic server-side web-search tool
into a top-level ``web_search_options``, which Venice rejects with
``400 Unrecognized key(s) in object: 'web_search_options'``. These tests pin
the trade: the tool is lifted out of the body and the same intent re-expressed
as a Venice model feature suffix.
Venice rejects the ``web_search_options`` litellm derives from an Anthropic
web-search tool, so the tool is swapped for a model-name suffix.
"""
from __future__ import annotations
@@ -203,8 +200,7 @@ def test_web_search_only_request_drops_tool_choice() -> None:
@pytest.mark.asyncio
async def test_claude_code_web_search_tool_is_accepted() -> None:
"""Claude Code always sends ``max_uses: 8``; Venice's single ``auto``
search already stays under any cap of one or more."""
"""Claude Code always sends ``max_uses: 8``."""
provider = VeniceUpstreamProvider(api_key="sk-test")
tool = {
"type": "web_search_20250305",
@@ -233,8 +229,6 @@ def test_max_uses_of_one_or_absent_is_accepted(max_uses: Any) -> None:
@pytest.mark.parametrize("max_uses", [0, -1, 1.5, True, "0", "8"])
def test_max_uses_other_than_a_positive_integer_is_refused(max_uses: Any) -> None:
"""``auto`` may still search, so a cap below one cannot be met, and a
malformed cap cannot be shown to be met."""
provider = VeniceUpstreamProvider(api_key="sk-test")
tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": max_uses}
@@ -256,8 +250,6 @@ def test_tool_named_web_search_without_the_type_marker_is_caught() -> None:
def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() -> None:
"""The fix at its cause: run the real litellm translation over the body
this provider produces and assert the rejected key is never derived."""
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( # noqa: E501
LiteLLMAnthropicMessagesAdapter,
)
@@ -267,8 +259,6 @@ def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() -
body = _body(tools=[WEB_SEARCH_TOOL, FUNCTION_TOOL])
def translate(request: dict[str, Any]) -> dict:
# litellm types the request as a TypedDict; these bodies are built
# from client JSON, so they are plain dicts at this seam.
translated, _ = adapter.translate_anthropic_to_openai(request) # type: ignore[arg-type]
return dict(translated)