mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
clean up
This commit is contained in:
@@ -2507,13 +2507,8 @@ class BaseUpstreamProvider:
|
||||
return await messages_dispatch.aggregate_anthropic_events_to_message(iterator)
|
||||
|
||||
def adapt_messages_request(self, body: dict, model_obj: Model) -> str:
|
||||
"""Rewrite an allowlisted /v1/messages body for this upstream.
|
||||
|
||||
Returns a suffix appended to the upstream model name, empty when the
|
||||
provider needs none. Subclasses override this to express an Anthropic
|
||||
feature the upstream spells differently; the base forwards the body
|
||||
untouched.
|
||||
"""
|
||||
"""Rewrite a /v1/messages body in place for this upstream and return a
|
||||
suffix for the upstream model name."""
|
||||
return ""
|
||||
|
||||
async def _dispatch_anthropic_messages(
|
||||
|
||||
@@ -467,10 +467,7 @@ async def dispatch_anthropic_messages(
|
||||
Shared by the bearer-key and x-cashu paths. Raises :class:`UpstreamError`
|
||||
on bad input or upstream failure.
|
||||
|
||||
``adapt_request`` is the provider's last word on the allowlisted body: it
|
||||
may rewrite it in place and returns a suffix for the upstream model name,
|
||||
which is how a provider expresses a feature litellm would otherwise
|
||||
translate into a parameter the upstream rejects.
|
||||
``adapt_request`` may rewrite the body and returns a model-name suffix.
|
||||
"""
|
||||
if not request_body:
|
||||
raise UpstreamError("Missing request body for /v1/messages", status_code=400)
|
||||
|
||||
+30
-66
@@ -14,16 +14,12 @@ if TYPE_CHECKING:
|
||||
|
||||
logger = get_logger(__name__)
|
||||
|
||||
# ``GET /models`` defaults to ``type=text``, which is why a Venice account
|
||||
# configured as a generic upstream never sees the rest of its catalog.
|
||||
# Venice's ``/models`` returns only text models unless asked for all.
|
||||
_MODELS_TYPE_PARAM = "all"
|
||||
|
||||
# Families this proxy can both route and price. Image, audio, music and video
|
||||
# are billed per clip or per second and return no usage object to settle
|
||||
# against, so exposing them would hand out unpriced inference.
|
||||
# Other families bill per clip or second and return no usage to settle.
|
||||
_SUPPORTED_TYPES = frozenset({"text", "embedding"})
|
||||
|
||||
# Venice prices text in USD per million tokens; Routstr prices per token.
|
||||
_USD_PER_MILLION = 1_000_000.0
|
||||
|
||||
_ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = {
|
||||
@@ -31,33 +27,19 @@ _ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = {
|
||||
"embedding": ("text->embedding", ["text"], ["embedding"]),
|
||||
}
|
||||
|
||||
# Venice runs search itself and reports it back through ``venice_parameters``;
|
||||
# it has no Anthropic-shaped server tool and rejects the ``web_search_options``
|
||||
# that litellm's Anthropic adapter derives from one. ``auto`` matches Anthropic
|
||||
# semantics, where declaring the tool leaves the decision to the model.
|
||||
# Citations are asked for because litellm's Anthropic response translation
|
||||
# carries no ``venice_parameters``, so the inline ``^n^`` markers Venice writes
|
||||
# into the text are the only way a caller sees that sources were used.
|
||||
# ``auto`` leaves the search decision to the model, as Anthropic does.
|
||||
# Citations are the caller's only sign of a search: litellm drops
|
||||
# ``venice_parameters`` from the response, leaving the inline ``^n^`` markers.
|
||||
_WEB_SEARCH_SUFFIX = ":enable_web_search=auto&enable_web_citations=true"
|
||||
|
||||
# Anthropic web-search constraints with no Venice equivalent. Honouring the
|
||||
# request means enforcing them, so a request that sets one is refused rather
|
||||
# than answered by a search that ignored it. ``max_uses`` is absent on purpose:
|
||||
# ``auto`` runs at most one search per request, so any cap of 1 or more is
|
||||
# already met, while domain filters and location would be silently ignored.
|
||||
# Only ``max_uses: 0``, a request for no search at all, cannot be honoured.
|
||||
# Refused rather than silently ignored, since Venice cannot enforce them.
|
||||
_UNENFORCEABLE_WEB_SEARCH_KEYS = frozenset(
|
||||
{"allowed_domains", "blocked_domains", "user_location"}
|
||||
)
|
||||
|
||||
|
||||
def _is_web_search_tool(tool: Any) -> bool:
|
||||
"""An Anthropic server-side web-search tool, by either of its markers.
|
||||
|
||||
Matches litellm's own detection (``litellm/llms/anthropic/
|
||||
experimental_pass_through/adapters/transformation.py``), so every tool it
|
||||
would turn into ``web_search_options`` is caught here first.
|
||||
"""
|
||||
"""Mirror litellm's detection, so every tool it would rewrite is caught."""
|
||||
if not isinstance(tool, dict):
|
||||
return False
|
||||
tool_type = tool.get("type")
|
||||
@@ -75,13 +57,19 @@ def _usd(entry: Any) -> float | None:
|
||||
return None
|
||||
|
||||
|
||||
class VeniceUpstreamProvider(BaseUpstreamProvider):
|
||||
"""Upstream provider for the Venice.ai API.
|
||||
def _is_unenforceable(key: str, value: Any) -> bool:
|
||||
if key in _UNENFORCEABLE_WEB_SEARCH_KEYS:
|
||||
return value is not None and value != []
|
||||
if key == "max_uses":
|
||||
# ``auto`` searches at most once, so only an integer cap of 1+ is met.
|
||||
is_count = isinstance(value, int) and not isinstance(value, bool)
|
||||
return value is not None and not (is_count and value >= 1)
|
||||
return False
|
||||
|
||||
Venice publishes a complete price book on its own catalog, so models are
|
||||
built from that rather than matched against OpenRouter, which has never
|
||||
heard of most of Venice's catalog.
|
||||
"""
|
||||
|
||||
class VeniceUpstreamProvider(BaseUpstreamProvider):
|
||||
"""Venice.ai upstream, priced from Venice's own catalog since OpenRouter
|
||||
lists little of it."""
|
||||
|
||||
provider_type = "venice"
|
||||
default_base_url = "https://api.venice.ai/api/v1"
|
||||
@@ -115,13 +103,10 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
|
||||
return model_id.removeprefix("venice/")
|
||||
|
||||
def adapt_messages_request(self, body: dict, model_obj: Model) -> str:
|
||||
"""Trade an Anthropic web-search tool for Venice's own search switch.
|
||||
"""Swap an Anthropic web-search tool for Venice's model-name suffix.
|
||||
|
||||
Left in the body, litellm's Anthropic adapter rewrites the tool into a
|
||||
top-level ``web_search_options``, which Venice answers with a 400. The
|
||||
tool is lifted out here and the same intent re-expressed as a model
|
||||
feature suffix, the one form of ``venice_parameters`` that survives
|
||||
that adapter.
|
||||
litellm would turn the tool into ``web_search_options``, which Venice
|
||||
rejects with a 400.
|
||||
"""
|
||||
tools = body.get("tools")
|
||||
if not isinstance(tools, list):
|
||||
@@ -130,28 +115,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
|
||||
if not search_tools:
|
||||
return ""
|
||||
|
||||
# A key carrying null or an empty list states no constraint, so it is
|
||||
# read as absent rather than refused. ``auto`` runs at most one search,
|
||||
# so only an integer ``max_uses`` of one or more is known to be met.
|
||||
unenforceable = sorted(
|
||||
{
|
||||
key
|
||||
for tool in search_tools
|
||||
for key, value in tool.items()
|
||||
if (
|
||||
key in _UNENFORCEABLE_WEB_SEARCH_KEYS
|
||||
and value is not None
|
||||
and value != []
|
||||
)
|
||||
or (
|
||||
key == "max_uses"
|
||||
and value is not None
|
||||
and not (
|
||||
isinstance(value, int)
|
||||
and not isinstance(value, bool)
|
||||
and value >= 1
|
||||
)
|
||||
)
|
||||
if _is_unenforceable(key, value)
|
||||
}
|
||||
)
|
||||
if unenforceable:
|
||||
@@ -175,15 +144,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
|
||||
|
||||
remaining = [tool for tool in tools if not _is_web_search_tool(tool)]
|
||||
if remaining:
|
||||
# A caller's ``tool_choice: any`` is kept and litellm maps it to
|
||||
# OpenAI ``required``, so one of the remaining function tools must
|
||||
# now be called where Anthropic would have let a search satisfy it.
|
||||
# Deliberate: OpenRouter never rewrites tool_choice for web search
|
||||
# either, and guessing an alternative would change caller intent.
|
||||
# ``tool_choice: any`` now requires a function tool; kept as is,
|
||||
# like OpenRouter, rather than guessing the caller's intent.
|
||||
body["tools"] = remaining
|
||||
else:
|
||||
body.pop("tools", None)
|
||||
# tool_choice without tools is rejected by OpenAI-shaped upstreams.
|
||||
# OpenAI-shaped upstreams reject tool_choice without tools.
|
||||
body.pop("tool_choice", None)
|
||||
|
||||
return _WEB_SEARCH_SUFFIX
|
||||
@@ -290,14 +256,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider):
|
||||
if not isinstance(raw, dict):
|
||||
return None
|
||||
|
||||
# The ``extended`` tier some models charge past a context threshold is
|
||||
# ignored: billing it would overcharge every request staying under it.
|
||||
# The long-context ``extended`` tier is ignored; billing it would
|
||||
# overcharge every shorter request.
|
||||
input_usd = _usd(raw.get("input"))
|
||||
output_usd = _usd(raw.get("output"))
|
||||
# Embeddings produce no completion tokens, so only they may omit an
|
||||
# output price. Anywhere else a missing or all-zero price would serve
|
||||
# completions free and a negative one would credit the caller, the
|
||||
# same guards ``generic.py`` applies to this price book.
|
||||
# Only embeddings may omit an output price. Free or negative prices
|
||||
# are dropped, as in ``generic.py``.
|
||||
if output_usd is None and model_type == "embedding":
|
||||
output_usd = 0.0
|
||||
if input_usd is None or output_usd is None:
|
||||
|
||||
@@ -1,11 +1,5 @@
|
||||
"""What Routstr actually puts on the wire for a Venice web-search request.
|
||||
|
||||
The unit tests stop at the kwargs handed to litellm. Everything that produced
|
||||
the reported ``400 Unrecognized key(s) in object: 'web_search_options'``
|
||||
happened *after* that point, inside litellm's Anthropic adapter, so this test
|
||||
runs the whole dispatch against a loopback OpenAI-compatible server and reads
|
||||
the bytes Venice would have received.
|
||||
"""
|
||||
"""The bytes Routstr sends Venice for a web-search request, captured past
|
||||
litellm's Anthropic adapter where the ``web_search_options`` 400 arose."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -122,10 +116,8 @@ async def test_web_search_request_reaches_venice_in_its_own_shape(
|
||||
|
||||
body = captured["body"]
|
||||
assert captured["path"] == "/v1/chat/completions"
|
||||
# The reported 400, at the only place it could be observed.
|
||||
assert "web_search_options" not in body
|
||||
assert body["model"] == (
|
||||
"deepseek-v4-flash-0731:enable_web_search=auto&enable_web_citations=true"
|
||||
)
|
||||
# The function tool still travels, in OpenAI's shape.
|
||||
assert [tool["function"]["name"] for tool in body["tools"]] == ["lookup"]
|
||||
|
||||
@@ -1,10 +1,4 @@
|
||||
"""Unit tests for ``VeniceUpstreamProvider.fetch_models``.
|
||||
|
||||
Venice answers ``/models`` with only its text catalog unless ``type`` is
|
||||
passed, which is why the same account configured as a generic upstream sees a
|
||||
different catalog. These tests pin that query parameter, the per-token pricing
|
||||
shape, and the families dropped as unpriceable.
|
||||
"""
|
||||
"""Unit tests for ``VeniceUpstreamProvider.fetch_models``."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -174,8 +168,7 @@ def test_embedding_models_are_listed() -> None:
|
||||
|
||||
|
||||
def test_families_billed_per_clip_are_dropped() -> None:
|
||||
"""Image, audio and video return no usage to settle against, so listing
|
||||
them here would hand out inference this provider cannot price."""
|
||||
"""Image, audio and video return no usage to settle against."""
|
||||
models, _ = _fetch()
|
||||
ids = {m.id for m in models}
|
||||
assert "venice-sd35" not in ids
|
||||
|
||||
@@ -1,10 +1,7 @@
|
||||
"""Venice web search over ``/v1/messages``.
|
||||
|
||||
litellm's Anthropic adapter rewrites an Anthropic server-side web-search tool
|
||||
into a top-level ``web_search_options``, which Venice rejects with
|
||||
``400 Unrecognized key(s) in object: 'web_search_options'``. These tests pin
|
||||
the trade: the tool is lifted out of the body and the same intent re-expressed
|
||||
as a Venice model feature suffix.
|
||||
Venice rejects the ``web_search_options`` litellm derives from an Anthropic
|
||||
web-search tool, so the tool is swapped for a model-name suffix.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -203,8 +200,7 @@ def test_web_search_only_request_drops_tool_choice() -> None:
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_claude_code_web_search_tool_is_accepted() -> None:
|
||||
"""Claude Code always sends ``max_uses: 8``; Venice's single ``auto``
|
||||
search already stays under any cap of one or more."""
|
||||
"""Claude Code always sends ``max_uses: 8``."""
|
||||
provider = VeniceUpstreamProvider(api_key="sk-test")
|
||||
tool = {
|
||||
"type": "web_search_20250305",
|
||||
@@ -233,8 +229,6 @@ def test_max_uses_of_one_or_absent_is_accepted(max_uses: Any) -> None:
|
||||
|
||||
@pytest.mark.parametrize("max_uses", [0, -1, 1.5, True, "0", "8"])
|
||||
def test_max_uses_other_than_a_positive_integer_is_refused(max_uses: Any) -> None:
|
||||
"""``auto`` may still search, so a cap below one cannot be met, and a
|
||||
malformed cap cannot be shown to be met."""
|
||||
provider = VeniceUpstreamProvider(api_key="sk-test")
|
||||
tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": max_uses}
|
||||
|
||||
@@ -256,8 +250,6 @@ def test_tool_named_web_search_without_the_type_marker_is_caught() -> None:
|
||||
|
||||
|
||||
def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() -> None:
|
||||
"""The fix at its cause: run the real litellm translation over the body
|
||||
this provider produces and assert the rejected key is never derived."""
|
||||
from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( # noqa: E501
|
||||
LiteLLMAnthropicMessagesAdapter,
|
||||
)
|
||||
@@ -267,8 +259,6 @@ def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() -
|
||||
body = _body(tools=[WEB_SEARCH_TOOL, FUNCTION_TOOL])
|
||||
|
||||
def translate(request: dict[str, Any]) -> dict:
|
||||
# litellm types the request as a TypedDict; these bodies are built
|
||||
# from client JSON, so they are plain dicts at this seam.
|
||||
translated, _ = adapter.translate_anthropic_to_openai(request) # type: ignore[arg-type]
|
||||
return dict(translated)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user