From f77896f19ac35fdd20b8ce3584613e98a8a09a54 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 00:19:17 +0200 Subject: [PATCH 01/12] feat: add venice upstream provider --- routstr/upstream/__init__.py | 2 + routstr/upstream/venice.py | 194 +++++++++++++++++++++++++ tests/unit/test_upstream_venice.py | 221 +++++++++++++++++++++++++++++ 3 files changed, 417 insertions(+) create mode 100644 routstr/upstream/venice.py create mode 100644 tests/unit/test_upstream_venice.py diff --git a/routstr/upstream/__init__.py b/routstr/upstream/__init__.py index edac0020..85d094e8 100644 --- a/routstr/upstream/__init__.py +++ b/routstr/upstream/__init__.py @@ -13,6 +13,7 @@ from .ppqai import PPQAIUpstreamProvider from .routstr import RoutstrUpstreamProvider from .tinfoil import TinfoilUpstreamProvider from .typesafe import TypeSafeUpstreamProvider +from .venice import VeniceUpstreamProvider from .xai import XAIUpstreamProvider upstream_provider_classes: list[type[BaseUpstreamProvider]] = [ @@ -30,6 +31,7 @@ upstream_provider_classes: list[type[BaseUpstreamProvider]] = [ RoutstrUpstreamProvider, TinfoilUpstreamProvider, TypeSafeUpstreamProvider, + VeniceUpstreamProvider, XAIUpstreamProvider, ] """List of all upstream classes""" diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py new file mode 100644 index 00000000..5949c205 --- /dev/null +++ b/routstr/upstream/venice.py @@ -0,0 +1,194 @@ +from __future__ import annotations + +from typing import TYPE_CHECKING, Any + +import httpx + +from ..core.logging import get_logger +from ..payment.models import Architecture, Model, Pricing, TopProvider +from .base import BaseUpstreamProvider + +if TYPE_CHECKING: + from ..core.db import UpstreamProviderRow + +logger = get_logger(__name__) + +# ``GET /models`` defaults to ``type=text``, which is why a Venice account +# configured as a generic upstream never sees the rest of its catalog. +_MODELS_TYPE_PARAM = "all" + +# Families this proxy can both route and price. Image, audio, music and video +# are billed per clip or per second and return no usage object to settle +# against, so exposing them would hand out unpriced inference. +_SUPPORTED_TYPES = frozenset({"text", "embedding"}) + +# Venice prices text in USD per million tokens; Routstr prices per token. +_USD_PER_MILLION = 1_000_000.0 + +_ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = { + "text": ("text->text", ["text"], ["text"]), + "embedding": ("text->embedding", ["text"], ["embedding"]), +} + + +def _usd(entry: Any) -> float | None: + """Read the USD leg of a Venice ``{usd, diem}`` price pair.""" + if isinstance(entry, dict): + value = entry.get("usd") + if isinstance(value, (int, float)) and not isinstance(value, bool): + return float(value) + return None + + +class VeniceUpstreamProvider(BaseUpstreamProvider): + """Upstream provider for the Venice.ai API. + + Venice publishes a complete price book on its own catalog, so models are + built from that rather than matched against OpenRouter, which has never + heard of most of Venice's catalog. + """ + + provider_type = "venice" + default_base_url = "https://api.venice.ai/api/v1" + platform_url = "https://venice.ai/settings/api" + + def __init__(self, api_key: str, provider_fee: float = 1.01): + super().__init__( + base_url=self.default_base_url, api_key=api_key, provider_fee=provider_fee + ) + + @classmethod + def _build_from_row( + cls, provider_row: "UpstreamProviderRow" + ) -> "VeniceUpstreamProvider": + return cls( + api_key=provider_row.api_key, + provider_fee=provider_row.provider_fee, + ) + + @classmethod + def get_provider_metadata(cls) -> dict[str, object]: + return { + "id": cls.provider_type, + "name": "Venice AI", + "default_base_url": cls.default_base_url, + "fixed_base_url": True, + "platform_url": cls.platform_url, + } + + def transform_model_name(self, model_id: str) -> str: + return model_id.removeprefix("venice/") + + async def _fetch_provider_models(self) -> dict: + url = f"{self.base_url.rstrip('/')}/models" + headers = {"Authorization": f"Bearer {self.api_key}"} if self.api_key else None + async with httpx.AsyncClient(timeout=30.0) as client: + response = await client.get( + url, params={"type": _MODELS_TYPE_PARAM}, headers=headers + ) + response.raise_for_status() + return response.json() + + async def fetch_models(self) -> list[Model]: + try: + payload = await self._fetch_provider_models() + except Exception as e: + logger.error( + "Error fetching Venice models", + extra={"error": str(e), "error_type": type(e).__name__}, + ) + return [] + + models: list[Model] = [] + skipped: list[str] = [] + for entry in payload.get("data", []): + if not isinstance(entry, dict): + continue + try: + model = self._parse_model(entry) + except Exception as e: + logger.warning( + "Failed to parse Venice model", + extra={ + "model_id": entry.get("id", "unknown"), + "error": str(e), + "error_type": type(e).__name__, + }, + ) + continue + if model is None: + skipped.append(str(entry.get("id", "unknown"))) + continue + models.append(model) + + if skipped: + logger.debug( + f"({len(skipped)}) Venice models skipped as unsupported or unpriced", + extra={"skipped_models": skipped}, + ) + return models + + def _parse_model(self, entry: dict[str, Any]) -> Model | None: + model_type = entry.get("type") + model_id = entry.get("id") + spec = entry.get("model_spec") + if not model_id or model_type not in _SUPPORTED_TYPES: + return None + if not isinstance(spec, dict) or spec.get("offline"): + return None + + pricing = self._parse_pricing(spec.get("pricing")) + if pricing is None: + return None + + modality, input_modalities, output_modalities = _ARCHITECTURES[str(model_type)] + capabilities = spec.get("capabilities") + if ( + model_type == "text" + and isinstance(capabilities, dict) + and capabilities.get("supportsVision") + ): + input_modalities = [*input_modalities, "image"] + modality = "text+image->text" + + context_length = spec.get("availableContextTokens") + max_completion_tokens = spec.get("maxCompletionTokens") + name = spec.get("name") or str(model_id) + + return Model( + id=str(model_id), + name=str(name), + created=int(entry.get("created") or 0), + description=str(spec.get("description") or f"Venice {model_type} model"), + context_length=int(context_length) if context_length else 0, + architecture=Architecture( + modality=modality, + input_modalities=input_modalities, + output_modalities=output_modalities, + tokenizer="Unknown", + instruct_type=None, + ), + pricing=pricing, + top_provider=TopProvider( + context_length=int(context_length) if context_length else None, + max_completion_tokens=int(max_completion_tokens) + if max_completion_tokens + else None, + ), + ) + + def _parse_pricing(self, raw: Any) -> Pricing | None: + if not isinstance(raw, dict): + return None + + # The ``extended`` tier some models charge past a context threshold is + # ignored: billing it would overcharge every request staying under it. + input_usd = _usd(raw.get("input")) + if input_usd is None: + return None + return Pricing( + prompt=input_usd / _USD_PER_MILLION, + completion=(_usd(raw.get("output")) or 0.0) / _USD_PER_MILLION, + input_cache_read=(_usd(raw.get("cache_input")) or 0.0) / _USD_PER_MILLION, + input_cache_write=(_usd(raw.get("cache_write")) or 0.0) / _USD_PER_MILLION, + ) diff --git a/tests/unit/test_upstream_venice.py b/tests/unit/test_upstream_venice.py new file mode 100644 index 00000000..1c6223a8 --- /dev/null +++ b/tests/unit/test_upstream_venice.py @@ -0,0 +1,221 @@ +"""Unit tests for ``VeniceUpstreamProvider.fetch_models``. + +Venice answers ``/models`` with only its text catalog unless ``type`` is +passed, which is why the same account configured as a generic upstream sees a +different catalog. These tests pin that query parameter, the per-token pricing +shape, and the families dropped as unpriceable. +""" + +from __future__ import annotations + +from typing import Any +from unittest.mock import patch + +import pytest + +from routstr.upstream.venice import VeniceUpstreamProvider + + +class _FakeResponse: + def __init__(self, payload: dict[str, Any]) -> None: + self._payload = payload + + def raise_for_status(self) -> None: + return None + + def json(self) -> dict[str, Any]: + return self._payload + + +class _FakeAsyncClient: + def __init__(self, payload: dict[str, Any], calls: list[dict[str, Any]]) -> None: + self._payload = payload + self._calls = calls + + async def __aenter__(self) -> "_FakeAsyncClient": + return self + + async def __aexit__(self, *_: object) -> None: + return None + + async def get( + self, + url: str, + params: dict[str, Any] | None = None, + headers: dict[str, str] | None = None, + ) -> _FakeResponse: + self._calls.append({"url": url, "params": params, "headers": headers}) + return _FakeResponse(self._payload) + + +CATALOG: dict[str, Any] = { + "object": "list", + "data": [ + { + "id": "venice-uncensored-1-2", + "type": "text", + "created": 1727966436, + "model_spec": { + "name": "Venice Uncensored 1.2", + "availableContextTokens": 128000, + "maxCompletionTokens": 8192, + "capabilities": {"supportsVision": True}, + "pricing": { + "input": {"usd": 0.2, "diem": 0.2}, + "output": {"usd": 0.9, "diem": 0.9}, + "cache_input": {"usd": 0.02, "diem": 0.02}, + "cache_write": {"usd": 0.25, "diem": 0.25}, + }, + }, + }, + { + "id": "text-embedding-bge-m3", + "type": "embedding", + "created": 1727966436, + "model_spec": { + "name": "BGE m3", + "availableContextTokens": 8192, + "pricing": {"input": {"usd": 0.01, "diem": 0.01}}, + }, + }, + { + "id": "unpriced-text", + "type": "text", + "created": 1727966436, + "model_spec": {"name": "Unpriced", "pricing": {}}, + }, + { + "id": "offline-model", + "type": "text", + "created": 1727966436, + "model_spec": { + "name": "Offline", + "offline": True, + "pricing": {"input": {"usd": 0.2, "diem": 0.2}}, + }, + }, + { + "id": "venice-sd35", + "type": "image", + "created": 1727966436, + "model_spec": { + "name": "Venice SD35", + "pricing": {"generation": {"usd": 0.01, "diem": 0.01}}, + }, + }, + { + "id": "flux-2-max-edit", + "type": "inpaint", + "created": 1727966436, + "model_spec": { + "name": "FLUX.2 Max Edit", + "pricing": {"inpaint": {"usd": 0.12, "diem": 0.12}}, + }, + }, + { + "id": "tts-kokoro", + "type": "tts", + "created": 1727966436, + "model_spec": { + "name": "Kokoro", + "pricing": {"input": {"usd": 3.5, "diem": 3.5}}, + }, + }, + { + "id": "unpriced-video", + "type": "video", + "created": 1727966436, + "model_spec": {"name": "Video"}, + }, + ], +} + + +def _fetch(payload: dict[str, Any] = CATALOG) -> tuple[list[Any], list[dict[str, Any]]]: + import asyncio + + calls: list[dict[str, Any]] = [] + provider = VeniceUpstreamProvider(api_key="sk-test") + with patch( + "routstr.upstream.venice.httpx.AsyncClient", + lambda *a, **kw: _FakeAsyncClient(payload, calls), + ): + models = asyncio.run(provider.fetch_models()) + return models, calls + + +def test_requests_every_model_family() -> None: + _, calls = _fetch() + assert calls[0]["params"] == {"type": "all"} + assert calls[0]["url"] == "https://api.venice.ai/api/v1/models" + assert calls[0]["headers"] == {"Authorization": "Bearer sk-test"} + + +def test_text_pricing_is_per_token() -> None: + models, _ = _fetch() + model = next(m for m in models if m.id == "venice-uncensored-1-2") + assert model.pricing.prompt == pytest.approx(0.2 / 1_000_000) + assert model.pricing.completion == pytest.approx(0.9 / 1_000_000) + assert model.pricing.input_cache_read == pytest.approx(0.02 / 1_000_000) + assert model.pricing.input_cache_write == pytest.approx(0.25 / 1_000_000) + assert model.context_length == 128000 + assert model.top_provider is not None + assert model.top_provider.max_completion_tokens == 8192 + assert model.architecture.input_modalities == ["text", "image"] + assert model.architecture.modality == "text+image->text" + + +def test_embedding_models_are_listed() -> None: + models, _ = _fetch() + model = next(m for m in models if m.id == "text-embedding-bge-m3") + assert model.architecture.output_modalities == ["embedding"] + assert model.pricing.prompt == pytest.approx(0.01 / 1_000_000) + assert model.pricing.completion == 0.0 + + +def test_families_billed_per_clip_are_dropped() -> None: + """Image, audio and video return no usage to settle against, so listing + them here would hand out inference this provider cannot price.""" + models, _ = _fetch() + ids = {m.id for m in models} + assert "venice-sd35" not in ids + assert "flux-2-max-edit" not in ids + assert "tts-kokoro" not in ids + assert "unpriced-video" not in ids + + +def test_offline_and_unpriced_models_are_dropped() -> None: + models, _ = _fetch() + ids = {m.id for m in models} + assert "offline-model" not in ids + assert "unpriced-text" not in ids + + +def test_model_name_drops_the_venice_prefix() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + assert provider.transform_model_name("venice/venice-uncensored-1-2") == ( + "venice-uncensored-1-2" + ) + assert provider.transform_model_name("venice-uncensored-1-2") == ( + "venice-uncensored-1-2" + ) + + +def test_provider_metadata_pins_the_base_url() -> None: + metadata = VeniceUpstreamProvider.get_provider_metadata() + assert metadata["id"] == "venice" + assert metadata["default_base_url"] == "https://api.venice.ai/api/v1" + assert metadata["fixed_base_url"] is True + + +def test_fetch_returns_empty_on_upstream_failure() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + + with patch.object( + VeniceUpstreamProvider, + "_fetch_provider_models", + side_effect=RuntimeError("boom"), + ): + import asyncio + + assert asyncio.run(provider.fetch_models()) == [] From 5d1004d3ce3cf13264993c1d830bd9014dbc9fd9 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 00:48:02 +0200 Subject: [PATCH 02/12] feat: translate anthropic web search to venice search on /v1/messages --- routstr/upstream/base.py | 11 ++ routstr/upstream/messages_dispatch.py | 10 +- routstr/upstream/venice.py | 89 +++++++++++ tests/unit/test_venice_web_search.py | 221 ++++++++++++++++++++++++++ 4 files changed, 330 insertions(+), 1 deletion(-) create mode 100644 tests/unit/test_venice_web_search.py diff --git a/routstr/upstream/base.py b/routstr/upstream/base.py index bc7244e1..3f9d7bde 100644 --- a/routstr/upstream/base.py +++ b/routstr/upstream/base.py @@ -2506,6 +2506,16 @@ class BaseUpstreamProvider: ) -> dict: return await messages_dispatch.aggregate_anthropic_events_to_message(iterator) + def adapt_messages_request(self, body: dict, model_obj: Model) -> str: + """Rewrite an allowlisted /v1/messages body for this upstream. + + Returns a suffix appended to the upstream model name, empty when the + provider needs none. Subclasses override this to express an Anthropic + feature the upstream spells differently; the base forwards the body + untouched. + """ + return "" + async def _dispatch_anthropic_messages( self, request_body: bytes | None, @@ -2520,6 +2530,7 @@ class BaseUpstreamProvider: api_key=self.api_key, provider_prefix=self.get_litellm_provider_prefix(), transform_model_name=self.transform_model_name, + adapt_request=lambda body: self.adapt_messages_request(body, model_obj), log_extra=log_extra, ) diff --git a/routstr/upstream/messages_dispatch.py b/routstr/upstream/messages_dispatch.py index 2856256c..488129a8 100644 --- a/routstr/upstream/messages_dispatch.py +++ b/routstr/upstream/messages_dispatch.py @@ -458,6 +458,7 @@ async def dispatch_anthropic_messages( api_key: str, provider_prefix: str, transform_model_name: Callable[[str], str], + adapt_request: Callable[[dict], str] | None = None, log_extra: dict[str, Any] | None = None, ) -> tuple[bool, Any, str | None]: """Call ``litellm.anthropic.messages.acreate`` and return @@ -465,6 +466,11 @@ async def dispatch_anthropic_messages( Shared by the bearer-key and x-cashu paths. Raises :class:`UpstreamError` on bad input or upstream failure. + + ``adapt_request`` is the provider's last word on the allowlisted body: it + may rewrite it in place and returns a suffix for the upstream model name, + which is how a provider expresses a feature litellm would otherwise + translate into a parameter the upstream rejects. """ if not request_body: raise UpstreamError("Missing request body for /v1/messages", status_code=400) @@ -499,13 +505,15 @@ async def dispatch_anthropic_messages( ) body = {k: v for k, v in body.items() if k in ALLOWED_MESSAGES_REQUEST_FIELDS} + model_suffix = adapt_request(body) if adapt_request else "" + # Convention: `model.id` is the canonical upstream model name; # `forwarded_model_id` is the public alias the internal API exposes # and echoes back to the client. requested_model = ( (model_obj.forwarded_model_id or model_obj.id) if model_obj else None ) - upstream_model = transform_model_name(model_obj.id) + upstream_model = f"{transform_model_name(model_obj.id)}{model_suffix}" litellm_model = f"{provider_prefix}{upstream_model}" kwargs: dict = { diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index 5949c205..30404335 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -4,6 +4,7 @@ from typing import TYPE_CHECKING, Any import httpx +from ..core.exceptions import UpstreamError from ..core.logging import get_logger from ..payment.models import Architecture, Model, Pricing, TopProvider from .base import BaseUpstreamProvider @@ -30,6 +31,37 @@ _ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = { "embedding": ("text->embedding", ["text"], ["embedding"]), } +# Venice runs search itself and reports it back through ``venice_parameters``; +# it has no Anthropic-shaped server tool and rejects the ``web_search_options`` +# that litellm's Anthropic adapter derives from one. ``auto`` matches Anthropic +# semantics, where declaring the tool leaves the decision to the model. +# Citations are asked for because litellm's Anthropic response translation +# carries no ``venice_parameters``, so inline ``[REF]n[/REF]`` markers in the +# text are the only way a caller sees which sources were used. +_WEB_SEARCH_SUFFIX = ":enable_web_search=auto&enable_web_citations=true" + +# Anthropic web-search constraints with no Venice equivalent. Honouring the +# request means enforcing them, so a request that sets one is refused rather +# than answered by a search that ignored it. +_UNENFORCEABLE_WEB_SEARCH_KEYS = frozenset( + {"max_uses", "allowed_domains", "blocked_domains", "user_location"} +) + + +def _is_web_search_tool(tool: Any) -> bool: + """An Anthropic server-side web-search tool, by either of its markers. + + Matches litellm's own detection (``litellm/llms/anthropic/ + experimental_pass_through/adapters/transformation.py``), so every tool it + would turn into ``web_search_options`` is caught here first. + """ + if not isinstance(tool, dict): + return False + tool_type = tool.get("type") + return ( + isinstance(tool_type, str) and tool_type.startswith("web_search") + ) or tool.get("name") == "web_search" + def _usd(entry: Any) -> float | None: """Read the USD leg of a Venice ``{usd, diem}`` price pair.""" @@ -79,6 +111,63 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): def transform_model_name(self, model_id: str) -> str: return model_id.removeprefix("venice/") + def adapt_messages_request(self, body: dict, model_obj: Model) -> str: + """Trade an Anthropic web-search tool for Venice's own search switch. + + Left in the body, litellm's Anthropic adapter rewrites the tool into a + top-level ``web_search_options``, which Venice answers with a 400. The + tool is lifted out here and the same intent re-expressed as a model + feature suffix, the one form of ``venice_parameters`` that survives + that adapter. + """ + tools = body.get("tools") + if not isinstance(tools, list): + return "" + search_tools = [tool for tool in tools if _is_web_search_tool(tool)] + if not search_tools: + return "" + + # A key carrying null or an empty list states no constraint, so it is + # read as absent rather than refused. + unenforceable = sorted( + { + key + for tool in search_tools + for key, value in tool.items() + if key in _UNENFORCEABLE_WEB_SEARCH_KEYS + and value is not None + and value != [] + } + ) + if unenforceable: + raise UpstreamError( + "Venice web search cannot honour these Anthropic web_search " + f"options: {', '.join(unenforceable)}", + status_code=400, + code="UNSUPPORTED_WEB_SEARCH_OPTION", + details={"unsupported_options": unenforceable}, + ) + + tool_choice = body.get("tool_choice") + if isinstance(tool_choice, dict) and tool_choice.get("name") == "web_search": + raise UpstreamError( + "Venice web search cannot be forced through tool_choice; it is " + "decided by the model", + status_code=400, + code="UNSUPPORTED_WEB_SEARCH_OPTION", + details={"unsupported_options": ["tool_choice"]}, + ) + + remaining = [tool for tool in tools if not _is_web_search_tool(tool)] + if remaining: + body["tools"] = remaining + else: + body.pop("tools", None) + # tool_choice without tools is rejected by OpenAI-shaped upstreams. + body.pop("tool_choice", None) + + return _WEB_SEARCH_SUFFIX + async def _fetch_provider_models(self) -> dict: url = f"{self.base_url.rstrip('/')}/models" headers = {"Authorization": f"Bearer {self.api_key}"} if self.api_key else None diff --git a/tests/unit/test_venice_web_search.py b/tests/unit/test_venice_web_search.py new file mode 100644 index 00000000..df2ac903 --- /dev/null +++ b/tests/unit/test_venice_web_search.py @@ -0,0 +1,221 @@ +"""Venice web search over ``/v1/messages``. + +litellm's Anthropic adapter rewrites an Anthropic server-side web-search tool +into a top-level ``web_search_options``, which Venice rejects with +``400 Unrecognized key(s) in object: 'web_search_options'``. These tests pin +the trade: the tool is lifted out of the body and the same intent re-expressed +as a Venice model feature suffix. +""" + +from __future__ import annotations + +import json +from typing import Any, AsyncIterator +from unittest.mock import AsyncMock, patch + +import pytest + +from routstr.core.exceptions import UpstreamError +from routstr.payment.models import Architecture, Model, Pricing +from routstr.upstream.base import BaseUpstreamProvider +from routstr.upstream.venice import VeniceUpstreamProvider + +WEB_SEARCH_TOOL = {"type": "web_search_20250305", "name": "web_search"} +FUNCTION_TOOL = { + "name": "lookup", + "description": "Look something up", + "input_schema": {"type": "object", "properties": {}}, +} + + +def _model(model_id: str = "deepseek-v4-flash-0731") -> Model: + return Model( + id=model_id, + name=model_id, + created=0, + description="", + context_length=8192, + architecture=Architecture( + modality="text->text", + input_modalities=["text"], + output_modalities=["text"], + tokenizer="Unknown", + instruct_type=None, + ), + pricing=Pricing(prompt=0.0, completion=0.0), + ) + + +def _body(**extra: Any) -> dict[str, Any]: + return { + "messages": [{"role": "user", "content": "what shipped today?"}], + "max_tokens": 64, + **extra, + } + + +async def _dispatch(provider: BaseUpstreamProvider, body: dict[str, Any]) -> dict: + """Run the real dispatcher, capturing the kwargs litellm would receive.""" + captured: dict[str, Any] = {} + + async def empty_iter() -> AsyncIterator[dict]: + if False: + yield {} + + async def fake_acreate(**kwargs: Any) -> AsyncIterator[dict]: + captured.update(kwargs) + return empty_iter() + + with patch( + "litellm.anthropic.messages.acreate", + new=AsyncMock(side_effect=fake_acreate), + ): + await provider._dispatch_anthropic_messages( + request_body=json.dumps( + {"model": "venice/x", "stream": True, **body} + ).encode(), + model_obj=_model(), + ) + return captured + + +@pytest.mark.asyncio +async def test_web_search_tool_never_reaches_venice_as_web_search_options() -> None: + """The reported 400: the derived parameter must not be sent at all.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + + kwargs = await _dispatch(provider, _body(tools=[WEB_SEARCH_TOOL])) + + assert "web_search_options" not in kwargs + assert "tools" not in kwargs + assert kwargs["model"] == ( + "openai/deepseek-v4-flash-0731:enable_web_search=auto&enable_web_citations=true" + ) + assert kwargs["api_base"] == "https://api.venice.ai/api/v1" + + +@pytest.mark.asyncio +async def test_function_tools_survive_alongside_web_search() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + + kwargs = await _dispatch( + provider, + _body( + tools=[WEB_SEARCH_TOOL, FUNCTION_TOOL], + tool_choice={"type": "tool", "name": "lookup"}, + ), + ) + + assert kwargs["tools"] == [FUNCTION_TOOL] + assert kwargs["tool_choice"] == {"type": "tool", "name": "lookup"} + assert "web_search_options" not in kwargs + assert kwargs["model"].endswith(":enable_web_search=auto&enable_web_citations=true") + + +@pytest.mark.asyncio +async def test_requests_without_web_search_are_untouched() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + + kwargs = await _dispatch(provider, _body(tools=[FUNCTION_TOOL])) + + assert kwargs["model"] == "openai/deepseek-v4-flash-0731" + assert kwargs["tools"] == [FUNCTION_TOOL] + + +@pytest.mark.asyncio +async def test_other_providers_keep_their_existing_behaviour() -> None: + """The base hook is a no-op, so no non-Venice upstream changes shape.""" + provider = BaseUpstreamProvider(base_url="http://test", api_key="k") + + kwargs = await _dispatch(provider, _body(tools=[WEB_SEARCH_TOOL])) + + assert kwargs["model"] == "openai/deepseek-v4-flash-0731" + assert kwargs["tools"] == [WEB_SEARCH_TOOL] + + +@pytest.mark.parametrize( + "tool", + [ + {"type": "web_search_20250305", "name": "web_search", "max_uses": 5}, + { + "type": "web_search_20250305", + "name": "web_search", + "allowed_domains": ["example.com"], + }, + {"type": "web_search_20250305", "name": "web_search", "blocked_domains": ["x"]}, + { + "type": "web_search_20250305", + "name": "web_search", + "user_location": {"type": "approximate", "country": "DE"}, + }, + ], +) +def test_constraints_venice_cannot_enforce_are_refused(tool: dict[str, Any]) -> None: + """Better an explicit 400 than a search that quietly ignored the limit.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + + with pytest.raises(UpstreamError) as excinfo: + provider.adapt_messages_request(_body(tools=[tool]), _model()) + + assert excinfo.value.status_code == 400 + assert excinfo.value.code == "UNSUPPORTED_WEB_SEARCH_OPTION" + + +@pytest.mark.parametrize( + "tool", + [ + {"type": "web_search_20250305", "name": "web_search", "max_uses": None}, + {"type": "web_search_20250305", "name": "web_search", "allowed_domains": []}, + ], +) +def test_constraint_keys_stating_nothing_are_read_as_absent( + tool: dict[str, Any], +) -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + + assert provider.adapt_messages_request(_body(tools=[tool]), _model()) != "" + + +def test_forcing_web_search_through_tool_choice_is_refused() -> None: + provider = VeniceUpstreamProvider(api_key="sk-test") + body = _body( + tools=[WEB_SEARCH_TOOL], + tool_choice={"type": "tool", "name": "web_search"}, + ) + + with pytest.raises(UpstreamError) as excinfo: + provider.adapt_messages_request(body, _model()) + + assert excinfo.value.status_code == 400 + + +def test_tool_named_web_search_without_the_type_marker_is_caught() -> None: + """litellm matches on either marker, so this one would also be rewritten.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + body = _body(tools=[{"name": "web_search"}]) + + assert provider.adapt_messages_request(body, _model()) != "" + assert "tools" not in body + + +def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() -> None: + """The fix at its cause: run the real litellm translation over the body + this provider produces and assert the rejected key is never derived.""" + from litellm.llms.anthropic.experimental_pass_through.adapters.transformation import ( # noqa: E501 + LiteLLMAnthropicMessagesAdapter, + ) + + provider = VeniceUpstreamProvider(api_key="sk-test") + adapter = LiteLLMAnthropicMessagesAdapter() + body = _body(tools=[WEB_SEARCH_TOOL, FUNCTION_TOOL]) + + # Unadapted, litellm derives the parameter Venice rejects. + before, _ = adapter.translate_anthropic_to_openai( + {"model": "m", **_body(tools=[WEB_SEARCH_TOOL])} + ) + assert "web_search_options" in before + + provider.adapt_messages_request(body, _model()) + after, _ = adapter.translate_anthropic_to_openai({"model": "m", **body}) + + assert "web_search_options" not in after From aedbd6036942bc666b2a1e12a53f6116130c8a88 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 00:55:20 +0200 Subject: [PATCH 03/12] test: pin the wire shape of a venice web-search request --- .../test_venice_web_search_wire.py | 131 ++++++++++++++++++ 1 file changed, 131 insertions(+) create mode 100644 tests/integration/test_venice_web_search_wire.py diff --git a/tests/integration/test_venice_web_search_wire.py b/tests/integration/test_venice_web_search_wire.py new file mode 100644 index 00000000..2636af1f --- /dev/null +++ b/tests/integration/test_venice_web_search_wire.py @@ -0,0 +1,131 @@ +"""What Routstr actually puts on the wire for a Venice web-search request. + +The unit tests stop at the kwargs handed to litellm. Everything that produced +the reported ``400 Unrecognized key(s) in object: 'web_search_options'`` +happened *after* that point, inside litellm's Anthropic adapter, so this test +runs the whole dispatch against a loopback OpenAI-compatible server and reads +the bytes Venice would have received. +""" + +from __future__ import annotations + +import json +import threading +from http.server import BaseHTTPRequestHandler, HTTPServer +from typing import Any, Iterator + +import pytest + +from routstr.payment.models import Architecture, Model, Pricing +from routstr.upstream.litellm_routing import configure_litellm +from routstr.upstream.venice import VeniceUpstreamProvider + +_CHUNKS = [ + { + "id": "chatcmpl-1", + "object": "chat.completion.chunk", + "created": 0, + "model": "deepseek-v4-flash-0731", + "choices": [{"index": 0, "delta": {"role": "assistant", "content": "ok"}}], + }, + { + "id": "chatcmpl-1", + "object": "chat.completion.chunk", + "created": 0, + "model": "deepseek-v4-flash-0731", + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 5, "completion_tokens": 2, "total_tokens": 7}, + }, +] + + +@pytest.fixture +def upstream() -> Iterator[tuple[str, dict[str, Any]]]: + """A loopback stand-in for ``api.venice.ai`` that records one request.""" + captured: dict[str, Any] = {} + + class Handler(BaseHTTPRequestHandler): + def do_POST(self) -> None: # noqa: N802 - http.server's spelling + length = int(self.headers.get("Content-Length", 0)) + captured["path"] = self.path + captured["body"] = json.loads(self.rfile.read(length)) + + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.end_headers() + for chunk in _CHUNKS: + self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode()) + self.wfile.write(b"data: [DONE]\n\n") + + def log_message(self, *args: Any) -> None: + return None + + server = HTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + yield f"http://127.0.0.1:{server.server_address[1]}/v1", captured + finally: + server.shutdown() + thread.join(timeout=5) + + +def _model() -> Model: + return Model( + id="deepseek-v4-flash-0731", + name="deepseek-v4-flash-0731", + created=0, + description="", + context_length=8192, + architecture=Architecture( + modality="text->text", + input_modalities=["text"], + output_modalities=["text"], + tokenizer="Unknown", + instruct_type=None, + ), + pricing=Pricing(prompt=0.0, completion=0.0), + ) + + +@pytest.mark.asyncio +async def test_web_search_request_reaches_venice_in_its_own_shape( + upstream: tuple[str, dict[str, Any]], +) -> None: + base_url, captured = upstream + # The app applies this at startup; without it litellm posts the Anthropic + # body to /responses, which Venice serves only in alpha. + configure_litellm() + + provider = VeniceUpstreamProvider(api_key="sk-test") + provider.base_url = base_url + + await provider._dispatch_anthropic_messages( + request_body=json.dumps( + { + "model": "venice/deepseek-v4-flash-0731", + "messages": [{"role": "user", "content": "what shipped today?"}], + "max_tokens": 64, + "stream": True, + "tools": [ + {"type": "web_search_20250305", "name": "web_search"}, + { + "name": "lookup", + "description": "Look something up", + "input_schema": {"type": "object", "properties": {}}, + }, + ], + } + ).encode(), + model_obj=_model(), + ) + + body = captured["body"] + assert captured["path"] == "/v1/chat/completions" + # The reported 400, at the only place it could be observed. + assert "web_search_options" not in body + assert body["model"] == ( + "deepseek-v4-flash-0731:enable_web_search=auto&enable_web_citations=true" + ) + # The function tool still travels, in OpenAI's shape. + assert [tool["function"]["name"] for tool in body["tools"]] == ["lookup"] From d5aa7d26d4bdd6cb76d88d06f6fcdff0bc7894c3 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 01:08:30 +0200 Subject: [PATCH 04/12] docs: correct venice citation marker format in web-search comment --- routstr/upstream/venice.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index 30404335..57e96b24 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -36,8 +36,8 @@ _ARCHITECTURES: dict[str, tuple[str, list[str], list[str]]] = { # that litellm's Anthropic adapter derives from one. ``auto`` matches Anthropic # semantics, where declaring the tool leaves the decision to the model. # Citations are asked for because litellm's Anthropic response translation -# carries no ``venice_parameters``, so inline ``[REF]n[/REF]`` markers in the -# text are the only way a caller sees which sources were used. +# carries no ``venice_parameters``, so the inline ``^n^`` markers Venice writes +# into the text are the only way a caller sees that sources were used. _WEB_SEARCH_SUFFIX = ":enable_web_search=auto&enable_web_citations=true" # Anthropic web-search constraints with no Venice equivalent. Honouring the From 04becf4607d5792b70c5d3c64271593cb412cffa Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 01:44:58 +0200 Subject: [PATCH 05/12] docs: document venice web search investigation and implementation --- VENICE_WEB_SEARCH.md | 151 +++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 151 insertions(+) create mode 100644 VENICE_WEB_SEARCH.md diff --git a/VENICE_WEB_SEARCH.md b/VENICE_WEB_SEARCH.md new file mode 100644 index 00000000..18abcc07 --- /dev/null +++ b/VENICE_WEB_SEARCH.md @@ -0,0 +1,151 @@ +# Venice web search through Routstr `/v1/messages` + +Status: **implemented on branch `feat/venice-provider`** and **verified against live Venice** (2026-09-25, see "Live verification"). Investigated 2026-09-24, built 2026-09-25. The production request body and a live Venice credential were unavailable; distinguish reproduced local behavior from the inferred production trigger below. + +## What shipped + +Two commits on `feat/venice-provider` (branched from `main`): + +- `f77896f1` `feat: add venice upstream provider` — `VeniceUpstreamProvider` ported text-and-embedding only from `feat/venice-provider-image-pricing`. Image, inpaint and upscale families are dropped rather than listed, because their price book lives in the image-billing commit that did not come along; listing them here would hand out unpriced inference. +- `5d1004d3` `feat: translate anthropic web search to venice search on /v1/messages` — the fix below. + +**The seam.** `BaseUpstreamProvider.adapt_messages_request(body, model_obj) -> str` is a provider's last word on an allowlisted Anthropic body: it may rewrite the body in place and returns a suffix for the upstream model name. `dispatch_anthropic_messages` calls it after the `ALLOWED_MESSAGES_REQUEST_FIELDS` filter and appends the suffix to `transform_model_name(model.id)`. The base implementation returns `""`, so no other provider changes shape. + +**The Venice override.** Any tool litellm would read as web search — `type` starting `web_search`, or `name == "web_search"`, the same two markers its adapter matches — is lifted out of `tools`, and the intent is re-expressed as the model feature suffix `:enable_web_search=auto&enable_web_citations=true`. `auto` matches Anthropic semantics, where declaring the tool leaves the decision to the model. Citations are requested because litellm's Anthropic response translation carries no `venice_parameters`, so inline `[REF]n[/REF]` markers are the only surviving signal of which sources were used. Remaining function tools and their `tool_choice` travel untouched; when the search tool was the only one, `tool_choice` is dropped with it, since an OpenAI-shaped upstream rejects a choice with no tools. + +**Refusals.** `max_uses`, `allowed_domains`, `blocked_domains` and `user_location` have no Venice equivalent, and a `tool_choice` naming `web_search` cannot be honoured because Venice's search is not a callable tool. Each returns 400 `UNSUPPORTED_WEB_SEARCH_OPTION` before the upstream call rather than a search that quietly ignored the constraint. A key carrying `null` or `[]` states no constraint and is read as absent. + +**Verification.** `tests/unit/test_venice_web_search.py` (13 tests) covers the adaptation, the refusals, and — running the real litellm adapter — asserts the unadapted body derives `web_search_options` while the adapted one does not. `tests/integration/test_venice_web_search_wire.py` runs the whole dispatch against a loopback OpenAI-compatible server and reads the bytes Venice would receive: `POST /v1/chat/completions`, no `web_search_options`, `model` carrying the suffix, the function tool in OpenAI shape. Full unit suite 1722 passed; ruff and mypy clean. + +That wire test also pins a dependency on startup config: without `configure_litellm()` (applied in `routstr/core/main.py`), litellm posts the Anthropic body to `/responses`, which Venice serves only in alpha. + +## Live verification + +Run 2026-09-25 against `api.venice.ai` with a real key. Each open question from the plan is now answered by observation rather than inference. + +**The suffix is honoured.** A `/v1/messages` request carrying an Anthropic `web_search_20250305` tool came back with a live figure and its source ("approximately $84,216.93 USD, according to CoinMarketCap.^6^") on `deepseek-v4-flash-0731`, and the same on `zai-org-glm-5-1` over a real stream. No 400. The control request without the tool searched nothing. + +**Streaming is intact.** The stream yields the full Anthropic event set — `message_start`, `content_block_start`, `content_block_delta`, `content_block_stop`, `message_delta`, `message_stop` — with usage on the final events. + +**Citations arrive as `^n^`, not `[REF]n[/REF]`.** The API reference describes the latter; live responses write superscript markers, matching Venice's own agent skill. Structured citations are confirmed lost: the Anthropic-shaped response carries only `content`, `id`, `model`, `role`, `stop_reason`, `stop_sequence`, `type`, `usage`, with no `venice_parameters`. The inline markers are the whole signal. + +**The capability gate is unnecessary.** All 123 text models in the live catalog report `supportsWebSearch: true` — none false, none missing the key. There is no Venice text model to refuse, so the `Model` field, `ModelRow` column and migration the plan called for are not worth building. Revisit only if Venice ships a text model without it. + +**`Pricing.web_search = 0.0` is right.** Venice bills search through the prompt: the same question cost 5,839 input tokens with search against 1,710 without, because the results are injected into the context. There is no separate per-search fee to price (Venice documents one only for `enable_x_search`, which this path never enables). Those tokens are billed by the existing per-token path, and `_calculate_usd_max_costs` reserves against the full context window, so an inflated prompt stays inside the reservation. + +Still unobserved: behaviour when Venice's search itself fails or returns nothing, and `enable_web_scraping`, which this path never turns on. + +## Incident and conclusion + +Routstr 0.4.7 logged a `/v1/messages` request (`09adf07c-1456-4ccb-8276-824016392219`) dispatched to `https://api.venice.ai/api/v1` with LiteLLM model `openai/deepseek-v4-flash-0731`. Venice returned HTTP 400: `Unrecognized key(s) in object: 'web_search_options'`. The proxy then logged `provider=generic`, `status_code=400`, `retry=true`. + +These labels describe different layers. `generic` is Routstr's provider row; `openai/` selects LiteLLM's OpenAI-compatible Chat Completions adapter, not the destination service. `api_base` still points to Venice. The installed LiteLLM does not recognize `venice/` as a provider prefix, so simply renaming it breaks routing. LiteLLM removes `openai/` when resolving the provider; the *outbound* model ID should be the bare Venice ID. Capture one sanitized outbound request to verify the wire payload rather than relying on the dispatch log. + +The 400 is about the **unsupported field**, not the prefix. Routstr allowlists `tools` but does not forward client-supplied `web_search_options`. The installed LiteLLM 1.93.2 Anthropic Messages adapter recognizes a tool whose `type` starts with `web_search` or whose `name` is `web_search`, removes it from ordinary function tools, and inserts `web_search_options: {}` into the OpenAI-shaped call. A local, credential-free repro with `web_search_20250305` produced that exact field; an ordinary function tool did not. The production log has no input `tools` field, so the specific incoming trigger remains **strongly indicated, not proved**. A sanitized copy of the incoming `tools` types/names would settle it. + +The shared `litellm.drop_params=True` setting is not sufficient to protect arbitrary OpenAI-compatible servers: the adapter creates this field *after* Routstr filters the incoming body. Similarly, the proxy's `correct_request` retries on client request fields, not on this post-translation field. Its `retry=true` means another candidate provider may be attempted for a 400, not that the same Venice request becomes valid. + +## Venice's actual search interfaces + +Venice documents **model-integrated web search** for `POST /chat/completions` using `venice_parameters.enable_web_search` (`"off"`, `"auto"`, `"on"`; default `"off"`). `"on"` forces search; `"auto"` leaves it to the model. `venice_parameters.enable_web_citations: true` asks for inline source references. The response may include `venice_parameters.web_search_citations`; citations arrive in the first streaming chunk or the non-streaming response. The model feature suffix is another documented way to set these without an extra request field: + +```text +:enable_web_search=auto&enable_web_citations=true +``` + +For standalone retrieval, Venice also has `POST /augment/search` and `/augment/scrape`, but that is a different architecture: Routstr would have to execute search, supply results to the model, handle citations and account for the extra call. Venice model metadata advertises `model_spec.capabilities.supportsWebSearch` for model-specific support; verify the actual configured model at runtime rather than assuming all Venice models support it. The incident alone does **not** prove `deepseek-v4-flash-0731` advertises this capability. + +Important documentation discrepancy: Venice's first-party `venice-chat` skill describes `tools: [{"type":"web_search"}]` as a built-in toggle, while the official Chat Completions OpenAPI schema currently says only function tools are supported. Treat the `venice_parameters`/suffix route as the documented baseline; test built-in `tools` on the live API before depending on it. Neither source documents accepting the top-level `web_search_options` field rejected in this incident. + +## Routstr implementation plan + +1. **Write a red regression at the actual seam.** Extend `tests/unit/test_messages_litellm_dispatch.py` with a generic provider pointing at Venice and an Anthropic `/v1/messages` request containing a server-side `web_search_20250305` tool. Exercise `BaseUpstreamProvider._dispatch_anthropic_messages` through `messages_dispatch.dispatch_anthropic_messages`. Use the real LiteLLM translation in a local, network-free adapter assertion, not only an `acreate` mock: assert that the current path produces `web_search_options` and that the proposed path does not. Cover both bearer-key and x-cashu callers because both use the same dispatcher. +2. **Add a narrowly scoped Venice capability branch** before `litellm.anthropic.messages.acreate` in `routstr/upstream/messages_dispatch.py`, with the provider identity supplied by `BaseUpstreamProvider` (or an explicit provider capability). Match the parsed Venice hostname exactly, not an unbounded substring or a model name; generic non-Venice hosts must remain unchanged. Keep `openai/` as the LiteLLM adapter prefix and `api_base` as Venice. Do not rewrite the public `Model.id` or `forwarded_model_id`. +3. **Translate intent, not merely delete it.** On a Venice route, remove only Anthropic *server-side web-search* tools from the `tools` sent into LiteLLM so its adapter cannot synthesize `web_search_options`. Preserve ordinary function tools and their `tool_choice`. Enable Venice search for this request with the documented suffix on the **upstream** model ID, e.g. `:enable_web_search=auto`, optionally adding `&enable_web_citations=true` if the response path preserves citations. This avoids relying on unknown `extra_body` behavior in LiteLLM's Anthropic adapter. Alternatively, pass `venice_parameters` only after a wire-level test demonstrates it survives that adapter. Never silently remove a requested search tool without enabling an equivalent service. +4. **Make unsupported semantics explicit.** Decide and test how to handle `max_uses`, `allowed_domains`/`blocked_domains`, forced `tool_choice` targeting web search, duplicate search tools, or a model without `supportsWebSearch`: Venice's search switch is not a one-to-one implementation of every Anthropic tool constraint. Where equivalence cannot be guaranteed, return a clear pre-dispatch 4xx or explicitly documented degraded behavior; avoid a success that pretends the requested constraints were enforced. Do not translate client-supplied arbitrary `venice_parameters` through the `/v1/messages` allowlist. +5. **Preserve the API contract.** Test streamed and non-streamed Anthropic-shaped responses, function tools coexisting with search, no-search Venice requests, non-Venice OpenAI-compatible requests, and handling of `venice_parameters.web_search_citations`. The existing LiteLLM → Anthropic response conversion may drop Venice-specific citation metadata; verify it with captured fixtures before promising search citations. If metadata is lost, either map it deliberately to the chosen client-visible format or document that search works without structured citations. +6. **Protect billing and routing.** `GenericUpstreamProvider.fetch_models` currently sets `Pricing.web_search=0.0`; check Venice's live web-search charges and returned usage/cost fields. Ensure reservation/max-cost estimation and final charge include any search fees before enabling paid searches, or fail closed if they cannot be priced. The 400 fallback behavior in `routstr/proxy.py` must not route a search-required request to a provider that silently loses search; inspect candidate capabilities and keep payment reversal correct. Keep the suffix out of catalog IDs, public response model IDs, and price lookups. +7. **Verify live with a Venice test key** after network-free tests: record sanitized outbound JSON and check absence of `web_search_options`, bare upstream model name plus the Venice suffix (if chosen), successful web-enabled reply, citations/usage shape, and billing reconciliation for `stream=true` and `false`. Check model capability from `/models` first. No live request was sent in this investigation. + +Acceptance: web-search requests on a Venice model that supports search either complete with search enabled and correctly billed, or fail before the upstream call with a specific unsupported-capability error; no request emits `web_search_options` toward Venice. Requests without search and other providers retain their existing behavior. No unsupported search constraints are silently accepted. + +## Related reports and prior art + +- [LiteLLM #10714](https://github.com/BerriAI/litellm/issues/10714) and its referenced [#10664](https://github.com/BerriAI/litellm/issues/10664) concern Anthropic `web_search_20250305` support in LiteLLM; these are historical context for adapter differences, **not** a verified patch for this Venice 400. +- [LiteLLM #14250](https://github.com/BerriAI/litellm/issues/14250) documents that even OpenAI Chat Completions web search via `web_search_options` is model-specific; an OpenAI-compatible endpoint need not implement it. +- [LiteLLM web-search interception integration](https://docs.litellm.ai/docs/web_search_interception) is an alternative architecture with an external search provider and an agentic follow-up, not a drop-in change to Routstr's current direct `litellm.anthropic.messages.acreate` path. A [follow-up duplicate-kwargs report](https://github.com/BerriAI/litellm/issues) was found in the broader search but not established as this issue's cause; do not infer a fix from it. +- First-party [Venice Chat skill](https://github.com/veniceai/skills/blob/main/skills/venice-chat/SKILL.md) gives provider-native search examples. The official API reference below takes precedence for the implementable request shape. Searches for an exact public Venice + LiteLLM `web_search_options` 400 fix did **not** yield a verified matching issue or merged patch. Do not claim an upstream fix exists without reproducing it in the pinned version. + +## Branch `feat/venice-provider-image-pricing` — a Venice provider class already exists + +Checked 2026-09-24 on that branch (two commits ahead of `main`, no PR open). `routstr/upstream/venice.py` adds `VeniceUpstreamProvider(BaseUpstreamProvider)` with `provider_type = "venice"`, a pinned `default_base_url = "https://api.venice.ai/api/v1"` (`fixed_base_url: True`), a catalog fetch across Venice's model families, text and per-image-tier pricing, and `transform_model_name` stripping a `venice/` prefix. Tests in `tests/unit/test_upstream_venice.py` are catalog and pricing only. + +It does **not** fix this incident. Verified at runtime on the branch: `VeniceUpstreamProvider.litellm_provider_prefix` is `None`, so `get_litellm_provider_prefix()` still resolves to `openai/` through `detect_litellm_prefix`, and `supports_anthropic_messages` is `False`, so `/v1/messages` still goes through `messages_dispatch` into LiteLLM's Anthropic adapter — the same code that synthesizes `web_search_options`. The file contains no web-search or `venice_parameters` handling. + +What it does change is **where the fix belongs**. With this class merged, step 2 of the plan above needs no hostname matching: provider identity is the class itself, so the Venice branch becomes a method on `VeniceUpstreamProvider` rather than a URL test inside the shared dispatcher. Adopt it and revise the plan as follows: + +- Put the translation on the provider, e.g. an override of `_dispatch_anthropic_messages` (or a narrow hook the base dispatcher calls) that strips Anthropic server-side web-search tools and enables Venice search. Keep `openai/` as the LiteLLM adapter prefix. +- Do **not** append the `:enable_web_search=…` suffix inside `transform_model_name`. `base.py` calls it on the chat/completions and model-listing paths too (around lines 705, 724, 795), so a suffix there would leak into unrelated requests. Scope it to the messages dispatch call. +- The incident ran on a `generic` row, not this class. Using it means re-creating the Venice upstream row as `provider_type="venice"`; `_build_from_row` takes only `api_key` and `provider_fee` because the base URL is pinned. A stale `generic` row keeps the old behavior. +- `_parse_pricing` returns text `Pricing` without a `web_search` rate (defaults to `0.0`), so per-search charges are still unpriced — the billing item in step 6 stands unchanged. + +The branch is unreviewed and carries an unrelated image-generation billing commit (27 files, ~3.9k insertions). Landing the web-search work on top of it couples this fix to that review. Decide explicitly: build on the branch, or implement against `main` and rebase once the provider lands. + +## Adding Venice support to LiteLLM + +Investigated 2026-09-24 against installed LiteLLM 1.93.2 and upstream `main` (published 1.102.1). + +### What already exists + +Venice is **already registered** in LiteLLM, but only as a bare JSON entry. `litellm/llms/openai_like/providers.json` contains, on both the pinned version and upstream `main`: + +```json +"veniceai": { + "base_url": "https://api.venice.ai/api/v1", + "api_key_env": "VENICE_AI_API_KEY" +} +``` + +Verified locally: `litellm.get_llm_provider("veniceai/deepseek-v4-flash-0731")` resolves to `("deepseek-v4-flash-0731", "veniceai")`, while `venice/...` raises `LLM Provider NOT provided`. `veniceai` is **not** in `litellm.provider_list` or the `LlmProviders` enum — it resolves through `JSONProviderRegistry`, which `get_llm_provider_logic.py` checks before the enum. Upstream `main` has no `litellm/llms/venice*` directory, no Venice entries in `model_prices_and_context_window.json`, and `docs.litellm.ai/docs/providers/venice` returns 404. The `venice` block in the installed `provider_endpoints_support_backup.json` describes a provider that was never merged. + +### The JSON entry does not fix this incident + +JSON providers inherit `OpenAIGPTConfig`, whose supported-parameter list includes `web_search_options`. Verified locally with the generated config class: `get_supported_openai_params` returns 26 params including `web_search_options`, and `map_openai_params({"web_search_options": {}}, drop_params=True)` keeps the field. So `litellm.drop_params` will not remove it, and switching Routstr's prefix from `openai/` to `veniceai/` still emits the field Venice rejects. `get_optional_params(..., custom_llm_provider="veniceai", extra_body={"venice_parameters": {...}})` does keep `extra_body` alongside `web_search_options`; whether that survives the Anthropic-messages adapter to the wire is **untested**, as no live request was made. + +### Prior attempts and maintainer stance + +- [#17962](https://github.com/BerriAI/litellm/pull/17962) **merged** — the two-line `providers.json` entry above, one file, no tests. +- [#17948](https://github.com/BerriAI/litellm/pull/17948) **closed unmerged** — a full `VeniceAIChatConfig(OpenAILikeChatConfig)` with a `VENICE_PARAMS` set (`enable_web_search`, `enable_web_citations`, `character_slug`, …) nested into `venice_parameters` by `transform_request`, plus enum, URL detection, docs, and 428 lines of tests. A maintainer replied that provider-specific params already pass through automatically and pointed at the providers.json path; the author closed it in favor of #17962. +- [#18248](https://github.com/BerriAI/litellm/pull/18248) **closed** (stale) — wired `veniceai` into `constants.py`, `types/utils.py`, URL detection, `provider_endpoints_support.json`, and docs. +- [#26970](https://github.com/BerriAI/litellm/pull/26970) (Venice model prices, fixes [#24229](https://github.com/BerriAI/litellm/issues/24229)) and [#23670](https://github.com/BerriAI/litellm/pull/23670) (docs) both **closed unmerged**. +- Feature requests [#8833](https://github.com/BerriAI/litellm/issues/8833) and [#9093](https://github.com/BerriAI/litellm/issues/9093) are closed. + +Treat that history as the main risk: the nesting problem this project needs was proposed once and rejected as unnecessary. A new PR must argue what `providers.json` cannot express, rather than restating the request. + +### Option A — extend the JSON provider system (recommended upstream path) + +`param_mappings` only renames a key; it cannot nest `enable_web_search` under `venice_parameters`, and nothing in the schema can mark an inherited param unsupported. Two small additive fields in `dynamic_config.py` close both gaps generically, for every OpenAI-compatible provider that rejects inherited OpenAI extras: + +- `unsupported_params: ["web_search_options"]` — removed from `get_supported_openai_params`, so `drop_params` handles it through the existing path. +- `nest_params_under: "venice_parameters"` with the member list — `map_openai_params`/`transform_request` build the nested object. + +Scope: `llms/openai_like/dynamic_config.py`, `providers.json`, `llms/openai_like/README.md`, plus tests under `tests/test_litellm/`. This stays inside the system the maintainer endorsed and benefits other providers, which is the strongest available argument for merge. + +### Option B — first-class Python provider + +Revive the #17948 + #18248 shape: `litellm/llms/venice_ai/chat/transformation.py`, `LlmProviders.VENICE_AI` in `types/utils.py`, `constants.py` provider list, `api.venice.ai` detection in `get_llm_provider_logic.py`, `__init__.py`/`utils.py` wiring, `ProviderConfigManager` registration, `model_prices_and_context_window.json` (+ backup) from Venice `/models`, `provider_endpoints_support.json`, `docs/my-website/docs/providers/venice.md` + `sidebars.js`, and tests under `tests/test_litellm/llms/venice_ai/`. Contributing requires a signed CLA, at least one test, and a Greptile review request. Only this option can also map an Anthropic `web_search_*` tool to `enable_web_search` inside LiteLLM, and only for callers that reach the chat path with that tool intact. + +Both options are upstream work on a third-party project with an uncertain merge outcome and a release lag. Neither removes the need for the Routstr-side plan above, which is the only change that fixes the incident on the pinned 1.93.2. + +### If Routstr adopts `veniceai/` later + +`detect_litellm_prefix` in `routstr/upstream/litellm_routing.py` would map `api.venice.ai` to `veniceai/`. Gate that on the installed LiteLLM version: the prefix resolves only while the JSON entry exists, it is absent from `litellm.provider_list`, and no Venice model carries LiteLLM pricing, so Routstr's own pricing path stays authoritative. On its own, the prefix change does not stop `web_search_options`. + +## Primary sources and local evidence + +- [Venice Chat Completions API](https://docs.venice.ai/api-reference/endpoint/chat/completions) — `venice_parameters`, search modes, response citations, strict request schema. +- [Venice Model Feature Suffix](https://docs.venice.ai/api-reference/endpoint/chat/model_feature_suffix) — `:=` and combined suffixes. +- [Venice Web Search API](https://docs.venice.ai/api-reference/endpoint/augment/search), [Web Search and Scraping guide](https://docs.venice.ai/guides/tools/web-retrieval), [Venice model catalog](https://docs.venice.ai/api-reference/endpoint/models/list). +- Local: `routstr/upstream/litellm_routing.py:24-116`, `routstr/upstream/base.py:361-372,2493-2508`, `routstr/upstream/messages_dispatch.py:59-78,479-531`, `routstr/upstream/generic.py:92-136,209-235`, `routstr/proxy.py:857-923`, `tests/unit/test_messages_litellm_dispatch.py`, `uv.lock` (LiteLLM 1.93.2). +- Installed dependency: `.venv/lib/python3.14/site-packages/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py:335-351,921-954` creates `web_search_options`; `litellm_core_utils/get_llm_provider_logic.py:206-230` strips the adapter prefix. These locations are version-specific and must be rechecked after dependency upgrades. +- LiteLLM JSON provider system: `llms/openai_like/providers.json`, `json_loader.py`, `dynamic_config.py`, `README.md`; upstream [providers.json on main](https://github.com/BerriAI/litellm/blob/main/litellm/llms/openai_like/providers.json) and [adding OpenAI-compatible providers](https://docs.litellm.ai/docs/contributing/adding_openai_compatible_providers). From 93f81b3f67771677ec02147f76b21033e0a5d6ba Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 01:50:50 +0200 Subject: [PATCH 06/12] fix: satisfy mypy on the litellm adapter call in venice web-search tests --- tests/unit/test_venice_web_search.py | 15 +++++++++------ 1 file changed, 9 insertions(+), 6 deletions(-) diff --git a/tests/unit/test_venice_web_search.py b/tests/unit/test_venice_web_search.py index df2ac903..5183cd51 100644 --- a/tests/unit/test_venice_web_search.py +++ b/tests/unit/test_venice_web_search.py @@ -206,16 +206,19 @@ def test_litellm_adapter_derives_no_web_search_options_from_the_adapted_body() - ) provider = VeniceUpstreamProvider(api_key="sk-test") - adapter = LiteLLMAnthropicMessagesAdapter() + adapter = LiteLLMAnthropicMessagesAdapter() # type: ignore[no-untyped-call] body = _body(tools=[WEB_SEARCH_TOOL, FUNCTION_TOOL]) + def translate(request: dict[str, Any]) -> dict: + # litellm types the request as a TypedDict; these bodies are built + # from client JSON, so they are plain dicts at this seam. + translated, _ = adapter.translate_anthropic_to_openai(request) # type: ignore[arg-type] + return dict(translated) + # Unadapted, litellm derives the parameter Venice rejects. - before, _ = adapter.translate_anthropic_to_openai( - {"model": "m", **_body(tools=[WEB_SEARCH_TOOL])} - ) + before = translate({"model": "m", **_body(tools=[WEB_SEARCH_TOOL])}) assert "web_search_options" in before provider.adapt_messages_request(body, _model()) - after, _ = adapter.translate_anthropic_to_openai({"model": "m", **body}) - assert "web_search_options" not in after + assert "web_search_options" not in translate({"model": "m", **body}) From a89b428bd016ec6c5899b3f3641e52a9fe80e9cd Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 23:57:09 +0200 Subject: [PATCH 07/12] fix: drop venice text models whose price book would bill completions free --- routstr/upstream/venice.py | 17 +++++-- tests/unit/test_upstream_venice.py | 72 ++++++++++++++++++++++++++++++ 2 files changed, 85 insertions(+), 4 deletions(-) diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index 57e96b24..0d0040fd 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -226,7 +226,7 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): if not isinstance(spec, dict) or spec.get("offline"): return None - pricing = self._parse_pricing(spec.get("pricing")) + pricing = self._parse_pricing(spec.get("pricing"), str(model_type)) if pricing is None: return None @@ -266,18 +266,27 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): ), ) - def _parse_pricing(self, raw: Any) -> Pricing | None: + def _parse_pricing(self, raw: Any, model_type: str) -> Pricing | None: if not isinstance(raw, dict): return None # The ``extended`` tier some models charge past a context threshold is # ignored: billing it would overcharge every request staying under it. input_usd = _usd(raw.get("input")) - if input_usd is None: + output_usd = _usd(raw.get("output")) + # Embeddings produce no completion tokens, so only they may omit an + # output price. Anywhere else a missing or all-zero price would serve + # completions free and a negative one would credit the caller, the + # same guards ``generic.py`` applies to this price book. + if output_usd is None and model_type == "embedding": + output_usd = 0.0 + if input_usd is None or output_usd is None: + return None + if input_usd < 0 or output_usd < 0 or (input_usd == 0 and output_usd == 0): return None return Pricing( prompt=input_usd / _USD_PER_MILLION, - completion=(_usd(raw.get("output")) or 0.0) / _USD_PER_MILLION, + completion=output_usd / _USD_PER_MILLION, input_cache_read=(_usd(raw.get("cache_input")) or 0.0) / _USD_PER_MILLION, input_cache_write=(_usd(raw.get("cache_write")) or 0.0) / _USD_PER_MILLION, ) diff --git a/tests/unit/test_upstream_venice.py b/tests/unit/test_upstream_venice.py index 1c6223a8..742c38ae 100644 --- a/tests/unit/test_upstream_venice.py +++ b/tests/unit/test_upstream_venice.py @@ -191,6 +191,78 @@ def test_offline_and_unpriced_models_are_dropped() -> None: assert "unpriced-text" not in ids +def _priced_entry(model_id: str, model_type: str, pricing: dict[str, Any]) -> dict: + return { + "id": model_id, + "type": model_type, + "created": 1727966436, + "model_spec": {"name": model_id, "pricing": pricing}, + } + + +@pytest.mark.parametrize( + "pricing", + [ + pytest.param({"input": {"usd": 0.2, "diem": 0.2}}, id="missing-output"), + pytest.param( + {"input": {"usd": 0.0, "diem": 0.0}, "output": {"usd": 0.0, "diem": 0.0}}, + id="both-zero", + ), + pytest.param( + {"input": {"usd": -0.2, "diem": 0.2}, "output": {"usd": 0.9, "diem": 0.9}}, + id="negative-input", + ), + pytest.param( + {"input": {"usd": 0.2, "diem": 0.2}, "output": {"usd": -0.9, "diem": 0.9}}, + id="negative-output", + ), + ], +) +def test_text_models_that_would_bill_free_or_negative_are_dropped( + pricing: dict[str, Any], +) -> None: + models, _ = _fetch( + {"object": "list", "data": [_priced_entry("bad-text", "text", pricing)]} + ) + assert models == [] + + +def test_embedding_with_only_an_input_price_is_listed() -> None: + models, _ = _fetch( + { + "object": "list", + "data": [ + _priced_entry("emb", "embedding", {"input": {"usd": 0.05, "diem": 0}}) + ], + } + ) + assert [m.id for m in models] == ["emb"] + assert models[0].pricing.prompt == pytest.approx(0.05 / 1_000_000) + assert models[0].pricing.completion == 0.0 + + +def test_embedding_with_a_negative_price_is_dropped() -> None: + models, _ = _fetch( + { + "object": "list", + "data": [ + _priced_entry("emb", "embedding", {"input": {"usd": -0.05, "diem": 0}}) + ], + } + ) + assert models == [] + + +def test_text_model_with_one_zero_price_is_listed() -> None: + """Only both-zero is free; a free prompt with a paid completion is priced.""" + pricing = {"input": {"usd": 0.0, "diem": 0}, "output": {"usd": 0.9, "diem": 0}} + models, _ = _fetch( + {"object": "list", "data": [_priced_entry("t", "text", pricing)]} + ) + assert [m.id for m in models] == ["t"] + assert models[0].pricing.completion == pytest.approx(0.9 / 1_000_000) + + def test_model_name_drops_the_venice_prefix() -> None: provider = VeniceUpstreamProvider(api_key="sk-test") assert provider.transform_model_name("venice/venice-uncensored-1-2") == ( From 87f462daf2a7a91e1f21b749405664cefcafd2cd Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 23:57:58 +0200 Subject: [PATCH 08/12] fix: accept anthropic web search max_uses of one or more on venice --- routstr/upstream/venice.py | 16 +++++++++---- tests/unit/test_venice_web_search.py | 36 +++++++++++++++++++++++++++- 2 files changed, 46 insertions(+), 6 deletions(-) diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index 0d0040fd..c243dd7a 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -42,9 +42,12 @@ _WEB_SEARCH_SUFFIX = ":enable_web_search=auto&enable_web_citations=true" # Anthropic web-search constraints with no Venice equivalent. Honouring the # request means enforcing them, so a request that sets one is refused rather -# than answered by a search that ignored it. +# than answered by a search that ignored it. ``max_uses`` is absent on purpose: +# ``auto`` runs at most one search per request, so any cap of 1 or more is +# already met, while domain filters and location would be silently ignored. +# Only ``max_uses: 0``, a request for no search at all, cannot be honoured. _UNENFORCEABLE_WEB_SEARCH_KEYS = frozenset( - {"max_uses", "allowed_domains", "blocked_domains", "user_location"} + {"allowed_domains", "blocked_domains", "user_location"} ) @@ -134,9 +137,12 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): key for tool in search_tools for key, value in tool.items() - if key in _UNENFORCEABLE_WEB_SEARCH_KEYS - and value is not None - and value != [] + if ( + key in _UNENFORCEABLE_WEB_SEARCH_KEYS + and value is not None + and value != [] + ) + or (key == "max_uses" and value == 0) } ) if unenforceable: diff --git a/tests/unit/test_venice_web_search.py b/tests/unit/test_venice_web_search.py index 5183cd51..960176d3 100644 --- a/tests/unit/test_venice_web_search.py +++ b/tests/unit/test_venice_web_search.py @@ -136,7 +136,6 @@ async def test_other_providers_keep_their_existing_behaviour() -> None: @pytest.mark.parametrize( "tool", [ - {"type": "web_search_20250305", "name": "web_search", "max_uses": 5}, { "type": "web_search_20250305", "name": "web_search", @@ -189,6 +188,41 @@ def test_forcing_web_search_through_tool_choice_is_refused() -> None: assert excinfo.value.status_code == 400 +@pytest.mark.asyncio +async def test_claude_code_web_search_tool_is_accepted() -> None: + """Claude Code always sends ``max_uses: 8``; Venice's single ``auto`` + search already stays under any cap of one or more.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + tool = { + "type": "web_search_20250305", + "name": "web_search", + "allowed_domains": None, + "blocked_domains": None, + "max_uses": 8, + } + + kwargs = await _dispatch(provider, _body(tools=[tool])) + + assert "web_search_options" not in kwargs + assert "tools" not in kwargs + assert kwargs["model"] == ( + "openai/deepseek-v4-flash-0731:enable_web_search=auto&enable_web_citations=true" + ) + + +def test_zero_max_uses_is_refused() -> None: + """``auto`` may still search, so a request for no search cannot be met.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": 0} + + with pytest.raises(UpstreamError) as excinfo: + provider.adapt_messages_request(_body(tools=[tool]), _model()) + + assert excinfo.value.status_code == 400 + assert excinfo.value.code == "UNSUPPORTED_WEB_SEARCH_OPTION" + assert excinfo.value.details == {"unsupported_options": ["max_uses"]} + + def test_tool_named_web_search_without_the_type_marker_is_caught() -> None: """litellm matches on either marker, so this one would also be rewritten.""" provider = VeniceUpstreamProvider(api_key="sk-test") From 826dd39212c42d893394c1a2811955bf99750d9a Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 23:58:01 +0200 Subject: [PATCH 09/12] test: pin venice tool_choice refusal details and web-search-only tool_choice drop --- tests/unit/test_venice_web_search.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tests/unit/test_venice_web_search.py b/tests/unit/test_venice_web_search.py index 960176d3..9bcae72b 100644 --- a/tests/unit/test_venice_web_search.py +++ b/tests/unit/test_venice_web_search.py @@ -186,6 +186,19 @@ def test_forcing_web_search_through_tool_choice_is_refused() -> None: provider.adapt_messages_request(body, _model()) assert excinfo.value.status_code == 400 + assert excinfo.value.code == "UNSUPPORTED_WEB_SEARCH_OPTION" + assert excinfo.value.details == {"unsupported_options": ["tool_choice"]} + + +def test_web_search_only_request_drops_tool_choice() -> None: + """Without tools left, a surviving tool_choice is rejected upstream.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + body = _body(tools=[WEB_SEARCH_TOOL], tool_choice={"type": "auto"}) + + provider.adapt_messages_request(body, _model()) + + assert "tools" not in body + assert "tool_choice" not in body @pytest.mark.asyncio From 49138fb46ba3a1cb6c033da3c85f503145639956 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 23:58:01 +0200 Subject: [PATCH 10/12] docs: note venice keeps tool_choice any when function tools remain beside web search --- routstr/upstream/venice.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index c243dd7a..deffb581 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -166,6 +166,11 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): remaining = [tool for tool in tools if not _is_web_search_tool(tool)] if remaining: + # A caller's ``tool_choice: any`` is kept and litellm maps it to + # OpenAI ``required``, so one of the remaining function tools must + # now be called where Anthropic would have let a search satisfy it. + # Deliberate: OpenRouter never rewrites tool_choice for web search + # either, and guessing an alternative would change caller intent. body["tools"] = remaining else: body.pop("tools", None) From aae417763e445eb8d6bbf0870809037b252ace59 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Fri, 25 Sep 2026 23:58:01 +0200 Subject: [PATCH 11/12] docs: remove superseded venice web search investigation log --- VENICE_WEB_SEARCH.md | 151 ------------------------------------------- 1 file changed, 151 deletions(-) delete mode 100644 VENICE_WEB_SEARCH.md diff --git a/VENICE_WEB_SEARCH.md b/VENICE_WEB_SEARCH.md deleted file mode 100644 index 18abcc07..00000000 --- a/VENICE_WEB_SEARCH.md +++ /dev/null @@ -1,151 +0,0 @@ -# Venice web search through Routstr `/v1/messages` - -Status: **implemented on branch `feat/venice-provider`** and **verified against live Venice** (2026-09-25, see "Live verification"). Investigated 2026-09-24, built 2026-09-25. The production request body and a live Venice credential were unavailable; distinguish reproduced local behavior from the inferred production trigger below. - -## What shipped - -Two commits on `feat/venice-provider` (branched from `main`): - -- `f77896f1` `feat: add venice upstream provider` — `VeniceUpstreamProvider` ported text-and-embedding only from `feat/venice-provider-image-pricing`. Image, inpaint and upscale families are dropped rather than listed, because their price book lives in the image-billing commit that did not come along; listing them here would hand out unpriced inference. -- `5d1004d3` `feat: translate anthropic web search to venice search on /v1/messages` — the fix below. - -**The seam.** `BaseUpstreamProvider.adapt_messages_request(body, model_obj) -> str` is a provider's last word on an allowlisted Anthropic body: it may rewrite the body in place and returns a suffix for the upstream model name. `dispatch_anthropic_messages` calls it after the `ALLOWED_MESSAGES_REQUEST_FIELDS` filter and appends the suffix to `transform_model_name(model.id)`. The base implementation returns `""`, so no other provider changes shape. - -**The Venice override.** Any tool litellm would read as web search — `type` starting `web_search`, or `name == "web_search"`, the same two markers its adapter matches — is lifted out of `tools`, and the intent is re-expressed as the model feature suffix `:enable_web_search=auto&enable_web_citations=true`. `auto` matches Anthropic semantics, where declaring the tool leaves the decision to the model. Citations are requested because litellm's Anthropic response translation carries no `venice_parameters`, so inline `[REF]n[/REF]` markers are the only surviving signal of which sources were used. Remaining function tools and their `tool_choice` travel untouched; when the search tool was the only one, `tool_choice` is dropped with it, since an OpenAI-shaped upstream rejects a choice with no tools. - -**Refusals.** `max_uses`, `allowed_domains`, `blocked_domains` and `user_location` have no Venice equivalent, and a `tool_choice` naming `web_search` cannot be honoured because Venice's search is not a callable tool. Each returns 400 `UNSUPPORTED_WEB_SEARCH_OPTION` before the upstream call rather than a search that quietly ignored the constraint. A key carrying `null` or `[]` states no constraint and is read as absent. - -**Verification.** `tests/unit/test_venice_web_search.py` (13 tests) covers the adaptation, the refusals, and — running the real litellm adapter — asserts the unadapted body derives `web_search_options` while the adapted one does not. `tests/integration/test_venice_web_search_wire.py` runs the whole dispatch against a loopback OpenAI-compatible server and reads the bytes Venice would receive: `POST /v1/chat/completions`, no `web_search_options`, `model` carrying the suffix, the function tool in OpenAI shape. Full unit suite 1722 passed; ruff and mypy clean. - -That wire test also pins a dependency on startup config: without `configure_litellm()` (applied in `routstr/core/main.py`), litellm posts the Anthropic body to `/responses`, which Venice serves only in alpha. - -## Live verification - -Run 2026-09-25 against `api.venice.ai` with a real key. Each open question from the plan is now answered by observation rather than inference. - -**The suffix is honoured.** A `/v1/messages` request carrying an Anthropic `web_search_20250305` tool came back with a live figure and its source ("approximately $84,216.93 USD, according to CoinMarketCap.^6^") on `deepseek-v4-flash-0731`, and the same on `zai-org-glm-5-1` over a real stream. No 400. The control request without the tool searched nothing. - -**Streaming is intact.** The stream yields the full Anthropic event set — `message_start`, `content_block_start`, `content_block_delta`, `content_block_stop`, `message_delta`, `message_stop` — with usage on the final events. - -**Citations arrive as `^n^`, not `[REF]n[/REF]`.** The API reference describes the latter; live responses write superscript markers, matching Venice's own agent skill. Structured citations are confirmed lost: the Anthropic-shaped response carries only `content`, `id`, `model`, `role`, `stop_reason`, `stop_sequence`, `type`, `usage`, with no `venice_parameters`. The inline markers are the whole signal. - -**The capability gate is unnecessary.** All 123 text models in the live catalog report `supportsWebSearch: true` — none false, none missing the key. There is no Venice text model to refuse, so the `Model` field, `ModelRow` column and migration the plan called for are not worth building. Revisit only if Venice ships a text model without it. - -**`Pricing.web_search = 0.0` is right.** Venice bills search through the prompt: the same question cost 5,839 input tokens with search against 1,710 without, because the results are injected into the context. There is no separate per-search fee to price (Venice documents one only for `enable_x_search`, which this path never enables). Those tokens are billed by the existing per-token path, and `_calculate_usd_max_costs` reserves against the full context window, so an inflated prompt stays inside the reservation. - -Still unobserved: behaviour when Venice's search itself fails or returns nothing, and `enable_web_scraping`, which this path never turns on. - -## Incident and conclusion - -Routstr 0.4.7 logged a `/v1/messages` request (`09adf07c-1456-4ccb-8276-824016392219`) dispatched to `https://api.venice.ai/api/v1` with LiteLLM model `openai/deepseek-v4-flash-0731`. Venice returned HTTP 400: `Unrecognized key(s) in object: 'web_search_options'`. The proxy then logged `provider=generic`, `status_code=400`, `retry=true`. - -These labels describe different layers. `generic` is Routstr's provider row; `openai/` selects LiteLLM's OpenAI-compatible Chat Completions adapter, not the destination service. `api_base` still points to Venice. The installed LiteLLM does not recognize `venice/` as a provider prefix, so simply renaming it breaks routing. LiteLLM removes `openai/` when resolving the provider; the *outbound* model ID should be the bare Venice ID. Capture one sanitized outbound request to verify the wire payload rather than relying on the dispatch log. - -The 400 is about the **unsupported field**, not the prefix. Routstr allowlists `tools` but does not forward client-supplied `web_search_options`. The installed LiteLLM 1.93.2 Anthropic Messages adapter recognizes a tool whose `type` starts with `web_search` or whose `name` is `web_search`, removes it from ordinary function tools, and inserts `web_search_options: {}` into the OpenAI-shaped call. A local, credential-free repro with `web_search_20250305` produced that exact field; an ordinary function tool did not. The production log has no input `tools` field, so the specific incoming trigger remains **strongly indicated, not proved**. A sanitized copy of the incoming `tools` types/names would settle it. - -The shared `litellm.drop_params=True` setting is not sufficient to protect arbitrary OpenAI-compatible servers: the adapter creates this field *after* Routstr filters the incoming body. Similarly, the proxy's `correct_request` retries on client request fields, not on this post-translation field. Its `retry=true` means another candidate provider may be attempted for a 400, not that the same Venice request becomes valid. - -## Venice's actual search interfaces - -Venice documents **model-integrated web search** for `POST /chat/completions` using `venice_parameters.enable_web_search` (`"off"`, `"auto"`, `"on"`; default `"off"`). `"on"` forces search; `"auto"` leaves it to the model. `venice_parameters.enable_web_citations: true` asks for inline source references. The response may include `venice_parameters.web_search_citations`; citations arrive in the first streaming chunk or the non-streaming response. The model feature suffix is another documented way to set these without an extra request field: - -```text -:enable_web_search=auto&enable_web_citations=true -``` - -For standalone retrieval, Venice also has `POST /augment/search` and `/augment/scrape`, but that is a different architecture: Routstr would have to execute search, supply results to the model, handle citations and account for the extra call. Venice model metadata advertises `model_spec.capabilities.supportsWebSearch` for model-specific support; verify the actual configured model at runtime rather than assuming all Venice models support it. The incident alone does **not** prove `deepseek-v4-flash-0731` advertises this capability. - -Important documentation discrepancy: Venice's first-party `venice-chat` skill describes `tools: [{"type":"web_search"}]` as a built-in toggle, while the official Chat Completions OpenAPI schema currently says only function tools are supported. Treat the `venice_parameters`/suffix route as the documented baseline; test built-in `tools` on the live API before depending on it. Neither source documents accepting the top-level `web_search_options` field rejected in this incident. - -## Routstr implementation plan - -1. **Write a red regression at the actual seam.** Extend `tests/unit/test_messages_litellm_dispatch.py` with a generic provider pointing at Venice and an Anthropic `/v1/messages` request containing a server-side `web_search_20250305` tool. Exercise `BaseUpstreamProvider._dispatch_anthropic_messages` through `messages_dispatch.dispatch_anthropic_messages`. Use the real LiteLLM translation in a local, network-free adapter assertion, not only an `acreate` mock: assert that the current path produces `web_search_options` and that the proposed path does not. Cover both bearer-key and x-cashu callers because both use the same dispatcher. -2. **Add a narrowly scoped Venice capability branch** before `litellm.anthropic.messages.acreate` in `routstr/upstream/messages_dispatch.py`, with the provider identity supplied by `BaseUpstreamProvider` (or an explicit provider capability). Match the parsed Venice hostname exactly, not an unbounded substring or a model name; generic non-Venice hosts must remain unchanged. Keep `openai/` as the LiteLLM adapter prefix and `api_base` as Venice. Do not rewrite the public `Model.id` or `forwarded_model_id`. -3. **Translate intent, not merely delete it.** On a Venice route, remove only Anthropic *server-side web-search* tools from the `tools` sent into LiteLLM so its adapter cannot synthesize `web_search_options`. Preserve ordinary function tools and their `tool_choice`. Enable Venice search for this request with the documented suffix on the **upstream** model ID, e.g. `:enable_web_search=auto`, optionally adding `&enable_web_citations=true` if the response path preserves citations. This avoids relying on unknown `extra_body` behavior in LiteLLM's Anthropic adapter. Alternatively, pass `venice_parameters` only after a wire-level test demonstrates it survives that adapter. Never silently remove a requested search tool without enabling an equivalent service. -4. **Make unsupported semantics explicit.** Decide and test how to handle `max_uses`, `allowed_domains`/`blocked_domains`, forced `tool_choice` targeting web search, duplicate search tools, or a model without `supportsWebSearch`: Venice's search switch is not a one-to-one implementation of every Anthropic tool constraint. Where equivalence cannot be guaranteed, return a clear pre-dispatch 4xx or explicitly documented degraded behavior; avoid a success that pretends the requested constraints were enforced. Do not translate client-supplied arbitrary `venice_parameters` through the `/v1/messages` allowlist. -5. **Preserve the API contract.** Test streamed and non-streamed Anthropic-shaped responses, function tools coexisting with search, no-search Venice requests, non-Venice OpenAI-compatible requests, and handling of `venice_parameters.web_search_citations`. The existing LiteLLM → Anthropic response conversion may drop Venice-specific citation metadata; verify it with captured fixtures before promising search citations. If metadata is lost, either map it deliberately to the chosen client-visible format or document that search works without structured citations. -6. **Protect billing and routing.** `GenericUpstreamProvider.fetch_models` currently sets `Pricing.web_search=0.0`; check Venice's live web-search charges and returned usage/cost fields. Ensure reservation/max-cost estimation and final charge include any search fees before enabling paid searches, or fail closed if they cannot be priced. The 400 fallback behavior in `routstr/proxy.py` must not route a search-required request to a provider that silently loses search; inspect candidate capabilities and keep payment reversal correct. Keep the suffix out of catalog IDs, public response model IDs, and price lookups. -7. **Verify live with a Venice test key** after network-free tests: record sanitized outbound JSON and check absence of `web_search_options`, bare upstream model name plus the Venice suffix (if chosen), successful web-enabled reply, citations/usage shape, and billing reconciliation for `stream=true` and `false`. Check model capability from `/models` first. No live request was sent in this investigation. - -Acceptance: web-search requests on a Venice model that supports search either complete with search enabled and correctly billed, or fail before the upstream call with a specific unsupported-capability error; no request emits `web_search_options` toward Venice. Requests without search and other providers retain their existing behavior. No unsupported search constraints are silently accepted. - -## Related reports and prior art - -- [LiteLLM #10714](https://github.com/BerriAI/litellm/issues/10714) and its referenced [#10664](https://github.com/BerriAI/litellm/issues/10664) concern Anthropic `web_search_20250305` support in LiteLLM; these are historical context for adapter differences, **not** a verified patch for this Venice 400. -- [LiteLLM #14250](https://github.com/BerriAI/litellm/issues/14250) documents that even OpenAI Chat Completions web search via `web_search_options` is model-specific; an OpenAI-compatible endpoint need not implement it. -- [LiteLLM web-search interception integration](https://docs.litellm.ai/docs/web_search_interception) is an alternative architecture with an external search provider and an agentic follow-up, not a drop-in change to Routstr's current direct `litellm.anthropic.messages.acreate` path. A [follow-up duplicate-kwargs report](https://github.com/BerriAI/litellm/issues) was found in the broader search but not established as this issue's cause; do not infer a fix from it. -- First-party [Venice Chat skill](https://github.com/veniceai/skills/blob/main/skills/venice-chat/SKILL.md) gives provider-native search examples. The official API reference below takes precedence for the implementable request shape. Searches for an exact public Venice + LiteLLM `web_search_options` 400 fix did **not** yield a verified matching issue or merged patch. Do not claim an upstream fix exists without reproducing it in the pinned version. - -## Branch `feat/venice-provider-image-pricing` — a Venice provider class already exists - -Checked 2026-09-24 on that branch (two commits ahead of `main`, no PR open). `routstr/upstream/venice.py` adds `VeniceUpstreamProvider(BaseUpstreamProvider)` with `provider_type = "venice"`, a pinned `default_base_url = "https://api.venice.ai/api/v1"` (`fixed_base_url: True`), a catalog fetch across Venice's model families, text and per-image-tier pricing, and `transform_model_name` stripping a `venice/` prefix. Tests in `tests/unit/test_upstream_venice.py` are catalog and pricing only. - -It does **not** fix this incident. Verified at runtime on the branch: `VeniceUpstreamProvider.litellm_provider_prefix` is `None`, so `get_litellm_provider_prefix()` still resolves to `openai/` through `detect_litellm_prefix`, and `supports_anthropic_messages` is `False`, so `/v1/messages` still goes through `messages_dispatch` into LiteLLM's Anthropic adapter — the same code that synthesizes `web_search_options`. The file contains no web-search or `venice_parameters` handling. - -What it does change is **where the fix belongs**. With this class merged, step 2 of the plan above needs no hostname matching: provider identity is the class itself, so the Venice branch becomes a method on `VeniceUpstreamProvider` rather than a URL test inside the shared dispatcher. Adopt it and revise the plan as follows: - -- Put the translation on the provider, e.g. an override of `_dispatch_anthropic_messages` (or a narrow hook the base dispatcher calls) that strips Anthropic server-side web-search tools and enables Venice search. Keep `openai/` as the LiteLLM adapter prefix. -- Do **not** append the `:enable_web_search=…` suffix inside `transform_model_name`. `base.py` calls it on the chat/completions and model-listing paths too (around lines 705, 724, 795), so a suffix there would leak into unrelated requests. Scope it to the messages dispatch call. -- The incident ran on a `generic` row, not this class. Using it means re-creating the Venice upstream row as `provider_type="venice"`; `_build_from_row` takes only `api_key` and `provider_fee` because the base URL is pinned. A stale `generic` row keeps the old behavior. -- `_parse_pricing` returns text `Pricing` without a `web_search` rate (defaults to `0.0`), so per-search charges are still unpriced — the billing item in step 6 stands unchanged. - -The branch is unreviewed and carries an unrelated image-generation billing commit (27 files, ~3.9k insertions). Landing the web-search work on top of it couples this fix to that review. Decide explicitly: build on the branch, or implement against `main` and rebase once the provider lands. - -## Adding Venice support to LiteLLM - -Investigated 2026-09-24 against installed LiteLLM 1.93.2 and upstream `main` (published 1.102.1). - -### What already exists - -Venice is **already registered** in LiteLLM, but only as a bare JSON entry. `litellm/llms/openai_like/providers.json` contains, on both the pinned version and upstream `main`: - -```json -"veniceai": { - "base_url": "https://api.venice.ai/api/v1", - "api_key_env": "VENICE_AI_API_KEY" -} -``` - -Verified locally: `litellm.get_llm_provider("veniceai/deepseek-v4-flash-0731")` resolves to `("deepseek-v4-flash-0731", "veniceai")`, while `venice/...` raises `LLM Provider NOT provided`. `veniceai` is **not** in `litellm.provider_list` or the `LlmProviders` enum — it resolves through `JSONProviderRegistry`, which `get_llm_provider_logic.py` checks before the enum. Upstream `main` has no `litellm/llms/venice*` directory, no Venice entries in `model_prices_and_context_window.json`, and `docs.litellm.ai/docs/providers/venice` returns 404. The `venice` block in the installed `provider_endpoints_support_backup.json` describes a provider that was never merged. - -### The JSON entry does not fix this incident - -JSON providers inherit `OpenAIGPTConfig`, whose supported-parameter list includes `web_search_options`. Verified locally with the generated config class: `get_supported_openai_params` returns 26 params including `web_search_options`, and `map_openai_params({"web_search_options": {}}, drop_params=True)` keeps the field. So `litellm.drop_params` will not remove it, and switching Routstr's prefix from `openai/` to `veniceai/` still emits the field Venice rejects. `get_optional_params(..., custom_llm_provider="veniceai", extra_body={"venice_parameters": {...}})` does keep `extra_body` alongside `web_search_options`; whether that survives the Anthropic-messages adapter to the wire is **untested**, as no live request was made. - -### Prior attempts and maintainer stance - -- [#17962](https://github.com/BerriAI/litellm/pull/17962) **merged** — the two-line `providers.json` entry above, one file, no tests. -- [#17948](https://github.com/BerriAI/litellm/pull/17948) **closed unmerged** — a full `VeniceAIChatConfig(OpenAILikeChatConfig)` with a `VENICE_PARAMS` set (`enable_web_search`, `enable_web_citations`, `character_slug`, …) nested into `venice_parameters` by `transform_request`, plus enum, URL detection, docs, and 428 lines of tests. A maintainer replied that provider-specific params already pass through automatically and pointed at the providers.json path; the author closed it in favor of #17962. -- [#18248](https://github.com/BerriAI/litellm/pull/18248) **closed** (stale) — wired `veniceai` into `constants.py`, `types/utils.py`, URL detection, `provider_endpoints_support.json`, and docs. -- [#26970](https://github.com/BerriAI/litellm/pull/26970) (Venice model prices, fixes [#24229](https://github.com/BerriAI/litellm/issues/24229)) and [#23670](https://github.com/BerriAI/litellm/pull/23670) (docs) both **closed unmerged**. -- Feature requests [#8833](https://github.com/BerriAI/litellm/issues/8833) and [#9093](https://github.com/BerriAI/litellm/issues/9093) are closed. - -Treat that history as the main risk: the nesting problem this project needs was proposed once and rejected as unnecessary. A new PR must argue what `providers.json` cannot express, rather than restating the request. - -### Option A — extend the JSON provider system (recommended upstream path) - -`param_mappings` only renames a key; it cannot nest `enable_web_search` under `venice_parameters`, and nothing in the schema can mark an inherited param unsupported. Two small additive fields in `dynamic_config.py` close both gaps generically, for every OpenAI-compatible provider that rejects inherited OpenAI extras: - -- `unsupported_params: ["web_search_options"]` — removed from `get_supported_openai_params`, so `drop_params` handles it through the existing path. -- `nest_params_under: "venice_parameters"` with the member list — `map_openai_params`/`transform_request` build the nested object. - -Scope: `llms/openai_like/dynamic_config.py`, `providers.json`, `llms/openai_like/README.md`, plus tests under `tests/test_litellm/`. This stays inside the system the maintainer endorsed and benefits other providers, which is the strongest available argument for merge. - -### Option B — first-class Python provider - -Revive the #17948 + #18248 shape: `litellm/llms/venice_ai/chat/transformation.py`, `LlmProviders.VENICE_AI` in `types/utils.py`, `constants.py` provider list, `api.venice.ai` detection in `get_llm_provider_logic.py`, `__init__.py`/`utils.py` wiring, `ProviderConfigManager` registration, `model_prices_and_context_window.json` (+ backup) from Venice `/models`, `provider_endpoints_support.json`, `docs/my-website/docs/providers/venice.md` + `sidebars.js`, and tests under `tests/test_litellm/llms/venice_ai/`. Contributing requires a signed CLA, at least one test, and a Greptile review request. Only this option can also map an Anthropic `web_search_*` tool to `enable_web_search` inside LiteLLM, and only for callers that reach the chat path with that tool intact. - -Both options are upstream work on a third-party project with an uncertain merge outcome and a release lag. Neither removes the need for the Routstr-side plan above, which is the only change that fixes the incident on the pinned 1.93.2. - -### If Routstr adopts `veniceai/` later - -`detect_litellm_prefix` in `routstr/upstream/litellm_routing.py` would map `api.venice.ai` to `veniceai/`. Gate that on the installed LiteLLM version: the prefix resolves only while the JSON entry exists, it is absent from `litellm.provider_list`, and no Venice model carries LiteLLM pricing, so Routstr's own pricing path stays authoritative. On its own, the prefix change does not stop `web_search_options`. - -## Primary sources and local evidence - -- [Venice Chat Completions API](https://docs.venice.ai/api-reference/endpoint/chat/completions) — `venice_parameters`, search modes, response citations, strict request schema. -- [Venice Model Feature Suffix](https://docs.venice.ai/api-reference/endpoint/chat/model_feature_suffix) — `:=` and combined suffixes. -- [Venice Web Search API](https://docs.venice.ai/api-reference/endpoint/augment/search), [Web Search and Scraping guide](https://docs.venice.ai/guides/tools/web-retrieval), [Venice model catalog](https://docs.venice.ai/api-reference/endpoint/models/list). -- Local: `routstr/upstream/litellm_routing.py:24-116`, `routstr/upstream/base.py:361-372,2493-2508`, `routstr/upstream/messages_dispatch.py:59-78,479-531`, `routstr/upstream/generic.py:92-136,209-235`, `routstr/proxy.py:857-923`, `tests/unit/test_messages_litellm_dispatch.py`, `uv.lock` (LiteLLM 1.93.2). -- Installed dependency: `.venv/lib/python3.14/site-packages/litellm/llms/anthropic/experimental_pass_through/adapters/transformation.py:335-351,921-954` creates `web_search_options`; `litellm_core_utils/get_llm_provider_logic.py:206-230` strips the adapter prefix. These locations are version-specific and must be rechecked after dependency upgrades. -- LiteLLM JSON provider system: `llms/openai_like/providers.json`, `json_loader.py`, `dynamic_config.py`, `README.md`; upstream [providers.json on main](https://github.com/BerriAI/litellm/blob/main/litellm/llms/openai_like/providers.json) and [adding OpenAI-compatible providers](https://docs.litellm.ai/docs/contributing/adding_openai_compatible_providers). From 6f2c10bfd0954fc06b6e59709550eb80b018dd89 Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Sat, 26 Sep 2026 00:10:08 +0200 Subject: [PATCH 12/12] fix: refuse venice web search max_uses unless it is an integer of one or more --- routstr/upstream/venice.py | 13 +++++++++++-- tests/unit/test_venice_web_search.py | 16 +++++++++++++--- 2 files changed, 24 insertions(+), 5 deletions(-) diff --git a/routstr/upstream/venice.py b/routstr/upstream/venice.py index deffb581..379a4a66 100644 --- a/routstr/upstream/venice.py +++ b/routstr/upstream/venice.py @@ -131,7 +131,8 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): return "" # A key carrying null or an empty list states no constraint, so it is - # read as absent rather than refused. + # read as absent rather than refused. ``auto`` runs at most one search, + # so only an integer ``max_uses`` of one or more is known to be met. unenforceable = sorted( { key @@ -142,7 +143,15 @@ class VeniceUpstreamProvider(BaseUpstreamProvider): and value is not None and value != [] ) - or (key == "max_uses" and value == 0) + or ( + key == "max_uses" + and value is not None + and not ( + isinstance(value, int) + and not isinstance(value, bool) + and value >= 1 + ) + ) } ) if unenforceable: diff --git a/tests/unit/test_venice_web_search.py b/tests/unit/test_venice_web_search.py index 9bcae72b..a836e956 100644 --- a/tests/unit/test_venice_web_search.py +++ b/tests/unit/test_venice_web_search.py @@ -223,10 +223,20 @@ async def test_claude_code_web_search_tool_is_accepted() -> None: ) -def test_zero_max_uses_is_refused() -> None: - """``auto`` may still search, so a request for no search cannot be met.""" +@pytest.mark.parametrize("max_uses", [1, None]) +def test_max_uses_of_one_or_absent_is_accepted(max_uses: Any) -> None: provider = VeniceUpstreamProvider(api_key="sk-test") - tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": 0} + tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": max_uses} + + assert provider.adapt_messages_request(_body(tools=[tool]), _model()) != "" + + +@pytest.mark.parametrize("max_uses", [0, -1, 1.5, True, "0", "8"]) +def test_max_uses_other_than_a_positive_integer_is_refused(max_uses: Any) -> None: + """``auto`` may still search, so a cap below one cannot be met, and a + malformed cap cannot be shown to be met.""" + provider = VeniceUpstreamProvider(api_key="sk-test") + tool = {"type": "web_search_20250305", "name": "web_search", "max_uses": max_uses} with pytest.raises(UpstreamError) as excinfo: provider.adapt_messages_request(_body(tools=[tool]), _model())