diff --git a/docs/api/endpoints.md b/docs/api/endpoints.md index d8b610ca..53743643 100644 --- a/docs/api/endpoints.md +++ b/docs/api/endpoints.md @@ -280,6 +280,28 @@ Billing is input-token based (output tokens are free on Jev); the response's published rate ($0.042 per million input tokens, output free). Override the model row if TypeSafe changes pricing. +## Decisions (OpenAI) + +### Create Decision + +Ask OpenAI's Decisions API (powered by `gpt-6-luna`) to pick from a finite set +of answers for text or image context. The API is in limited preview: accounts +that are not enrolled get `403 Decision API is not enabled for this user`. + +```http +POST /v1/decisions +``` + +The request body is forwarded to `https://api.openai.com/v1/decisions` +unchanged; Routstr only reads the top-level `model` to route and price the +request, and settles from the response's `usage` like embeddings. + +**Notes:** + +- Only the `openai` provider serves this endpoint. A model that no OpenAI + provider on the node offers returns `400 unsupported_request`. +- Pricing uses the node's catalog rate for the requested model. + ## Images (Coming Soon) ### Create Image diff --git a/docs/api/overview.md b/docs/api/overview.md index 8a4a445d..9a737e36 100644 --- a/docs/api/overview.md +++ b/docs/api/overview.md @@ -105,6 +105,7 @@ Standard OpenAI-compatible endpoints: - **Chat Completions**: `/v1/chat/completions` - **Embeddings**: `/v1/embeddings` - **System One**: `/v1/systemone` (TypeSafe decision models) +- **Decisions**: `/v1/decisions` (OpenAI Decisions API) - **Completions**: `/v1/completions` *(planned)* - **Images**: `/v1/images/generations` *(planned)* - **Audio**: `/v1/audio/transcriptions` *(planned)* diff --git a/routstr/proxy.py b/routstr/proxy.py index 38e2dfdc..1d45420f 100644 --- a/routstr/proxy.py +++ b/routstr/proxy.py @@ -276,6 +276,7 @@ _ALLOWED_ENDPOINTS: dict[str, frozenset[str]] = { # -> {answers, usage}. Non-streaming, JSON in/out; billed from the # response's usage exactly like embeddings. "systemone": frozenset({"POST"}), + "decisions": frozenset({"POST"}), "models": frozenset({"GET"}), "attestation": frozenset({"GET"}), "tee/attestation": frozenset({"GET"}), @@ -695,6 +696,20 @@ async def _proxy( request=request, ) + if _canonical_api_path(path) == "decisions": + candidates = [ + (model, upstream) + for model, upstream in candidates + if upstream.supports_decisions + ] + if not candidates: + return create_error_response( + "unsupported_request", + f"No Decisions-capable provider found for model '{model_id}'", + 400, + request=request, + ) + # Reserve/max-cost checks use the best-ranked candidate; the failover loop # below rebinds (model_obj, upstream) per candidate so forwarding and # settlement always use the model of the provider actually being tried. diff --git a/routstr/upstream/base.py b/routstr/upstream/base.py index 8f46d90b..dfcb6c9b 100644 --- a/routstr/upstream/base.py +++ b/routstr/upstream/base.py @@ -311,7 +311,7 @@ def _openai_completion_path(path: str) -> str | None: def _x_cashu_path_has_settlement_handler(path: str) -> bool: canonical = path.rstrip("/") return _openai_completion_path(canonical) is not None or canonical.endswith( - ("embeddings", "messages", "messages/count_tokens", "systemone") + ("embeddings", "messages", "messages/count_tokens", "systemone", "decisions") ) @@ -334,6 +334,7 @@ class BaseUpstreamProvider: platform_url: str | None = None supports_anthropic_messages: bool = False + supports_decisions: bool = False # When None, the prefix is detected from `base_url` at dispatch time # (see `get_litellm_provider_prefix`). Subclasses set this to lock the # provider regardless of URL. @@ -3308,6 +3309,7 @@ class BaseUpstreamProvider: or path.endswith("messages") or path.endswith("messages/count_tokens") or path.endswith("systemone") + or path.endswith("decisions") ): if path.endswith("messages"): client_wants_streaming = False diff --git a/routstr/upstream/openai.py b/routstr/upstream/openai.py index f2f03cc5..81b64b8c 100644 --- a/routstr/upstream/openai.py +++ b/routstr/upstream/openai.py @@ -13,6 +13,7 @@ class OpenAIUpstreamProvider(BaseUpstreamProvider): provider_type = "openai" default_base_url = "https://api.openai.com/v1" platform_url = "https://platform.openai.com/api-keys" + supports_decisions = True def __init__(self, api_key: str, provider_fee: float = 1.01): super().__init__( diff --git a/tests/integration/test_openai_decisions.py b/tests/integration/test_openai_decisions.py new file mode 100644 index 00000000..18a54a84 --- /dev/null +++ b/tests/integration/test_openai_decisions.py @@ -0,0 +1,173 @@ +"""Integration test: POST /v1/decisions routed to OpenAI and billed from usage.""" + +import json +from typing import Any, AsyncGenerator +from unittest.mock import patch + +import httpx +import pytest +from httpx import AsyncClient + +from routstr.payment.models import Architecture, Model, Pricing +from routstr.proxy import refresh_model_maps +from routstr.upstream.base import BaseUpstreamProvider +from routstr.upstream.openai import OpenAIUpstreamProvider + +DECISIONS_REQUEST = { + "model": "gpt-6-luna", + "state": {"ticket": "Checkout shows a blank page after I click Pay."}, + "questions": { + "team": { + "type": "choice", + "instructions": "Which team should own this ticket?", + "criteria": {"payments": "Checkout and billing", "account": "Login"}, + } + }, +} + +DECISIONS_RESPONSE = { + "model": "gpt-6-luna", + "answers": {"team": {"type": "choice", "choice": "payments"}}, + "usage": {"input_tokens": 1000, "output_tokens": 200}, +} + + +def _luna_model(prompt_sats: float) -> Model: + return Model( + id="gpt-6-luna", + name="gpt-6-luna", + created=1, + description="GPT-6 Luna", + context_length=400_000, + architecture=Architecture( + modality="text+image->text", + input_modalities=["text", "image"], + output_modalities=["text"], + tokenizer="GPT", + instruct_type=None, + ), + pricing=Pricing(prompt=prompt_sats, completion=0.0, max_cost=50.0), + sats_pricing=Pricing(prompt=prompt_sats, completion=0.0, max_cost=50.0), + ) + + +class _StaticOpenAIProvider(OpenAIUpstreamProvider): + def __init__(self, model: Model) -> None: + super().__init__("key-openai", 1.0) + self._static_model = model + + def get_cached_models(self) -> list[Model]: + return [self._static_model] + + async def refresh_models_cache(self) -> None: + pass + + +class _StaticGenericProvider(BaseUpstreamProvider): + def __init__(self, model: Model) -> None: + super().__init__("https://generic.test/v1", "key-generic", 1.0) + self._static_model = model + + def get_cached_models(self) -> list[Model]: + return [self._static_model] + + async def refresh_models_cache(self) -> None: + pass + + +async def _install( + upstreams: list[BaseUpstreamProvider], +) -> AsyncGenerator[None, None]: + from routstr import proxy + + original_upstreams = proxy.get_upstreams() + with patch("routstr.proxy._upstreams", upstreams): + await refresh_model_maps() + yield + with patch("routstr.proxy._upstreams", original_upstreams): + await refresh_model_maps() + + +@pytest.fixture +async def luna_on_openai_and_generic( + patched_db_engine: None, +) -> AsyncGenerator[None, None]: + async for _ in _install( + [ + _StaticGenericProvider(_luna_model(0.0005)), + _StaticOpenAIProvider(_luna_model(0.001)), + ] + ): + yield + + +@pytest.fixture +async def luna_on_generic_only( + patched_db_engine: None, +) -> AsyncGenerator[None, None]: + async for _ in _install([_StaticGenericProvider(_luna_model(0.0005))]): + yield + + +@pytest.mark.integration +@pytest.mark.asyncio +async def test_decisions_routed_to_openai_and_billed_from_usage( + authenticated_client: AsyncClient, + luna_on_openai_and_generic: None, +) -> None: + sent_requests: list[httpx.Request] = [] + + async def fake_transport( + request: httpx.Request, *args: Any, **kwargs: Any + ) -> httpx.Response: + sent_requests.append(request) + return httpx.Response( + 200, + content=json.dumps(DECISIONS_RESPONSE).encode(), + headers={"content-type": "application/json"}, + ) + + with ( + patch( + "httpx.AsyncHTTPTransport.handle_async_request", + side_effect=fake_transport, + ), + patch( + "routstr.payment.cost_calculation.sats_usd_price", + return_value=0.0005, + ), + ): + response = await authenticated_client.post( + "/v1/decisions", json=DECISIONS_REQUEST + ) + + assert response.status_code == 200, response.text + payload = response.json() + assert payload["answers"]["team"]["choice"] == "payments" + + assert len(sent_requests) == 1 + assert str(sent_requests[0].url) == "https://api.openai.com/v1/decisions" + assert ( + json.loads(sent_requests[0].content)["questions"] + == (DECISIONS_REQUEST["questions"]) + ) + + assert payload["cost"]["input_msats"] == 1000 + assert payload["cost"]["output_msats"] == 0 + assert payload["cost"]["total_msats"] == 1000 + + +@pytest.mark.integration +@pytest.mark.asyncio +async def test_decisions_without_capable_provider_is_rejected( + authenticated_client: AsyncClient, + luna_on_generic_only: None, +) -> None: + with patch("httpx.AsyncHTTPTransport.handle_async_request") as transport: + response = await authenticated_client.post( + "/v1/decisions", json=DECISIONS_REQUEST + ) + + assert response.status_code == 400, response.text + assert "Decisions" in response.json()["error"]["message"] + transport.assert_not_called() diff --git a/tests/unit/test_openai_decisions.py b/tests/unit/test_openai_decisions.py new file mode 100644 index 00000000..eb8093e8 --- /dev/null +++ b/tests/unit/test_openai_decisions.py @@ -0,0 +1,46 @@ +from __future__ import annotations + +import os + +os.environ.setdefault("UPSTREAM_BASE_URL", "http://test") +os.environ.setdefault("UPSTREAM_API_KEY", "test") + +import pytest # noqa: E402 + +from routstr.proxy import _forwarding_allowed # noqa: E402 +from routstr.upstream.base import ( # noqa: E402 + BaseUpstreamProvider, + _x_cashu_path_has_settlement_handler, +) +from routstr.upstream.openai import OpenAIUpstreamProvider # noqa: E402 +from routstr.upstream.openrouter import OpenRouterUpstreamProvider # noqa: E402 + + +@pytest.mark.parametrize("path", ["v1/decisions", "decisions", "v1/decisions/"]) +def test_decisions_is_forwarded(path: str) -> None: + assert _forwarding_allowed(path, "POST") is True + + +@pytest.mark.parametrize( + ("path", "method"), + [ + ("v1/decisions", "GET"), + ("v1/decisionsdump", "POST"), + ("v1/decisions/dec_123", "POST"), + ("v1/alpha/decisions", "POST"), + ("v1/decisions/../admin", "POST"), + ], +) +def test_decisions_lookalikes_are_refused(path: str, method: str) -> None: + assert _forwarding_allowed(path, method) is False + + +def test_decisions_has_x_cashu_settlement_handler() -> None: + assert _x_cashu_path_has_settlement_handler("v1/decisions") is True + assert _x_cashu_path_has_settlement_handler("decisions/") is True + + +def test_only_openai_supports_decisions() -> None: + assert OpenAIUpstreamProvider.supports_decisions is True + assert BaseUpstreamProvider.supports_decisions is False + assert OpenRouterUpstreamProvider.supports_decisions is False