feat: proxy OpenAI Decisions API at /v1/decisions

This commit is contained in:
9qeklajc
2026-09-30 00:13:40 +02:00
parent 325d9beb44
commit 9ff8b01d0b
7 changed files with 261 additions and 1 deletions
+22
View File
@@ -280,6 +280,28 @@ Billing is input-token based (output tokens are free on Jev); the response's
published rate ($0.042 per million input tokens, output free). Override the
model row if TypeSafe changes pricing.
## Decisions (OpenAI)
### Create Decision
Ask OpenAI's Decisions API (powered by `gpt-6-luna`) to pick from a finite set
of answers for text or image context. The API is in limited preview: accounts
that are not enrolled get `403 Decision API is not enabled for this user`.
```http
POST /v1/decisions
```
The request body is forwarded to `https://api.openai.com/v1/decisions`
unchanged; Routstr only reads the top-level `model` to route and price the
request, and settles from the response's `usage` like embeddings.
**Notes:**
- Only the `openai` provider serves this endpoint. A model that no OpenAI
provider on the node offers returns `400 unsupported_request`.
- Pricing uses the node's catalog rate for the requested model.
## Images (Coming Soon)
### Create Image
+1
View File
@@ -105,6 +105,7 @@ Standard OpenAI-compatible endpoints:
- **Chat Completions**: `/v1/chat/completions`
- **Embeddings**: `/v1/embeddings`
- **System One**: `/v1/systemone` (TypeSafe decision models)
- **Decisions**: `/v1/decisions` (OpenAI Decisions API)
- **Completions**: `/v1/completions` *(planned)*
- **Images**: `/v1/images/generations` *(planned)*
- **Audio**: `/v1/audio/transcriptions` *(planned)*
+15
View File
@@ -276,6 +276,7 @@ _ALLOWED_ENDPOINTS: dict[str, frozenset[str]] = {
# -> {answers, usage}. Non-streaming, JSON in/out; billed from the
# response's usage exactly like embeddings.
"systemone": frozenset({"POST"}),
"decisions": frozenset({"POST"}),
"models": frozenset({"GET"}),
"attestation": frozenset({"GET"}),
"tee/attestation": frozenset({"GET"}),
@@ -695,6 +696,20 @@ async def _proxy(
request=request,
)
if _canonical_api_path(path) == "decisions":
candidates = [
(model, upstream)
for model, upstream in candidates
if upstream.supports_decisions
]
if not candidates:
return create_error_response(
"unsupported_request",
f"No Decisions-capable provider found for model '{model_id}'",
400,
request=request,
)
# Reserve/max-cost checks use the best-ranked candidate; the failover loop
# below rebinds (model_obj, upstream) per candidate so forwarding and
# settlement always use the model of the provider actually being tried.
+3 -1
View File
@@ -311,7 +311,7 @@ def _openai_completion_path(path: str) -> str | None:
def _x_cashu_path_has_settlement_handler(path: str) -> bool:
canonical = path.rstrip("/")
return _openai_completion_path(canonical) is not None or canonical.endswith(
("embeddings", "messages", "messages/count_tokens", "systemone")
("embeddings", "messages", "messages/count_tokens", "systemone", "decisions")
)
@@ -334,6 +334,7 @@ class BaseUpstreamProvider:
platform_url: str | None = None
supports_anthropic_messages: bool = False
supports_decisions: bool = False
# When None, the prefix is detected from `base_url` at dispatch time
# (see `get_litellm_provider_prefix`). Subclasses set this to lock the
# provider regardless of URL.
@@ -3308,6 +3309,7 @@ class BaseUpstreamProvider:
or path.endswith("messages")
or path.endswith("messages/count_tokens")
or path.endswith("systemone")
or path.endswith("decisions")
):
if path.endswith("messages"):
client_wants_streaming = False
+1
View File
@@ -13,6 +13,7 @@ class OpenAIUpstreamProvider(BaseUpstreamProvider):
provider_type = "openai"
default_base_url = "https://api.openai.com/v1"
platform_url = "https://platform.openai.com/api-keys"
supports_decisions = True
def __init__(self, api_key: str, provider_fee: float = 1.01):
super().__init__(
+173
View File
@@ -0,0 +1,173 @@
"""Integration test: POST /v1/decisions routed to OpenAI and billed from usage."""
import json
from typing import Any, AsyncGenerator
from unittest.mock import patch
import httpx
import pytest
from httpx import AsyncClient
from routstr.payment.models import Architecture, Model, Pricing
from routstr.proxy import refresh_model_maps
from routstr.upstream.base import BaseUpstreamProvider
from routstr.upstream.openai import OpenAIUpstreamProvider
DECISIONS_REQUEST = {
"model": "gpt-6-luna",
"state": {"ticket": "Checkout shows a blank page after I click Pay."},
"questions": {
"team": {
"type": "choice",
"instructions": "Which team should own this ticket?",
"criteria": {"payments": "Checkout and billing", "account": "Login"},
}
},
}
DECISIONS_RESPONSE = {
"model": "gpt-6-luna",
"answers": {"team": {"type": "choice", "choice": "payments"}},
"usage": {"input_tokens": 1000, "output_tokens": 200},
}
def _luna_model(prompt_sats: float) -> Model:
return Model(
id="gpt-6-luna",
name="gpt-6-luna",
created=1,
description="GPT-6 Luna",
context_length=400_000,
architecture=Architecture(
modality="text+image->text",
input_modalities=["text", "image"],
output_modalities=["text"],
tokenizer="GPT",
instruct_type=None,
),
pricing=Pricing(prompt=prompt_sats, completion=0.0, max_cost=50.0),
sats_pricing=Pricing(prompt=prompt_sats, completion=0.0, max_cost=50.0),
)
class _StaticOpenAIProvider(OpenAIUpstreamProvider):
def __init__(self, model: Model) -> None:
super().__init__("key-openai", 1.0)
self._static_model = model
def get_cached_models(self) -> list[Model]:
return [self._static_model]
async def refresh_models_cache(self) -> None:
pass
class _StaticGenericProvider(BaseUpstreamProvider):
def __init__(self, model: Model) -> None:
super().__init__("https://generic.test/v1", "key-generic", 1.0)
self._static_model = model
def get_cached_models(self) -> list[Model]:
return [self._static_model]
async def refresh_models_cache(self) -> None:
pass
async def _install(
upstreams: list[BaseUpstreamProvider],
) -> AsyncGenerator[None, None]:
from routstr import proxy
original_upstreams = proxy.get_upstreams()
with patch("routstr.proxy._upstreams", upstreams):
await refresh_model_maps()
yield
with patch("routstr.proxy._upstreams", original_upstreams):
await refresh_model_maps()
@pytest.fixture
async def luna_on_openai_and_generic(
patched_db_engine: None,
) -> AsyncGenerator[None, None]:
async for _ in _install(
[
_StaticGenericProvider(_luna_model(0.0005)),
_StaticOpenAIProvider(_luna_model(0.001)),
]
):
yield
@pytest.fixture
async def luna_on_generic_only(
patched_db_engine: None,
) -> AsyncGenerator[None, None]:
async for _ in _install([_StaticGenericProvider(_luna_model(0.0005))]):
yield
@pytest.mark.integration
@pytest.mark.asyncio
async def test_decisions_routed_to_openai_and_billed_from_usage(
authenticated_client: AsyncClient,
luna_on_openai_and_generic: None,
) -> None:
sent_requests: list[httpx.Request] = []
async def fake_transport(
request: httpx.Request, *args: Any, **kwargs: Any
) -> httpx.Response:
sent_requests.append(request)
return httpx.Response(
200,
content=json.dumps(DECISIONS_RESPONSE).encode(),
headers={"content-type": "application/json"},
)
with (
patch(
"httpx.AsyncHTTPTransport.handle_async_request",
side_effect=fake_transport,
),
patch(
"routstr.payment.cost_calculation.sats_usd_price",
return_value=0.0005,
),
):
response = await authenticated_client.post(
"/v1/decisions", json=DECISIONS_REQUEST
)
assert response.status_code == 200, response.text
payload = response.json()
assert payload["answers"]["team"]["choice"] == "payments"
assert len(sent_requests) == 1
assert str(sent_requests[0].url) == "https://api.openai.com/v1/decisions"
assert (
json.loads(sent_requests[0].content)["questions"]
== (DECISIONS_REQUEST["questions"])
)
assert payload["cost"]["input_msats"] == 1000
assert payload["cost"]["output_msats"] == 0
assert payload["cost"]["total_msats"] == 1000
@pytest.mark.integration
@pytest.mark.asyncio
async def test_decisions_without_capable_provider_is_rejected(
authenticated_client: AsyncClient,
luna_on_generic_only: None,
) -> None:
with patch("httpx.AsyncHTTPTransport.handle_async_request") as transport:
response = await authenticated_client.post(
"/v1/decisions", json=DECISIONS_REQUEST
)
assert response.status_code == 400, response.text
assert "Decisions" in response.json()["error"]["message"]
transport.assert_not_called()
+46
View File
@@ -0,0 +1,46 @@
from __future__ import annotations
import os
os.environ.setdefault("UPSTREAM_BASE_URL", "http://test")
os.environ.setdefault("UPSTREAM_API_KEY", "test")
import pytest # noqa: E402
from routstr.proxy import _forwarding_allowed # noqa: E402
from routstr.upstream.base import ( # noqa: E402
BaseUpstreamProvider,
_x_cashu_path_has_settlement_handler,
)
from routstr.upstream.openai import OpenAIUpstreamProvider # noqa: E402
from routstr.upstream.openrouter import OpenRouterUpstreamProvider # noqa: E402
@pytest.mark.parametrize("path", ["v1/decisions", "decisions", "v1/decisions/"])
def test_decisions_is_forwarded(path: str) -> None:
assert _forwarding_allowed(path, "POST") is True
@pytest.mark.parametrize(
("path", "method"),
[
("v1/decisions", "GET"),
("v1/decisionsdump", "POST"),
("v1/decisions/dec_123", "POST"),
("v1/alpha/decisions", "POST"),
("v1/decisions/../admin", "POST"),
],
)
def test_decisions_lookalikes_are_refused(path: str, method: str) -> None:
assert _forwarding_allowed(path, method) is False
def test_decisions_has_x_cashu_settlement_handler() -> None:
assert _x_cashu_path_has_settlement_handler("v1/decisions") is True
assert _x_cashu_path_has_settlement_handler("decisions/") is True
def test_only_openai_supports_decisions() -> None:
assert OpenAIUpstreamProvider.supports_decisions is True
assert BaseUpstreamProvider.supports_decisions is False
assert OpenRouterUpstreamProvider.supports_decisions is False