mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
feat: proxy OpenAI Decisions API at /v1/decisions
This commit is contained in:
@@ -280,6 +280,28 @@ Billing is input-token based (output tokens are free on Jev); the response's
|
||||
published rate ($0.042 per million input tokens, output free). Override the
|
||||
model row if TypeSafe changes pricing.
|
||||
|
||||
## Decisions (OpenAI)
|
||||
|
||||
### Create Decision
|
||||
|
||||
Ask OpenAI's Decisions API (powered by `gpt-6-luna`) to pick from a finite set
|
||||
of answers for text or image context. The API is in limited preview: accounts
|
||||
that are not enrolled get `403 Decision API is not enabled for this user`.
|
||||
|
||||
```http
|
||||
POST /v1/decisions
|
||||
```
|
||||
|
||||
The request body is forwarded to `https://api.openai.com/v1/decisions`
|
||||
unchanged; Routstr only reads the top-level `model` to route and price the
|
||||
request, and settles from the response's `usage` like embeddings.
|
||||
|
||||
**Notes:**
|
||||
|
||||
- Only the `openai` provider serves this endpoint. A model that no OpenAI
|
||||
provider on the node offers returns `400 unsupported_request`.
|
||||
- Pricing uses the node's catalog rate for the requested model.
|
||||
|
||||
## Images (Coming Soon)
|
||||
|
||||
### Create Image
|
||||
|
||||
@@ -105,6 +105,7 @@ Standard OpenAI-compatible endpoints:
|
||||
- **Chat Completions**: `/v1/chat/completions`
|
||||
- **Embeddings**: `/v1/embeddings`
|
||||
- **System One**: `/v1/systemone` (TypeSafe decision models)
|
||||
- **Decisions**: `/v1/decisions` (OpenAI Decisions API)
|
||||
- **Completions**: `/v1/completions` *(planned)*
|
||||
- **Images**: `/v1/images/generations` *(planned)*
|
||||
- **Audio**: `/v1/audio/transcriptions` *(planned)*
|
||||
|
||||
@@ -276,6 +276,7 @@ _ALLOWED_ENDPOINTS: dict[str, frozenset[str]] = {
|
||||
# -> {answers, usage}. Non-streaming, JSON in/out; billed from the
|
||||
# response's usage exactly like embeddings.
|
||||
"systemone": frozenset({"POST"}),
|
||||
"decisions": frozenset({"POST"}),
|
||||
"models": frozenset({"GET"}),
|
||||
"attestation": frozenset({"GET"}),
|
||||
"tee/attestation": frozenset({"GET"}),
|
||||
@@ -695,6 +696,20 @@ async def _proxy(
|
||||
request=request,
|
||||
)
|
||||
|
||||
if _canonical_api_path(path) == "decisions":
|
||||
candidates = [
|
||||
(model, upstream)
|
||||
for model, upstream in candidates
|
||||
if upstream.supports_decisions
|
||||
]
|
||||
if not candidates:
|
||||
return create_error_response(
|
||||
"unsupported_request",
|
||||
f"No Decisions-capable provider found for model '{model_id}'",
|
||||
400,
|
||||
request=request,
|
||||
)
|
||||
|
||||
# Reserve/max-cost checks use the best-ranked candidate; the failover loop
|
||||
# below rebinds (model_obj, upstream) per candidate so forwarding and
|
||||
# settlement always use the model of the provider actually being tried.
|
||||
|
||||
@@ -311,7 +311,7 @@ def _openai_completion_path(path: str) -> str | None:
|
||||
def _x_cashu_path_has_settlement_handler(path: str) -> bool:
|
||||
canonical = path.rstrip("/")
|
||||
return _openai_completion_path(canonical) is not None or canonical.endswith(
|
||||
("embeddings", "messages", "messages/count_tokens", "systemone")
|
||||
("embeddings", "messages", "messages/count_tokens", "systemone", "decisions")
|
||||
)
|
||||
|
||||
|
||||
@@ -334,6 +334,7 @@ class BaseUpstreamProvider:
|
||||
platform_url: str | None = None
|
||||
|
||||
supports_anthropic_messages: bool = False
|
||||
supports_decisions: bool = False
|
||||
# When None, the prefix is detected from `base_url` at dispatch time
|
||||
# (see `get_litellm_provider_prefix`). Subclasses set this to lock the
|
||||
# provider regardless of URL.
|
||||
@@ -3308,6 +3309,7 @@ class BaseUpstreamProvider:
|
||||
or path.endswith("messages")
|
||||
or path.endswith("messages/count_tokens")
|
||||
or path.endswith("systemone")
|
||||
or path.endswith("decisions")
|
||||
):
|
||||
if path.endswith("messages"):
|
||||
client_wants_streaming = False
|
||||
|
||||
@@ -13,6 +13,7 @@ class OpenAIUpstreamProvider(BaseUpstreamProvider):
|
||||
provider_type = "openai"
|
||||
default_base_url = "https://api.openai.com/v1"
|
||||
platform_url = "https://platform.openai.com/api-keys"
|
||||
supports_decisions = True
|
||||
|
||||
def __init__(self, api_key: str, provider_fee: float = 1.01):
|
||||
super().__init__(
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
"""Integration test: POST /v1/decisions routed to OpenAI and billed from usage."""
|
||||
|
||||
import json
|
||||
from typing import Any, AsyncGenerator
|
||||
from unittest.mock import patch
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
from httpx import AsyncClient
|
||||
|
||||
from routstr.payment.models import Architecture, Model, Pricing
|
||||
from routstr.proxy import refresh_model_maps
|
||||
from routstr.upstream.base import BaseUpstreamProvider
|
||||
from routstr.upstream.openai import OpenAIUpstreamProvider
|
||||
|
||||
DECISIONS_REQUEST = {
|
||||
"model": "gpt-6-luna",
|
||||
"state": {"ticket": "Checkout shows a blank page after I click Pay."},
|
||||
"questions": {
|
||||
"team": {
|
||||
"type": "choice",
|
||||
"instructions": "Which team should own this ticket?",
|
||||
"criteria": {"payments": "Checkout and billing", "account": "Login"},
|
||||
}
|
||||
},
|
||||
}
|
||||
|
||||
DECISIONS_RESPONSE = {
|
||||
"model": "gpt-6-luna",
|
||||
"answers": {"team": {"type": "choice", "choice": "payments"}},
|
||||
"usage": {"input_tokens": 1000, "output_tokens": 200},
|
||||
}
|
||||
|
||||
|
||||
def _luna_model(prompt_sats: float) -> Model:
|
||||
return Model(
|
||||
id="gpt-6-luna",
|
||||
name="gpt-6-luna",
|
||||
created=1,
|
||||
description="GPT-6 Luna",
|
||||
context_length=400_000,
|
||||
architecture=Architecture(
|
||||
modality="text+image->text",
|
||||
input_modalities=["text", "image"],
|
||||
output_modalities=["text"],
|
||||
tokenizer="GPT",
|
||||
instruct_type=None,
|
||||
),
|
||||
pricing=Pricing(prompt=prompt_sats, completion=0.0, max_cost=50.0),
|
||||
sats_pricing=Pricing(prompt=prompt_sats, completion=0.0, max_cost=50.0),
|
||||
)
|
||||
|
||||
|
||||
class _StaticOpenAIProvider(OpenAIUpstreamProvider):
|
||||
def __init__(self, model: Model) -> None:
|
||||
super().__init__("key-openai", 1.0)
|
||||
self._static_model = model
|
||||
|
||||
def get_cached_models(self) -> list[Model]:
|
||||
return [self._static_model]
|
||||
|
||||
async def refresh_models_cache(self) -> None:
|
||||
pass
|
||||
|
||||
|
||||
class _StaticGenericProvider(BaseUpstreamProvider):
|
||||
def __init__(self, model: Model) -> None:
|
||||
super().__init__("https://generic.test/v1", "key-generic", 1.0)
|
||||
self._static_model = model
|
||||
|
||||
def get_cached_models(self) -> list[Model]:
|
||||
return [self._static_model]
|
||||
|
||||
async def refresh_models_cache(self) -> None:
|
||||
pass
|
||||
|
||||
|
||||
async def _install(
|
||||
upstreams: list[BaseUpstreamProvider],
|
||||
) -> AsyncGenerator[None, None]:
|
||||
from routstr import proxy
|
||||
|
||||
original_upstreams = proxy.get_upstreams()
|
||||
with patch("routstr.proxy._upstreams", upstreams):
|
||||
await refresh_model_maps()
|
||||
yield
|
||||
with patch("routstr.proxy._upstreams", original_upstreams):
|
||||
await refresh_model_maps()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def luna_on_openai_and_generic(
|
||||
patched_db_engine: None,
|
||||
) -> AsyncGenerator[None, None]:
|
||||
async for _ in _install(
|
||||
[
|
||||
_StaticGenericProvider(_luna_model(0.0005)),
|
||||
_StaticOpenAIProvider(_luna_model(0.001)),
|
||||
]
|
||||
):
|
||||
yield
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
async def luna_on_generic_only(
|
||||
patched_db_engine: None,
|
||||
) -> AsyncGenerator[None, None]:
|
||||
async for _ in _install([_StaticGenericProvider(_luna_model(0.0005))]):
|
||||
yield
|
||||
|
||||
|
||||
@pytest.mark.integration
|
||||
@pytest.mark.asyncio
|
||||
async def test_decisions_routed_to_openai_and_billed_from_usage(
|
||||
authenticated_client: AsyncClient,
|
||||
luna_on_openai_and_generic: None,
|
||||
) -> None:
|
||||
sent_requests: list[httpx.Request] = []
|
||||
|
||||
async def fake_transport(
|
||||
request: httpx.Request, *args: Any, **kwargs: Any
|
||||
) -> httpx.Response:
|
||||
sent_requests.append(request)
|
||||
return httpx.Response(
|
||||
200,
|
||||
content=json.dumps(DECISIONS_RESPONSE).encode(),
|
||||
headers={"content-type": "application/json"},
|
||||
)
|
||||
|
||||
with (
|
||||
patch(
|
||||
"httpx.AsyncHTTPTransport.handle_async_request",
|
||||
side_effect=fake_transport,
|
||||
),
|
||||
patch(
|
||||
"routstr.payment.cost_calculation.sats_usd_price",
|
||||
return_value=0.0005,
|
||||
),
|
||||
):
|
||||
response = await authenticated_client.post(
|
||||
"/v1/decisions", json=DECISIONS_REQUEST
|
||||
)
|
||||
|
||||
assert response.status_code == 200, response.text
|
||||
payload = response.json()
|
||||
assert payload["answers"]["team"]["choice"] == "payments"
|
||||
|
||||
assert len(sent_requests) == 1
|
||||
assert str(sent_requests[0].url) == "https://api.openai.com/v1/decisions"
|
||||
assert (
|
||||
json.loads(sent_requests[0].content)["questions"]
|
||||
== (DECISIONS_REQUEST["questions"])
|
||||
)
|
||||
|
||||
assert payload["cost"]["input_msats"] == 1000
|
||||
assert payload["cost"]["output_msats"] == 0
|
||||
assert payload["cost"]["total_msats"] == 1000
|
||||
|
||||
|
||||
@pytest.mark.integration
|
||||
@pytest.mark.asyncio
|
||||
async def test_decisions_without_capable_provider_is_rejected(
|
||||
authenticated_client: AsyncClient,
|
||||
luna_on_generic_only: None,
|
||||
) -> None:
|
||||
with patch("httpx.AsyncHTTPTransport.handle_async_request") as transport:
|
||||
response = await authenticated_client.post(
|
||||
"/v1/decisions", json=DECISIONS_REQUEST
|
||||
)
|
||||
|
||||
assert response.status_code == 400, response.text
|
||||
assert "Decisions" in response.json()["error"]["message"]
|
||||
transport.assert_not_called()
|
||||
@@ -0,0 +1,46 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
os.environ.setdefault("UPSTREAM_BASE_URL", "http://test")
|
||||
os.environ.setdefault("UPSTREAM_API_KEY", "test")
|
||||
|
||||
import pytest # noqa: E402
|
||||
|
||||
from routstr.proxy import _forwarding_allowed # noqa: E402
|
||||
from routstr.upstream.base import ( # noqa: E402
|
||||
BaseUpstreamProvider,
|
||||
_x_cashu_path_has_settlement_handler,
|
||||
)
|
||||
from routstr.upstream.openai import OpenAIUpstreamProvider # noqa: E402
|
||||
from routstr.upstream.openrouter import OpenRouterUpstreamProvider # noqa: E402
|
||||
|
||||
|
||||
@pytest.mark.parametrize("path", ["v1/decisions", "decisions", "v1/decisions/"])
|
||||
def test_decisions_is_forwarded(path: str) -> None:
|
||||
assert _forwarding_allowed(path, "POST") is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("path", "method"),
|
||||
[
|
||||
("v1/decisions", "GET"),
|
||||
("v1/decisionsdump", "POST"),
|
||||
("v1/decisions/dec_123", "POST"),
|
||||
("v1/alpha/decisions", "POST"),
|
||||
("v1/decisions/../admin", "POST"),
|
||||
],
|
||||
)
|
||||
def test_decisions_lookalikes_are_refused(path: str, method: str) -> None:
|
||||
assert _forwarding_allowed(path, method) is False
|
||||
|
||||
|
||||
def test_decisions_has_x_cashu_settlement_handler() -> None:
|
||||
assert _x_cashu_path_has_settlement_handler("v1/decisions") is True
|
||||
assert _x_cashu_path_has_settlement_handler("decisions/") is True
|
||||
|
||||
|
||||
def test_only_openai_supports_decisions() -> None:
|
||||
assert OpenAIUpstreamProvider.supports_decisions is True
|
||||
assert BaseUpstreamProvider.supports_decisions is False
|
||||
assert OpenRouterUpstreamProvider.supports_decisions is False
|
||||
Reference in New Issue
Block a user