mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
CORE-UPSTREAM-5XX-NOT-NODE-DOWN
An upstream-attributable failure (provider 5xx, EHBP timeout, transport error
after a Cashu token was redeemed) was forwarded to the caller as the
provider's own 5xx. Callers read that as *this node* being down: they marked
the node unhealthy, dropped it from rotation, or refused to retry an upstream
blip that the node had already retried across candidates.
The node is healthy in those cases — it accepted the request, authenticated
it, reserved payment, tried every candidate, and reverted the reservation.
That is now visible from the response alone:
status 424 (UPSTREAM_ERROR_STATUS)
error.code UPSTREAM_UNAVAILABLE
header X-Routstr-Error-Scope: upstream
error.upstream_status / error.details.upstream_status
the provider's own status, preserved
Applied in routstr/upstream/base.py (forward_upstream_error_response and the
post-redemption x-cashu paths for both chat-completions and responses),
routstr/payment/helpers.py (create_error_response / create_upstream_error_response),
routstr/proxy.py (424 is retryable across candidates on the bearer, EHBP and
unauthenticated-GET loops), routstr/upstream/ehbp.py and
routstr/upstream/tinfoil.py (attestation host). New module
routstr/core/error_scope.py holds the contract constants and the mapping.
Deliberately unchanged:
* rate limits keep 429 + UPSTREAM_RATE_LIMIT (including a rate limit wrapped
in a provider 5xx envelope: status 429, code unchanged) — the retry hint is
worth more than the status class;
* provider-side 4xx passes through unchanged;
* genuine node faults stay 500 and carry no scope header (UpstreamError gained
scope=, set to "node" on internal-exception paths) so "node broken" is still
distinguishable from "upstream broken".
The new status is exported through CORS (x-routstr-error-scope) so browser
clients can read the attribution.
Tests: 15 stale assertions of the old 5xx contract updated (renames keep their
intent: refund still happens, bodies still redacted, pinned requests still do
not fall back), plus new acceptance coverage for 424 + scope header +
upstream_status on the provider, bearer, x-cashu, /v1/messages, EHBP and
unauthenticated-GET paths, node faults staying 500 with no header, and failover
past a 424 to a healthy candidate returning 200.
Docs: docs/api/errors.md (status table, new "Upstream attribution" section,
upstream error examples, retry list now includes 424), docs/api/overview.md
and docs/api/endpoints.md.
Full unit suite: 1649 passed, 1 skipped.
150 lines
5.1 KiB
Python
150 lines
5.1 KiB
Python
from __future__ import annotations
|
|
|
|
import json
|
|
from unittest.mock import AsyncMock, MagicMock
|
|
|
|
import pytest
|
|
|
|
from routstr.core.error_scope import (
|
|
ERROR_SCOPE_HEADER,
|
|
ERROR_SCOPE_UPSTREAM,
|
|
UPSTREAM_ERROR_STATUS,
|
|
)
|
|
from routstr.core.exceptions import EhbpTimeoutError, UpstreamError
|
|
from routstr.upstream import ehbp as ehbp_module
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# forward_ehbp_x_cashu_request — timeout fails closed with a refund + 424
|
|
#
|
|
# The EHBP hop is an upstream: a timeout there must not be reported as a 5xx
|
|
# (which would read as *this node* being down). See
|
|
# CORE-UPSTREAM-5XX-NOT-NODE-DOWN.
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
async def _request() -> MagicMock:
|
|
request = MagicMock()
|
|
request.state.request_id = "req-123"
|
|
request.method = "POST"
|
|
request.query_params = {}
|
|
request.headers = {}
|
|
request.body = AsyncMock(return_value=b"opaque")
|
|
return request
|
|
|
|
|
|
def _ehbp_upstream_mocks() -> tuple[MagicMock, MagicMock]:
|
|
"""Upstream and model mocks sufficient to reach the forwarding call."""
|
|
profile = MagicMock()
|
|
profile.client_target_url_header = None
|
|
profile.allow_client_target_override = False
|
|
profile.proxy_only_headers = frozenset()
|
|
profile.usage_response_header = None
|
|
|
|
target = MagicMock()
|
|
target.url = "https://inference.tinfoil.sh/v1/chat/completions"
|
|
target.headers = {}
|
|
target.profile = None
|
|
|
|
upstream = MagicMock()
|
|
upstream.prepare_headers.return_value = {}
|
|
upstream.get_ehbp_forwarding_target.return_value = target
|
|
upstream.get_confidential_inference_profile.return_value = profile
|
|
upstream.prepare_params.return_value = {}
|
|
|
|
model_obj = MagicMock()
|
|
model_obj.id = "tinfoil-kimi-k2-6"
|
|
model_obj.forwarded_model_id = "kimi-k2-6"
|
|
return upstream, model_obj
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_x_cashu_timeout_refunds_and_returns_424(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
monkeypatch.setattr(
|
|
ehbp_module,
|
|
"recieve_token",
|
|
AsyncMock(return_value=(1000, "msat", None)),
|
|
)
|
|
monkeypatch.setattr(
|
|
ehbp_module, "store_cashu_transaction", AsyncMock(return_value=None)
|
|
)
|
|
send_cashu_refund_mock = AsyncMock(return_value="refund-token")
|
|
monkeypatch.setattr(ehbp_module, "send_cashu_refund", send_cashu_refund_mock)
|
|
monkeypatch.setattr(
|
|
ehbp_module,
|
|
"forward_with_trailer",
|
|
AsyncMock(side_effect=EhbpTimeoutError("EHBP upstream timed out")),
|
|
)
|
|
|
|
upstream, model_obj = _ehbp_upstream_mocks()
|
|
|
|
response = await ehbp_module.forward_ehbp_x_cashu_request(
|
|
request=await _request(),
|
|
x_cashu_token="cashu-token",
|
|
path="v1/chat/completions",
|
|
max_cost_for_model=5000,
|
|
model_obj=model_obj,
|
|
upstream=upstream,
|
|
)
|
|
|
|
# 424, not 504: the timeout belongs to the upstream hop, and a 5xx would
|
|
# tell the caller this node is broken.
|
|
assert response.status_code == UPSTREAM_ERROR_STATUS
|
|
assert response.headers[ERROR_SCOPE_HEADER] == ERROR_SCOPE_UPSTREAM
|
|
assert response.headers["X-Cashu"] == "refund-token"
|
|
body = json.loads(bytes(response.body))
|
|
assert body["error"]["type"] == "upstream_timeout"
|
|
assert body["error"]["code"] == "UPSTREAM_TIMEOUT"
|
|
send_cashu_refund_mock.assert_awaited_once_with(1000, "msat", None, "req-123")
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# forward_ehbp_request — the bearer path must let the timeout through, so
|
|
# proxy.py can answer 424 instead of flattening it to a generic 500
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_bearer_timeout_propagates_424(
|
|
monkeypatch: pytest.MonkeyPatch,
|
|
) -> None:
|
|
"""A timed-out bearer request must not be rewritten to a 500.
|
|
|
|
``forward_ehbp_request`` ends in a bare ``except Exception`` that turns any
|
|
error into ``UpstreamError(..., status_code=500)``. The ``except
|
|
UpstreamError: raise`` above it is the only thing preserving the upstream
|
|
timeout status that ``proxy.py`` returns to the client, so this test pins
|
|
that handler.
|
|
"""
|
|
monkeypatch.setattr(
|
|
ehbp_module,
|
|
"forward_with_trailer",
|
|
AsyncMock(
|
|
side_effect=EhbpTimeoutError(
|
|
"EHBP upstream inference.tinfoil.sh timed out after 600s connecting"
|
|
)
|
|
),
|
|
)
|
|
upstream, model_obj = _ehbp_upstream_mocks()
|
|
key = MagicMock()
|
|
key.hashed_key = "abcdef1234567890"
|
|
|
|
with pytest.raises(EhbpTimeoutError) as exc_info:
|
|
await ehbp_module.forward_ehbp_request(
|
|
request=await _request(),
|
|
path="v1/chat/completions",
|
|
headers={},
|
|
request_body=b"opaque",
|
|
upstream=upstream,
|
|
key=key,
|
|
max_cost_for_model=5000,
|
|
session=MagicMock(),
|
|
model_obj=model_obj,
|
|
)
|
|
|
|
assert exc_info.value.status_code == UPSTREAM_ERROR_STATUS
|
|
assert exc_info.value.code == "UPSTREAM_TIMEOUT"
|
|
assert exc_info.value.scope == ERROR_SCOPE_UPSTREAM
|
|
assert isinstance(exc_info.value, UpstreamError)
|