mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
fix: disable upstream stream timeouts by default
This commit is contained in:
@@ -72,6 +72,12 @@ ROUTSTR_SECRET_KEY=
|
||||
# UPSTREAM_POOL_TIMEOUT=5
|
||||
# UPSTREAM_READ_TIMEOUT=900
|
||||
|
||||
# Upstream Streaming Guards (0 disables; keep above reasoning models' think time)
|
||||
# UPSTREAM_FIRST_TOKEN_TIMEOUT_SECONDS=0
|
||||
# UPSTREAM_STREAM_IDLE_TIMEOUT_SECONDS=0
|
||||
# UPSTREAM_ALLOWED_FAILS=3
|
||||
# UPSTREAM_COOLDOWN_SECONDS=30
|
||||
|
||||
# Logging
|
||||
# LOG_LEVEL=INFO
|
||||
# ENABLE_CONSOLE_LOGGING=true
|
||||
|
||||
@@ -40,14 +40,15 @@ class Settings(BaseSettings):
|
||||
upstream_5xx_retry_attempts: int = Field(
|
||||
default=1, ge=0, env="UPSTREAM_5XX_RETRY_ATTEMPTS"
|
||||
)
|
||||
# Streaming guards, both disabled by 0. A stream that never produces a
|
||||
# Streaming guards, off by default (0). A stream that never produces a
|
||||
# first chunk can still fail over; one that stalls later can only be
|
||||
# aborted and billed for what it delivered.
|
||||
# aborted and billed for what it delivered. Reasoning models can stay
|
||||
# silent for minutes, so set these above the longest expected think time.
|
||||
upstream_first_token_timeout_seconds: float = Field(
|
||||
default=60.0, ge=0, env="UPSTREAM_FIRST_TOKEN_TIMEOUT_SECONDS"
|
||||
default=0.0, ge=0, env="UPSTREAM_FIRST_TOKEN_TIMEOUT_SECONDS"
|
||||
)
|
||||
upstream_stream_idle_timeout_seconds: float = Field(
|
||||
default=120.0, ge=0, env="UPSTREAM_STREAM_IDLE_TIMEOUT_SECONDS"
|
||||
default=0.0, ge=0, env="UPSTREAM_STREAM_IDLE_TIMEOUT_SECONDS"
|
||||
)
|
||||
# Circuit breaker: timeouts/5xx per (provider, model) within a minute that
|
||||
# take the pair out of candidate selection. 0 seconds disables it.
|
||||
|
||||
@@ -15,7 +15,7 @@ from routstr.core.error_scope import (
|
||||
ERROR_SCOPE_UPSTREAM,
|
||||
)
|
||||
from routstr.core.exceptions import UpstreamError
|
||||
from routstr.core.settings import settings
|
||||
from routstr.core.settings import Settings, settings
|
||||
from routstr.upstream.base import BaseUpstreamProvider
|
||||
from routstr.upstream.cooldown import is_cooling_down, record_failure
|
||||
from routstr.upstream.stream_timeout import open_guarded_stream
|
||||
@@ -149,15 +149,12 @@ async def test_zero_first_token_timeout_disables_the_guard(
|
||||
assert [chunk async for chunk in stream] == [b"first"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_idle_timeout_ends_the_stream_without_raising(
|
||||
fast_timeouts: None,
|
||||
) -> None:
|
||||
stream = await open_guarded_stream(_response(_stalls_after_first()), "test")
|
||||
|
||||
# The stalled stream ends after the delivered bytes; the caller's finalizer
|
||||
# then settles actual usage instead of the request hanging.
|
||||
assert [chunk async for chunk in stream] == [b"first"]
|
||||
def test_stream_guards_are_off_by_default() -> None:
|
||||
# Reasoning models can think silently for minutes; on by default, the
|
||||
# guards would fail requests that succeed without them.
|
||||
fields = Settings.__fields__
|
||||
assert fields["upstream_first_token_timeout_seconds"].default == 0
|
||||
assert fields["upstream_stream_idle_timeout_seconds"].default == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
|
||||
Reference in New Issue
Block a user