mirror of
https://github.com/Routstr/routstr-core.git
synced 2026-10-05 12:28:22 +00:00
Merge pull request #766 from Routstr/fix/openrouter-models-fetch-retry
fix(models): retry the OpenRouter catalogue fetch before giving up
This commit is contained in:
+87
-50
@@ -241,61 +241,98 @@ def _has_valid_pricing(model: dict) -> bool:
|
||||
return True
|
||||
|
||||
|
||||
async def async_fetch_openrouter_models(source_filter: str | None = None) -> list[dict]:
|
||||
"""Asynchronously fetch model information from OpenRouter API."""
|
||||
# OpenRouter occasionally answers /models with a truncated body, emptying the
|
||||
# catalogue behind one log line. Retry, but keep 3 attempts within roughly the
|
||||
# old single-attempt budget: this fetch blocks startup and the refresh loop.
|
||||
OPENROUTER_MODELS_MAX_ATTEMPTS = 3
|
||||
OPENROUTER_MODELS_TIMEOUT_SECONDS = 10
|
||||
OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS = 0.5
|
||||
|
||||
|
||||
def _is_transient(error: BaseException) -> bool:
|
||||
if isinstance(error, httpx.HTTPStatusError):
|
||||
return error.response.status_code >= 500
|
||||
return True
|
||||
|
||||
|
||||
def _parse_models_response(response: httpx.Response | BaseException) -> list[dict]:
|
||||
if isinstance(response, BaseException):
|
||||
raise response
|
||||
response.raise_for_status()
|
||||
return [
|
||||
model
|
||||
for model in response.json().get("data", [])
|
||||
if ":free" not in model.get("id", "").lower()
|
||||
]
|
||||
|
||||
|
||||
async def _fetch_openrouter_models_once(source_filter: str | None) -> list[dict]:
|
||||
"""One attempt. Raises if /models is unusable; embeddings are best-effort."""
|
||||
base_url = "https://openrouter.ai/api/v1"
|
||||
timeout = OPENROUTER_MODELS_TIMEOUT_SECONDS
|
||||
|
||||
try:
|
||||
async with httpx.AsyncClient() as client:
|
||||
models_response, embeddings_response = await asyncio.gather(
|
||||
client.get(f"{base_url}/models", timeout=30),
|
||||
client.get(f"{base_url}/embeddings/models", timeout=30),
|
||||
return_exceptions=True,
|
||||
)
|
||||
async with httpx.AsyncClient() as client:
|
||||
models_response, embeddings_response = await asyncio.gather(
|
||||
client.get(f"{base_url}/models", timeout=timeout),
|
||||
client.get(f"{base_url}/embeddings/models", timeout=timeout),
|
||||
return_exceptions=True,
|
||||
)
|
||||
|
||||
def process_models_response(
|
||||
response: httpx.Response | BaseException,
|
||||
) -> list[dict]:
|
||||
if not isinstance(response, BaseException):
|
||||
response.raise_for_status()
|
||||
data = response.json()
|
||||
return [
|
||||
model
|
||||
for model in data.get("data", [])
|
||||
if ":free" not in model.get("id", "").lower()
|
||||
]
|
||||
# Losing /models is what empties the node, so it fails the attempt and
|
||||
# the caller retries. A missing embeddings half must not do the same.
|
||||
models_data = _parse_models_response(models_response)
|
||||
try:
|
||||
models_data.extend(_parse_models_response(embeddings_response))
|
||||
except Exception as e:
|
||||
logger.warning(f"Skipping OpenRouter embeddings models: {e}")
|
||||
|
||||
# Apply source filter and exclusions
|
||||
filtered_models = []
|
||||
for model in models_data:
|
||||
model_id = model.get("id", "")
|
||||
|
||||
if source_filter:
|
||||
source_prefix = f"{source_filter}/"
|
||||
if not model_id.startswith(source_prefix):
|
||||
continue
|
||||
|
||||
model = dict(model)
|
||||
model["id"] = model_id[len(source_prefix) :]
|
||||
model_id = model["id"]
|
||||
|
||||
if "(free)" in model.get("name", ""):
|
||||
continue
|
||||
|
||||
if not _has_valid_pricing(model):
|
||||
continue
|
||||
|
||||
filtered_models.append(model)
|
||||
|
||||
return filtered_models
|
||||
|
||||
|
||||
async def async_fetch_openrouter_models(source_filter: str | None = None) -> list[dict]:
|
||||
"""Fetch the OpenRouter catalogue; ``[]`` once every attempt has failed."""
|
||||
for attempt in range(1, OPENROUTER_MODELS_MAX_ATTEMPTS + 1):
|
||||
try:
|
||||
return await _fetch_openrouter_models_once(source_filter)
|
||||
except Exception as e:
|
||||
last_attempt = attempt == OPENROUTER_MODELS_MAX_ATTEMPTS
|
||||
if last_attempt or not _is_transient(e):
|
||||
logger.error(
|
||||
f"Error (async) fetching models from OpenRouter API "
|
||||
f"after {attempt} attempt(s): {e}"
|
||||
)
|
||||
return []
|
||||
logger.warning(
|
||||
f"OpenRouter models fetch attempt {attempt}/"
|
||||
f"{OPENROUTER_MODELS_MAX_ATTEMPTS} failed: {e}; retrying"
|
||||
)
|
||||
# Jittered so nodes do not retry in lockstep.
|
||||
backoff = OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS * attempt
|
||||
await asyncio.sleep(backoff * random.uniform(0.5, 1.5))
|
||||
|
||||
models_data: list[dict] = []
|
||||
models_data.extend(process_models_response(models_response))
|
||||
models_data.extend(process_models_response(embeddings_response))
|
||||
|
||||
# Apply source filter and exclusions
|
||||
filtered_models = []
|
||||
for model in models_data:
|
||||
model_id = model.get("id", "")
|
||||
|
||||
if source_filter:
|
||||
source_prefix = f"{source_filter}/"
|
||||
if not model_id.startswith(source_prefix):
|
||||
continue
|
||||
|
||||
model = dict(model)
|
||||
model["id"] = model_id[len(source_prefix) :]
|
||||
model_id = model["id"]
|
||||
|
||||
if "(free)" in model.get("name", ""):
|
||||
continue
|
||||
|
||||
if not _has_valid_pricing(model):
|
||||
continue
|
||||
|
||||
filtered_models.append(model)
|
||||
|
||||
return filtered_models
|
||||
except Exception as e:
|
||||
logger.error(f"Error (async) fetching models from OpenRouter API: {e}")
|
||||
return []
|
||||
return []
|
||||
|
||||
|
||||
def _build_model_from_row(
|
||||
|
||||
@@ -0,0 +1,186 @@
|
||||
"""Retry behaviour for the OpenRouter catalogue fetch."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, Callable
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from routstr.payment import models as models_module
|
||||
from routstr.payment.models import async_fetch_openrouter_models
|
||||
|
||||
MODELS_URL = "https://openrouter.ai/api/v1/models"
|
||||
EMBEDDINGS_URL = "https://openrouter.ai/api/v1/embeddings/models"
|
||||
|
||||
|
||||
def _model(model_id: str) -> dict[str, Any]:
|
||||
return {
|
||||
"id": model_id,
|
||||
"name": model_id,
|
||||
"pricing": {"prompt": "0.000001", "completion": "0.000002"},
|
||||
}
|
||||
|
||||
|
||||
def _ok_response(url: str, payload: dict[str, Any]) -> httpx.Response:
|
||||
return httpx.Response(
|
||||
200,
|
||||
request=httpx.Request("GET", url),
|
||||
content=json.dumps(payload).encode(),
|
||||
headers={"content-type": "application/json"},
|
||||
)
|
||||
|
||||
|
||||
def _error_response(url: str, status: int) -> httpx.Response:
|
||||
return httpx.Response(status, request=httpx.Request("GET", url), content=b"nope")
|
||||
|
||||
|
||||
def _truncated_response(url: str) -> httpx.Response:
|
||||
"""A body cut mid-JSON — the shape OpenRouter actually sent the node."""
|
||||
return httpx.Response(
|
||||
200,
|
||||
request=httpx.Request("GET", url),
|
||||
content=b'{"data": [{"id": "vendor/model-a", "name": "Model A", "pric',
|
||||
headers={"content-type": "application/json"},
|
||||
)
|
||||
|
||||
|
||||
def _payload_for(url: str) -> dict[str, Any]:
|
||||
if url.endswith("/embeddings/models"):
|
||||
return {"data": [_model("vendor/embed-1")]}
|
||||
return {"data": [_model("vendor/model-a")]}
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _no_retry_backoff(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
monkeypatch.setattr(models_module, "OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS", 0)
|
||||
|
||||
|
||||
def _install_get(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
handler: Callable[[str, int], httpx.Response],
|
||||
) -> dict[str, int]:
|
||||
"""Patch ``httpx.AsyncClient.get`` and count calls per endpoint."""
|
||||
counts: dict[str, int] = {}
|
||||
|
||||
async def fake_get(
|
||||
self: httpx.AsyncClient, url: Any, **kwargs: Any
|
||||
) -> httpx.Response:
|
||||
key = str(url)
|
||||
counts[key] = counts.get(key, 0) + 1
|
||||
return handler(key, counts[key])
|
||||
|
||||
monkeypatch.setattr(httpx.AsyncClient, "get", fake_get)
|
||||
return counts
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_truncated_body_is_retried_and_recovers(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A truncated body on the first attempt must not empty the catalogue."""
|
||||
|
||||
def handler(url: str, call: int) -> httpx.Response:
|
||||
if call == 1:
|
||||
return _truncated_response(url)
|
||||
return _ok_response(url, _payload_for(url))
|
||||
|
||||
counts = _install_get(monkeypatch, handler)
|
||||
|
||||
result = await async_fetch_openrouter_models()
|
||||
|
||||
assert [model["id"] for model in result] == ["vendor/model-a", "vendor/embed-1"]
|
||||
assert counts[MODELS_URL] == 2
|
||||
assert counts[EMBEDDINGS_URL] == 2
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_gives_up_after_max_attempts_and_logs_the_error(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""After every attempt fails: log once and return an empty catalogue."""
|
||||
errors: list[str] = []
|
||||
monkeypatch.setattr(models_module.logger, "error", lambda msg: errors.append(msg))
|
||||
|
||||
counts = _install_get(monkeypatch, lambda url, call: _truncated_response(url))
|
||||
|
||||
result = await async_fetch_openrouter_models()
|
||||
|
||||
assert result == []
|
||||
assert counts[MODELS_URL] == models_module.OPENROUTER_MODELS_MAX_ATTEMPTS
|
||||
assert len(errors) == 1
|
||||
assert "after 3 attempt(s)" in errors[0]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_embeddings_outage_still_yields_the_main_catalogue(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""The secondary endpoint is best-effort: it cannot empty the catalogue."""
|
||||
|
||||
def handler(url: str, call: int) -> httpx.Response:
|
||||
if url == EMBEDDINGS_URL:
|
||||
return _error_response(url, 503)
|
||||
return _ok_response(url, _payload_for(url))
|
||||
|
||||
counts = _install_get(monkeypatch, handler)
|
||||
|
||||
result = await async_fetch_openrouter_models()
|
||||
|
||||
assert [model["id"] for model in result] == ["vendor/model-a"]
|
||||
assert counts[MODELS_URL] == 1
|
||||
assert counts[EMBEDDINGS_URL] == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_main_catalogue_server_error_is_retried(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""A 5xx on /models is transient, so the attempt is retried."""
|
||||
|
||||
def handler(url: str, call: int) -> httpx.Response:
|
||||
if url == MODELS_URL and call == 1:
|
||||
return _error_response(url, 503)
|
||||
return _ok_response(url, _payload_for(url))
|
||||
|
||||
counts = _install_get(monkeypatch, handler)
|
||||
|
||||
result = await async_fetch_openrouter_models()
|
||||
|
||||
assert [model["id"] for model in result] == ["vendor/model-a", "vendor/embed-1"]
|
||||
assert counts[MODELS_URL] == 2
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_client_error_is_not_retried(monkeypatch: pytest.MonkeyPatch) -> None:
|
||||
"""Retrying a 4xx only adds load to an upstream that already said no."""
|
||||
counts = _install_get(monkeypatch, lambda url, call: _error_response(url, 401))
|
||||
|
||||
result = await async_fetch_openrouter_models()
|
||||
|
||||
assert result == []
|
||||
assert counts[MODELS_URL] == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_source_filter_and_free_models_are_still_applied(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
) -> None:
|
||||
"""The moved filter loop still strips prefixes and drops free tiers."""
|
||||
payload = {
|
||||
"data": [
|
||||
_model("openai/gpt-x"),
|
||||
_model("openai/gpt-x:free"),
|
||||
_model("other/y"),
|
||||
]
|
||||
}
|
||||
|
||||
def handler(url: str, call: int) -> httpx.Response:
|
||||
return _ok_response(url, payload if url == MODELS_URL else {"data": []})
|
||||
|
||||
_install_get(monkeypatch, handler)
|
||||
|
||||
result = await async_fetch_openrouter_models(source_filter="openai")
|
||||
|
||||
assert [model["id"] for model in result] == ["gpt-x"]
|
||||
Reference in New Issue
Block a user