Merge pull request #766 from Routstr/fix/openrouter-models-fetch-retry

fix(models): retry the OpenRouter catalogue fetch before giving up
This commit is contained in:
9qeklajc
2026-09-23 23:08:47 +02:00
committed by GitHub
2 changed files with 273 additions and 50 deletions
+87 -50
View File
@@ -241,61 +241,98 @@ def _has_valid_pricing(model: dict) -> bool:
return True
async def async_fetch_openrouter_models(source_filter: str | None = None) -> list[dict]:
"""Asynchronously fetch model information from OpenRouter API."""
# OpenRouter occasionally answers /models with a truncated body, emptying the
# catalogue behind one log line. Retry, but keep 3 attempts within roughly the
# old single-attempt budget: this fetch blocks startup and the refresh loop.
OPENROUTER_MODELS_MAX_ATTEMPTS = 3
OPENROUTER_MODELS_TIMEOUT_SECONDS = 10
OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS = 0.5
def _is_transient(error: BaseException) -> bool:
if isinstance(error, httpx.HTTPStatusError):
return error.response.status_code >= 500
return True
def _parse_models_response(response: httpx.Response | BaseException) -> list[dict]:
if isinstance(response, BaseException):
raise response
response.raise_for_status()
return [
model
for model in response.json().get("data", [])
if ":free" not in model.get("id", "").lower()
]
async def _fetch_openrouter_models_once(source_filter: str | None) -> list[dict]:
"""One attempt. Raises if /models is unusable; embeddings are best-effort."""
base_url = "https://openrouter.ai/api/v1"
timeout = OPENROUTER_MODELS_TIMEOUT_SECONDS
try:
async with httpx.AsyncClient() as client:
models_response, embeddings_response = await asyncio.gather(
client.get(f"{base_url}/models", timeout=30),
client.get(f"{base_url}/embeddings/models", timeout=30),
return_exceptions=True,
)
async with httpx.AsyncClient() as client:
models_response, embeddings_response = await asyncio.gather(
client.get(f"{base_url}/models", timeout=timeout),
client.get(f"{base_url}/embeddings/models", timeout=timeout),
return_exceptions=True,
)
def process_models_response(
response: httpx.Response | BaseException,
) -> list[dict]:
if not isinstance(response, BaseException):
response.raise_for_status()
data = response.json()
return [
model
for model in data.get("data", [])
if ":free" not in model.get("id", "").lower()
]
# Losing /models is what empties the node, so it fails the attempt and
# the caller retries. A missing embeddings half must not do the same.
models_data = _parse_models_response(models_response)
try:
models_data.extend(_parse_models_response(embeddings_response))
except Exception as e:
logger.warning(f"Skipping OpenRouter embeddings models: {e}")
# Apply source filter and exclusions
filtered_models = []
for model in models_data:
model_id = model.get("id", "")
if source_filter:
source_prefix = f"{source_filter}/"
if not model_id.startswith(source_prefix):
continue
model = dict(model)
model["id"] = model_id[len(source_prefix) :]
model_id = model["id"]
if "(free)" in model.get("name", ""):
continue
if not _has_valid_pricing(model):
continue
filtered_models.append(model)
return filtered_models
async def async_fetch_openrouter_models(source_filter: str | None = None) -> list[dict]:
"""Fetch the OpenRouter catalogue; ``[]`` once every attempt has failed."""
for attempt in range(1, OPENROUTER_MODELS_MAX_ATTEMPTS + 1):
try:
return await _fetch_openrouter_models_once(source_filter)
except Exception as e:
last_attempt = attempt == OPENROUTER_MODELS_MAX_ATTEMPTS
if last_attempt or not _is_transient(e):
logger.error(
f"Error (async) fetching models from OpenRouter API "
f"after {attempt} attempt(s): {e}"
)
return []
logger.warning(
f"OpenRouter models fetch attempt {attempt}/"
f"{OPENROUTER_MODELS_MAX_ATTEMPTS} failed: {e}; retrying"
)
# Jittered so nodes do not retry in lockstep.
backoff = OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS * attempt
await asyncio.sleep(backoff * random.uniform(0.5, 1.5))
models_data: list[dict] = []
models_data.extend(process_models_response(models_response))
models_data.extend(process_models_response(embeddings_response))
# Apply source filter and exclusions
filtered_models = []
for model in models_data:
model_id = model.get("id", "")
if source_filter:
source_prefix = f"{source_filter}/"
if not model_id.startswith(source_prefix):
continue
model = dict(model)
model["id"] = model_id[len(source_prefix) :]
model_id = model["id"]
if "(free)" in model.get("name", ""):
continue
if not _has_valid_pricing(model):
continue
filtered_models.append(model)
return filtered_models
except Exception as e:
logger.error(f"Error (async) fetching models from OpenRouter API: {e}")
return []
return []
def _build_model_from_row(
@@ -0,0 +1,186 @@
"""Retry behaviour for the OpenRouter catalogue fetch."""
from __future__ import annotations
import json
from typing import Any, Callable
import httpx
import pytest
from routstr.payment import models as models_module
from routstr.payment.models import async_fetch_openrouter_models
MODELS_URL = "https://openrouter.ai/api/v1/models"
EMBEDDINGS_URL = "https://openrouter.ai/api/v1/embeddings/models"
def _model(model_id: str) -> dict[str, Any]:
return {
"id": model_id,
"name": model_id,
"pricing": {"prompt": "0.000001", "completion": "0.000002"},
}
def _ok_response(url: str, payload: dict[str, Any]) -> httpx.Response:
return httpx.Response(
200,
request=httpx.Request("GET", url),
content=json.dumps(payload).encode(),
headers={"content-type": "application/json"},
)
def _error_response(url: str, status: int) -> httpx.Response:
return httpx.Response(status, request=httpx.Request("GET", url), content=b"nope")
def _truncated_response(url: str) -> httpx.Response:
"""A body cut mid-JSON — the shape OpenRouter actually sent the node."""
return httpx.Response(
200,
request=httpx.Request("GET", url),
content=b'{"data": [{"id": "vendor/model-a", "name": "Model A", "pric',
headers={"content-type": "application/json"},
)
def _payload_for(url: str) -> dict[str, Any]:
if url.endswith("/embeddings/models"):
return {"data": [_model("vendor/embed-1")]}
return {"data": [_model("vendor/model-a")]}
@pytest.fixture(autouse=True)
def _no_retry_backoff(monkeypatch: pytest.MonkeyPatch) -> None:
monkeypatch.setattr(models_module, "OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS", 0)
def _install_get(
monkeypatch: pytest.MonkeyPatch,
handler: Callable[[str, int], httpx.Response],
) -> dict[str, int]:
"""Patch ``httpx.AsyncClient.get`` and count calls per endpoint."""
counts: dict[str, int] = {}
async def fake_get(
self: httpx.AsyncClient, url: Any, **kwargs: Any
) -> httpx.Response:
key = str(url)
counts[key] = counts.get(key, 0) + 1
return handler(key, counts[key])
monkeypatch.setattr(httpx.AsyncClient, "get", fake_get)
return counts
@pytest.mark.asyncio
async def test_truncated_body_is_retried_and_recovers(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A truncated body on the first attempt must not empty the catalogue."""
def handler(url: str, call: int) -> httpx.Response:
if call == 1:
return _truncated_response(url)
return _ok_response(url, _payload_for(url))
counts = _install_get(monkeypatch, handler)
result = await async_fetch_openrouter_models()
assert [model["id"] for model in result] == ["vendor/model-a", "vendor/embed-1"]
assert counts[MODELS_URL] == 2
assert counts[EMBEDDINGS_URL] == 2
@pytest.mark.asyncio
async def test_gives_up_after_max_attempts_and_logs_the_error(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""After every attempt fails: log once and return an empty catalogue."""
errors: list[str] = []
monkeypatch.setattr(models_module.logger, "error", lambda msg: errors.append(msg))
counts = _install_get(monkeypatch, lambda url, call: _truncated_response(url))
result = await async_fetch_openrouter_models()
assert result == []
assert counts[MODELS_URL] == models_module.OPENROUTER_MODELS_MAX_ATTEMPTS
assert len(errors) == 1
assert "after 3 attempt(s)" in errors[0]
@pytest.mark.asyncio
async def test_embeddings_outage_still_yields_the_main_catalogue(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""The secondary endpoint is best-effort: it cannot empty the catalogue."""
def handler(url: str, call: int) -> httpx.Response:
if url == EMBEDDINGS_URL:
return _error_response(url, 503)
return _ok_response(url, _payload_for(url))
counts = _install_get(monkeypatch, handler)
result = await async_fetch_openrouter_models()
assert [model["id"] for model in result] == ["vendor/model-a"]
assert counts[MODELS_URL] == 1
assert counts[EMBEDDINGS_URL] == 1
@pytest.mark.asyncio
async def test_main_catalogue_server_error_is_retried(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""A 5xx on /models is transient, so the attempt is retried."""
def handler(url: str, call: int) -> httpx.Response:
if url == MODELS_URL and call == 1:
return _error_response(url, 503)
return _ok_response(url, _payload_for(url))
counts = _install_get(monkeypatch, handler)
result = await async_fetch_openrouter_models()
assert [model["id"] for model in result] == ["vendor/model-a", "vendor/embed-1"]
assert counts[MODELS_URL] == 2
@pytest.mark.asyncio
async def test_client_error_is_not_retried(monkeypatch: pytest.MonkeyPatch) -> None:
"""Retrying a 4xx only adds load to an upstream that already said no."""
counts = _install_get(monkeypatch, lambda url, call: _error_response(url, 401))
result = await async_fetch_openrouter_models()
assert result == []
assert counts[MODELS_URL] == 1
@pytest.mark.asyncio
async def test_source_filter_and_free_models_are_still_applied(
monkeypatch: pytest.MonkeyPatch,
) -> None:
"""The moved filter loop still strips prefixes and drops free tiers."""
payload = {
"data": [
_model("openai/gpt-x"),
_model("openai/gpt-x:free"),
_model("other/y"),
]
}
def handler(url: str, call: int) -> httpx.Response:
return _ok_response(url, payload if url == MODELS_URL else {"data": []})
_install_get(monkeypatch, handler)
result = await async_fetch_openrouter_models(source_filter="openai")
assert [model["id"] for model in result] == ["gpt-x"]