From 5478fa1d8b2a658adc0bee2d35c6c7479695196b Mon Sep 17 00:00:00 2001 From: 9qeklajc Date: Wed, 23 Sep 2026 21:52:49 +0200 Subject: [PATCH] fix(models): retry the OpenRouter catalogue fetch before giving up MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit OpenRouter occasionally answers /models with a truncated body, so the JSON parse fails. Any exception returned [] for the whole fetch behind one log line, leaving an enabled upstream advertising zero models, with no retry. Split into a single-attempt helper plus a wrapper that retries it 3 times and then returns [] as before. Only /models can fail an attempt; a failing /embeddings/models logs a warning and contributes nothing, so a secondary outage cannot empty the catalogue. Only transient errors retry — a 4xx gives up immediately. The per-attempt timeout drops to 10s so three attempts stay near the old budget: this fetch blocks startup and the refresh loop. Co-Authored-By: Claude Opus 5 (1M context) --- routstr/payment/models.py | 137 ++++++++----- .../test_openrouter_models_fetch_retry.py | 186 ++++++++++++++++++ 2 files changed, 273 insertions(+), 50 deletions(-) create mode 100644 tests/unit/test_openrouter_models_fetch_retry.py diff --git a/routstr/payment/models.py b/routstr/payment/models.py index bdd8fa73..8d7788a1 100644 --- a/routstr/payment/models.py +++ b/routstr/payment/models.py @@ -241,61 +241,98 @@ def _has_valid_pricing(model: dict) -> bool: return True -async def async_fetch_openrouter_models(source_filter: str | None = None) -> list[dict]: - """Asynchronously fetch model information from OpenRouter API.""" +# OpenRouter occasionally answers /models with a truncated body, emptying the +# catalogue behind one log line. Retry, but keep 3 attempts within roughly the +# old single-attempt budget: this fetch blocks startup and the refresh loop. +OPENROUTER_MODELS_MAX_ATTEMPTS = 3 +OPENROUTER_MODELS_TIMEOUT_SECONDS = 10 +OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS = 0.5 + + +def _is_transient(error: BaseException) -> bool: + if isinstance(error, httpx.HTTPStatusError): + return error.response.status_code >= 500 + return True + + +def _parse_models_response(response: httpx.Response | BaseException) -> list[dict]: + if isinstance(response, BaseException): + raise response + response.raise_for_status() + return [ + model + for model in response.json().get("data", []) + if ":free" not in model.get("id", "").lower() + ] + + +async def _fetch_openrouter_models_once(source_filter: str | None) -> list[dict]: + """One attempt. Raises if /models is unusable; embeddings are best-effort.""" base_url = "https://openrouter.ai/api/v1" + timeout = OPENROUTER_MODELS_TIMEOUT_SECONDS - try: - async with httpx.AsyncClient() as client: - models_response, embeddings_response = await asyncio.gather( - client.get(f"{base_url}/models", timeout=30), - client.get(f"{base_url}/embeddings/models", timeout=30), - return_exceptions=True, - ) + async with httpx.AsyncClient() as client: + models_response, embeddings_response = await asyncio.gather( + client.get(f"{base_url}/models", timeout=timeout), + client.get(f"{base_url}/embeddings/models", timeout=timeout), + return_exceptions=True, + ) - def process_models_response( - response: httpx.Response | BaseException, - ) -> list[dict]: - if not isinstance(response, BaseException): - response.raise_for_status() - data = response.json() - return [ - model - for model in data.get("data", []) - if ":free" not in model.get("id", "").lower() - ] + # Losing /models is what empties the node, so it fails the attempt and + # the caller retries. A missing embeddings half must not do the same. + models_data = _parse_models_response(models_response) + try: + models_data.extend(_parse_models_response(embeddings_response)) + except Exception as e: + logger.warning(f"Skipping OpenRouter embeddings models: {e}") + + # Apply source filter and exclusions + filtered_models = [] + for model in models_data: + model_id = model.get("id", "") + + if source_filter: + source_prefix = f"{source_filter}/" + if not model_id.startswith(source_prefix): + continue + + model = dict(model) + model["id"] = model_id[len(source_prefix) :] + model_id = model["id"] + + if "(free)" in model.get("name", ""): + continue + + if not _has_valid_pricing(model): + continue + + filtered_models.append(model) + + return filtered_models + + +async def async_fetch_openrouter_models(source_filter: str | None = None) -> list[dict]: + """Fetch the OpenRouter catalogue; ``[]`` once every attempt has failed.""" + for attempt in range(1, OPENROUTER_MODELS_MAX_ATTEMPTS + 1): + try: + return await _fetch_openrouter_models_once(source_filter) + except Exception as e: + last_attempt = attempt == OPENROUTER_MODELS_MAX_ATTEMPTS + if last_attempt or not _is_transient(e): + logger.error( + f"Error (async) fetching models from OpenRouter API " + f"after {attempt} attempt(s): {e}" + ) return [] + logger.warning( + f"OpenRouter models fetch attempt {attempt}/" + f"{OPENROUTER_MODELS_MAX_ATTEMPTS} failed: {e}; retrying" + ) + # Jittered so nodes do not retry in lockstep. + backoff = OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS * attempt + await asyncio.sleep(backoff * random.uniform(0.5, 1.5)) - models_data: list[dict] = [] - models_data.extend(process_models_response(models_response)) - models_data.extend(process_models_response(embeddings_response)) - - # Apply source filter and exclusions - filtered_models = [] - for model in models_data: - model_id = model.get("id", "") - - if source_filter: - source_prefix = f"{source_filter}/" - if not model_id.startswith(source_prefix): - continue - - model = dict(model) - model["id"] = model_id[len(source_prefix) :] - model_id = model["id"] - - if "(free)" in model.get("name", ""): - continue - - if not _has_valid_pricing(model): - continue - - filtered_models.append(model) - - return filtered_models - except Exception as e: - logger.error(f"Error (async) fetching models from OpenRouter API: {e}") - return [] + return [] def _build_model_from_row( diff --git a/tests/unit/test_openrouter_models_fetch_retry.py b/tests/unit/test_openrouter_models_fetch_retry.py new file mode 100644 index 00000000..862bfffe --- /dev/null +++ b/tests/unit/test_openrouter_models_fetch_retry.py @@ -0,0 +1,186 @@ +"""Retry behaviour for the OpenRouter catalogue fetch.""" + +from __future__ import annotations + +import json +from typing import Any, Callable + +import httpx +import pytest + +from routstr.payment import models as models_module +from routstr.payment.models import async_fetch_openrouter_models + +MODELS_URL = "https://openrouter.ai/api/v1/models" +EMBEDDINGS_URL = "https://openrouter.ai/api/v1/embeddings/models" + + +def _model(model_id: str) -> dict[str, Any]: + return { + "id": model_id, + "name": model_id, + "pricing": {"prompt": "0.000001", "completion": "0.000002"}, + } + + +def _ok_response(url: str, payload: dict[str, Any]) -> httpx.Response: + return httpx.Response( + 200, + request=httpx.Request("GET", url), + content=json.dumps(payload).encode(), + headers={"content-type": "application/json"}, + ) + + +def _error_response(url: str, status: int) -> httpx.Response: + return httpx.Response(status, request=httpx.Request("GET", url), content=b"nope") + + +def _truncated_response(url: str) -> httpx.Response: + """A body cut mid-JSON — the shape OpenRouter actually sent the node.""" + return httpx.Response( + 200, + request=httpx.Request("GET", url), + content=b'{"data": [{"id": "vendor/model-a", "name": "Model A", "pric', + headers={"content-type": "application/json"}, + ) + + +def _payload_for(url: str) -> dict[str, Any]: + if url.endswith("/embeddings/models"): + return {"data": [_model("vendor/embed-1")]} + return {"data": [_model("vendor/model-a")]} + + +@pytest.fixture(autouse=True) +def _no_retry_backoff(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(models_module, "OPENROUTER_MODELS_RETRY_BACKOFF_SECONDS", 0) + + +def _install_get( + monkeypatch: pytest.MonkeyPatch, + handler: Callable[[str, int], httpx.Response], +) -> dict[str, int]: + """Patch ``httpx.AsyncClient.get`` and count calls per endpoint.""" + counts: dict[str, int] = {} + + async def fake_get( + self: httpx.AsyncClient, url: Any, **kwargs: Any + ) -> httpx.Response: + key = str(url) + counts[key] = counts.get(key, 0) + 1 + return handler(key, counts[key]) + + monkeypatch.setattr(httpx.AsyncClient, "get", fake_get) + return counts + + +@pytest.mark.asyncio +async def test_truncated_body_is_retried_and_recovers( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A truncated body on the first attempt must not empty the catalogue.""" + + def handler(url: str, call: int) -> httpx.Response: + if call == 1: + return _truncated_response(url) + return _ok_response(url, _payload_for(url)) + + counts = _install_get(monkeypatch, handler) + + result = await async_fetch_openrouter_models() + + assert [model["id"] for model in result] == ["vendor/model-a", "vendor/embed-1"] + assert counts[MODELS_URL] == 2 + assert counts[EMBEDDINGS_URL] == 2 + + +@pytest.mark.asyncio +async def test_gives_up_after_max_attempts_and_logs_the_error( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """After every attempt fails: log once and return an empty catalogue.""" + errors: list[str] = [] + monkeypatch.setattr(models_module.logger, "error", lambda msg: errors.append(msg)) + + counts = _install_get(monkeypatch, lambda url, call: _truncated_response(url)) + + result = await async_fetch_openrouter_models() + + assert result == [] + assert counts[MODELS_URL] == models_module.OPENROUTER_MODELS_MAX_ATTEMPTS + assert len(errors) == 1 + assert "after 3 attempt(s)" in errors[0] + + +@pytest.mark.asyncio +async def test_embeddings_outage_still_yields_the_main_catalogue( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The secondary endpoint is best-effort: it cannot empty the catalogue.""" + + def handler(url: str, call: int) -> httpx.Response: + if url == EMBEDDINGS_URL: + return _error_response(url, 503) + return _ok_response(url, _payload_for(url)) + + counts = _install_get(monkeypatch, handler) + + result = await async_fetch_openrouter_models() + + assert [model["id"] for model in result] == ["vendor/model-a"] + assert counts[MODELS_URL] == 1 + assert counts[EMBEDDINGS_URL] == 1 + + +@pytest.mark.asyncio +async def test_main_catalogue_server_error_is_retried( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """A 5xx on /models is transient, so the attempt is retried.""" + + def handler(url: str, call: int) -> httpx.Response: + if url == MODELS_URL and call == 1: + return _error_response(url, 503) + return _ok_response(url, _payload_for(url)) + + counts = _install_get(monkeypatch, handler) + + result = await async_fetch_openrouter_models() + + assert [model["id"] for model in result] == ["vendor/model-a", "vendor/embed-1"] + assert counts[MODELS_URL] == 2 + + +@pytest.mark.asyncio +async def test_client_error_is_not_retried(monkeypatch: pytest.MonkeyPatch) -> None: + """Retrying a 4xx only adds load to an upstream that already said no.""" + counts = _install_get(monkeypatch, lambda url, call: _error_response(url, 401)) + + result = await async_fetch_openrouter_models() + + assert result == [] + assert counts[MODELS_URL] == 1 + + +@pytest.mark.asyncio +async def test_source_filter_and_free_models_are_still_applied( + monkeypatch: pytest.MonkeyPatch, +) -> None: + """The moved filter loop still strips prefixes and drops free tiers.""" + payload = { + "data": [ + _model("openai/gpt-x"), + _model("openai/gpt-x:free"), + _model("other/y"), + ] + } + + def handler(url: str, call: int) -> httpx.Response: + return _ok_response(url, payload if url == MODELS_URL else {"data": []}) + + _install_get(monkeypatch, handler) + + result = await async_fetch_openrouter_models(source_filter="openai") + + assert [model["id"] for model in result] == ["gpt-x"]