From ec0fd1143b2d880c0eb182b688b687d4f08bf6d7 Mon Sep 17 00:00:00 2001 From: Jeroen Ubbink Date: Sun, 5 Jul 2026 15:40:57 +0200 Subject: [PATCH] fix(upstream): preserve source modality instead of flattening it The generic provider computed modality as "text->text" for any vision model and "text" otherwise, discarding the source's own modality and mislabelling image-capable models. Carry OpenRouter's modality string through verbatim, and when a source doesn't supply one, derive it from the captured input/output modalities in the same "in->out" shape (e.g. "text+image->text"). Co-Authored-By: Claude Opus 4.8 --- routstr/upstream/generic.py | 11 +++++++---- routstr/upstream/pricing_resolver.py | 2 ++ tests/unit/test_upstream_generic.py | 9 +++++++-- 3 files changed, 16 insertions(+), 6 deletions(-) diff --git a/routstr/upstream/generic.py b/routstr/upstream/generic.py index 5f388825..0a040b9c 100644 --- a/routstr/upstream/generic.py +++ b/routstr/upstream/generic.py @@ -147,10 +147,13 @@ class GenericUpstreamProvider(BaseUpstreamProvider): else: enabled = True - modality = ( - "text->text" - if "image" in resolved.input_modalities - else "text" + # Prefer the source's own modality string (OpenRouter ships + # one, e.g. "text+image->text"); otherwise derive it from the + # captured input/output modalities in the same "in->out" shape + # rather than flattening vision models to "text->text". + modality = resolved.modality or ( + f"{'+'.join(resolved.input_modalities)}" + f"->{'+'.join(resolved.output_modalities)}" ) # A source can carry a price but no context (e.g. a litellm diff --git a/routstr/upstream/pricing_resolver.py b/routstr/upstream/pricing_resolver.py index f956334c..f5c1cb32 100644 --- a/routstr/upstream/pricing_resolver.py +++ b/routstr/upstream/pricing_resolver.py @@ -31,6 +31,7 @@ class ResolvedPricing: completion: float context_length: int | None source: str + modality: str | None = None max_completion_tokens: int | None = None input_cache_read: float = 0.0 input_cache_write: float = 0.0 @@ -159,6 +160,7 @@ def _from_openrouter(model_id: str, feed: list[dict]) -> ResolvedPricing | None: completion=completion, context_length=_as_int(entry.get("context_length")), source="openrouter", + modality=architecture.get("modality"), max_completion_tokens=_as_int(top_provider.get("max_completion_tokens")), input_cache_read=_as_float(pricing.get("input_cache_read")) or 0.0, input_cache_write=_as_float(pricing.get("input_cache_write")) or 0.0, diff --git a/tests/unit/test_upstream_generic.py b/tests/unit/test_upstream_generic.py index 65c9ada2..1bef8867 100644 --- a/tests/unit/test_upstream_generic.py +++ b/tests/unit/test_upstream_generic.py @@ -104,6 +104,9 @@ async def test_native_model_spec_resolves_and_captures_metadata() -> None: assert model.pricing.completion == pytest.approx(1.5 / 1_000_000) assert model.context_length == 65536 assert "image" in model.architecture.input_modalities + # Vision capability must be reflected in the combined modality string, not + # flattened to "text->text". + assert model.architecture.modality == "text+image->text" # A native price never needs the OpenRouter feed. or_feed.assert_not_awaited() @@ -237,8 +240,8 @@ async def test_unknown_to_litellm_resolves_via_openrouter() -> None: "name": "Exotic 9000", "context_length": 65536, "architecture": { - "modality": "text->text", - "input_modalities": ["text"], + "modality": "text+image->text", + "input_modalities": ["text", "image"], "output_modalities": ["text"], "tokenizer": "Other", "instruct_type": None, @@ -263,6 +266,8 @@ async def test_unknown_to_litellm_resolves_via_openrouter() -> None: assert model.pricing.prompt == pytest.approx(5e-06) assert model.pricing.completion == pytest.approx(1e-05) assert model.context_length == 65536 + # The feed's own modality string is carried through verbatim, not recomputed. + assert model.architecture.modality == "text+image->text" or_feed.assert_awaited()