fix(upstream): preserve source modality instead of flattening it

The generic provider computed modality as "text->text" for any vision model
and "text" otherwise, discarding the source's own modality and mislabelling
image-capable models. Carry OpenRouter's modality string through verbatim, and
when a source doesn't supply one, derive it from the captured input/output
modalities in the same "in->out" shape (e.g. "text+image->text").

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Jeroen Ubbink
2026-07-05 15:40:57 +02:00
co-authored by Claude Opus 4.8
parent ae9748db02
commit ec0fd1143b
3 changed files with 16 additions and 6 deletions
+7 -4
View File
@@ -147,10 +147,13 @@ class GenericUpstreamProvider(BaseUpstreamProvider):
else:
enabled = True
modality = (
"text->text"
if "image" in resolved.input_modalities
else "text"
# Prefer the source's own modality string (OpenRouter ships
# one, e.g. "text+image->text"); otherwise derive it from the
# captured input/output modalities in the same "in->out" shape
# rather than flattening vision models to "text->text".
modality = resolved.modality or (
f"{'+'.join(resolved.input_modalities)}"
f"->{'+'.join(resolved.output_modalities)}"
)
# A source can carry a price but no context (e.g. a litellm
+2
View File
@@ -31,6 +31,7 @@ class ResolvedPricing:
completion: float
context_length: int | None
source: str
modality: str | None = None
max_completion_tokens: int | None = None
input_cache_read: float = 0.0
input_cache_write: float = 0.0
@@ -159,6 +160,7 @@ def _from_openrouter(model_id: str, feed: list[dict]) -> ResolvedPricing | None:
completion=completion,
context_length=_as_int(entry.get("context_length")),
source="openrouter",
modality=architecture.get("modality"),
max_completion_tokens=_as_int(top_provider.get("max_completion_tokens")),
input_cache_read=_as_float(pricing.get("input_cache_read")) or 0.0,
input_cache_write=_as_float(pricing.get("input_cache_write")) or 0.0,
+7 -2
View File
@@ -104,6 +104,9 @@ async def test_native_model_spec_resolves_and_captures_metadata() -> None:
assert model.pricing.completion == pytest.approx(1.5 / 1_000_000)
assert model.context_length == 65536
assert "image" in model.architecture.input_modalities
# Vision capability must be reflected in the combined modality string, not
# flattened to "text->text".
assert model.architecture.modality == "text+image->text"
# A native price never needs the OpenRouter feed.
or_feed.assert_not_awaited()
@@ -237,8 +240,8 @@ async def test_unknown_to_litellm_resolves_via_openrouter() -> None:
"name": "Exotic 9000",
"context_length": 65536,
"architecture": {
"modality": "text->text",
"input_modalities": ["text"],
"modality": "text+image->text",
"input_modalities": ["text", "image"],
"output_modalities": ["text"],
"tokenizer": "Other",
"instruct_type": None,
@@ -263,6 +266,8 @@ async def test_unknown_to_litellm_resolves_via_openrouter() -> None:
assert model.pricing.prompt == pytest.approx(5e-06)
assert model.pricing.completion == pytest.approx(1e-05)
assert model.context_length == 65536
# The feed's own modality string is carried through verbatim, not recomputed.
assert model.architecture.modality == "text+image->text"
or_feed.assert_awaited()