diff --git a/routstr/payment/cost_calculation.py b/routstr/payment/cost_calculation.py index 820bd174..84f3847f 100644 --- a/routstr/payment/cost_calculation.py +++ b/routstr/payment/cost_calculation.py @@ -41,6 +41,28 @@ class CostDataError(BaseModel): code: str +def _empty_cost(cls: type[CostData] = CostData) -> CostData: + """Build an all-zero cost object — a full refund for an empty response. + + Shared by the two paths that must not bill: an upstream response with no + usage data at all, and one that reports a USD cost but carries zero tokens + in every bucket. + """ + return cls( + base_msats=0, + input_msats=0, + output_msats=0, + total_msats=0, + total_usd=0.0, + input_tokens=0, + output_tokens=0, + cache_read_input_tokens=0, + cache_creation_input_tokens=0, + cache_read_msats=0, + cache_creation_msats=0, + ) + + async def calculate_cost( response_data: dict, max_cost: int, @@ -83,19 +105,7 @@ async def calculate_cost( else None, }, ) - return MaxCostData( - base_msats=0, - input_msats=0, - output_msats=0, - total_msats=0, - total_usd=0.0, - input_tokens=0, - output_tokens=0, - cache_read_input_tokens=0, - cache_creation_input_tokens=0, - cache_read_msats=0, - cache_creation_msats=0, - ) + return _empty_cost(MaxCostData) usage_data = response_data.get("usage") or {} if not isinstance(usage_data, dict): @@ -109,6 +119,27 @@ async def calculate_cost( # Try USD cost first usd_cost = _resolve_usd_cost(usage_data, response_data) if usd_cost > 0: + truly_empty = ( + input_tokens == 0 + and output_tokens == 0 + and cache_read_tokens == 0 + and cache_creation_tokens == 0 + ) + if truly_empty: + logger.warning( + "Upstream reported a USD cost but the response carries no " + "tokens at all (input, output, cache-read and cache-creation " + "are all zero) — refunding in full rather than billing the " + "USD-derived cost for an empty response.", + extra={ + "model": response_data.get("model", "unknown"), + "usd_cost": usd_cost, + "usage_keys": sorted(usage_data.keys()) + if isinstance(usage_data, dict) + else None, + }, + ) + return _empty_cost() if input_tokens == 0 and output_tokens == 0: logger.warning( "Upstream reported a USD cost but no token counts — " diff --git a/tests/unit/test_cost_calculation_caching.py b/tests/unit/test_cost_calculation_caching.py index 41d85385..93c68877 100644 --- a/tests/unit/test_cost_calculation_caching.py +++ b/tests/unit/test_cost_calculation_caching.py @@ -404,6 +404,67 @@ async def test_deepseek_malformed_hit_tokens_coerce_to_zero() -> None: assert result.cache_read_input_tokens == 0 +# ============================================================================ +# Truly-empty response with a non-zero USD cost → full refund +# +# When an upstream reports a USD cost but the response carries NO tokens at all +# (input, output, cache-read and cache-creation all zero), billing the +# USD-derived cost charges the user for nothing. Refund in full. The gate is +# tightened relative to PR #489: a cache-read/-creation-only turn legitimately +# reports zero prompt/completion tokens with a real cost and must still bill. +# ============================================================================ +@pytest.mark.asyncio +async def test_truly_empty_usd_cost_response_is_refunded( + mock_fixed_pricing: None, +) -> None: + """0 input + 0 output + 0 cache tokens with a non-zero USD cost → refund.""" + response = { + "model": "gpt-4", + "usage": { + "prompt_tokens": 0, + "completion_tokens": 0, + "total_cost": 0.01, # non-zero USD cost despite no tokens + }, + } + result = await calculate_cost(response, max_cost=100000) + + assert isinstance(result, CostData) + assert result.total_msats == 0 # full refund + assert result.input_msats == 0 + assert result.output_msats == 0 + assert result.total_usd == 0.0 + assert result.input_tokens == 0 + assert result.output_tokens == 0 + assert result.cache_read_input_tokens == 0 + assert result.cache_creation_input_tokens == 0 + + +@pytest.mark.asyncio +async def test_cache_read_only_usd_cost_response_is_billed( + mock_fixed_pricing: None, +) -> None: + """Cache-read-only turn (0 prompt/completion, non-zero cost) still bills.""" + response = { + "model": "claude-3-5-sonnet", + "usage": { + "input_tokens": 0, + "output_tokens": 0, + "cache_read_input_tokens": 1000, # real cached usage + "cache_creation_input_tokens": 0, + "total_cost": 0.01, # non-zero USD cost + }, + } + result = await calculate_cost(response, max_cost=100000) + + assert isinstance(result, CostData) + # NOT refunded — the USD cost is billed in full. Pinning the exact value + # guards against any future regression that would over-refund a cache-only + # turn (the bug in PR #489, which refunded whenever prompt+completion == 0). + assert result.total_msats == 200000 + assert result.total_usd == 0.01 + assert result.cache_read_input_tokens == 1000 + + # ============================================================================ # Test 13: Missing Usage Block # ============================================================================