diff --git a/.wallet/wallet.sqlite3-shm b/.wallet/wallet.sqlite3-shm new file mode 100644 index 00000000..43eb655d Binary files /dev/null and b/.wallet/wallet.sqlite3-shm differ diff --git a/.wallet/wallet.sqlite3-wal b/.wallet/wallet.sqlite3-wal new file mode 100644 index 00000000..234524fc Binary files /dev/null and b/.wallet/wallet.sqlite3-wal differ diff --git a/AGENTS.md b/AGENTS.md new file mode 100644 index 00000000..e4082dca --- /dev/null +++ b/AGENTS.md @@ -0,0 +1,87 @@ +# Routstr Core - Agent Documentation + +This folder contains detailed documentation about the Routstr project, designed for AI agents and developers to understand how the system works. + +## Quick Navigation + +| Document | Description | +| ---------------------------------------------------------------- | ---------------------------------------------------------- | +| **[01-project-overview.md](.agents/01-project-overview.md)** | High-level architecture, concepts, and directory structure | +| **[02-api-endpoints.md](.agents/02-api-endpoints.md)** | Complete API reference with examples | +| **[03-payment-flow.md](.agents/03-payment-flow.md)** | How Cashu payments work end-to-end | +| **[04-upstream-providers.md](.agents/04-upstream-providers.md)** | Provider architecture and routing | +| **[05-nostr-integration.md](.agents/05-nostr-integration.md)** | NIP-91 discovery and announcements | +| **[06-database-models.md](.agents/06-database-models.md)** | Database schema and SQLModel classes | +| **[07-child-keys.md](.agents/07-child-keys.md)** | Sub-account system with spending limits | +| **[08-docker-deployment.md](.agents/08-docker-deployment.md)** | Docker and production deployment | +| **[09-ui-architecture.md](.agents/09-ui-architecture.md)** | Admin dashboard structure | + +## System Summary + +**Routstr** is a decentralized AI inference marketplace: + +- **Anyone can sell** AI inference for Bitcoin (sats) +- **Anyone can buy** using privacy-preserving eCash (Cashu) +- **No accounts, no KYC**, no central authority + +### Key Technologies + +- **FastAPI**: OpenAI-compatible proxy API +- **Cashu**: Private eCash payments on Bitcoin +- **Nostr**: Censorship-resistant discovery (NIP-91) +- **Next.js**: Admin dashboard +- **SQLite**: Data persistence + +### Request Flow + +``` +1. Client pays with Cashu token (Bearer or x-cashu header) +2. Proxy validates token, reserves cost +3. Request forwarded to upstream AI (OpenAI, Anthropic, etc.) +4. Token usage calculated, balance adjusted +5. Response returned with cost info +``` + +### Pricing + +- Fixed: flat fee per request +- Token-based: per-1K input/output tokens +- Provider fee: markup on upstream costs +- Exchange fee: BTC/USD conversion + +## Quick Code Reference + +### Entry Point + +```python +# routstr/core/main.py +app = FastAPI() +app.include_router(proxy_router) +app.include_router(balance_router) +app.include_router(models_router) +``` + +### Adding a Provider + +```python +# routstr/upstream/myprovider.py +class MyProvider(BaseUpstreamProvider): + provider_type = "myprovider" + def transform_model_name(self, model_id): return model_id +``` + +### Key Files + +| File | Purpose | +| ---------------------------- | ----------------- | +| `routstr/core/main.py` | FastAPI app setup | +| `routstr/proxy.py` | Request routing | +| `routstr/auth.py` | Auth & payment | +| `routstr/wallet.py` | Cashu operations | +| `routstr/payment/price.py` | BTC pricing | +| `routstr/nostr/listing.py` | NIP-91 publish | +| `routstr/nostr/discovery.py` | Provider search | + +## Version + +Current: **0.4.1** diff --git a/TEST_SUITE_OVERVIEW.md b/TEST_SUITE_OVERVIEW.md new file mode 100644 index 00000000..42fd8277 --- /dev/null +++ b/TEST_SUITE_OVERVIEW.md @@ -0,0 +1,180 @@ +# Test Suite Overview: Routstr Core + +**~40 test files** split into **unit tests** (25 files) and **integration tests** (11 files + utilities). The suite covers virtually every aspect of the system. + +--- + +## ๐Ÿ” Wallet & Authentication + +| What's tested | Key files | +|---|---| +| Cashu token receive/send/credit operations | `test_wallet.py` | +| Token mint-swapping (foreign mint โ†’ primary mint) with fee estimation, melt failures, edge cases | `test_wallet.py` | +| API key creation from Cashu tokens, duplicate token handling, invalid token rejection | `test_wallet_authentication.py` | +| Authority header validation (missing, malformed, Bearer format, case insensitivity, XSS/nul-byte injection) | `test_wallet_authentication.py` | +| Wallet info endpoints (`/v1/wallet/`, `/v1/wallet/info`) โ€” data consistency, zero balance, expired keys, key isolation | `test_wallet_information.py` | +| Top-up flow: valid tokens, multiple denominations, spent token rejection, concurrent top-ups, zero-amount edge cases, stress tests (50 sequential) | `test_wallet_topup.py` | + +--- + +## ๐Ÿ’ฐ Balance & Billing + +| What's tested | Key files | +|---|---| +| Refund endpoint: x-cashu token refund, API-key refund, sat/msat unit handling, swept/collected token edge cases, 404/410/409 errors | `test_balance.py` | +| **Critical bug fix**: Balance never goes negative on cost overrun (when `tolerance_percentage` discounts the reservation and actual cost exceeds it) | `test_balance_negative_on_cost_overrun.py` | +| Concurrent cost overruns: parallel finalizations must never drive balance negative or give free inference | `test_balance_negative_on_cost_overrun.py` | +| Key validity dates, balance limits, daily reset policies, periodic reset jobs, orphan zero-balance key prevention | `test_key_logic.py` | +| Max-cost calculation: known/unknown/disabled models, tolerance percentage discounts, fixed vs token pricing | `test_payment_helpers.py` | +| `CashuTransaction` source field defaults, API-key-sourced transactions | `test_balance.py` | + +--- + +## ๐Ÿ”€ Proxy & Request Routing + +| What's tested | Key files | +|---|---| +| POST proxy forwarding (JSON payloads, streaming SSE, non-streaming, content-type preservation, large payloads) | `test_proxy_post_endpoints.py` | +| Model-specific endpoints, malformed JSON, insufficient balance, rate limiting, partial streaming failures | `test_proxy_post_endpoints.py` | +| Concurrent proxy requests (10 concurrent) and database state changes | `test_proxy_post_endpoints.py` | +| 404 catch-all handler: HTML for browsers, JSON for API clients, fallback when UI bundle missing | `test_proxy_not_found.py` | +| Embeddings endpoint and model case-insensitivity | `test_embeddings.py` | +| Model test endpoint: admin auth required, unsupported types rejected, oversized payloads rejected, correct upstream path used | `test_model_test_endpoint_security.py` | + +--- + +## ๐ŸŒŠ Streaming & SSE + +| What's tested | Key files | +|---|---| +| **Streaming SSE framing for every provider**: OpenAI-style plain, OpenRouter keepalive comments, comments glued to data chunks, JSON split across TCP boundaries, byte-by-byte fragmentation, Gemini CRLF framing, Azure leading role chunks | `test_streaming_sse_providers.py` | +| **OpenRouter regression**: `: OPENROUTER PROCESSING` keepalive comments must not crash clients with `Unexpected token ':'` | `test_streaming_sse_providers.py` | +| **Gemini combined content+usage chunk regression**: Content delivered when usage is in the same chunk | `test_streaming_sse_providers.py` | +| Mid-stream error events, model name override, multi-line non-JSON `data:` blocks with re-prefixing | `test_streaming_sse_providers.py` | +| Stream ID injection: chat completion IDs and model names injected into streamed chunks | `test_stream_id_injection.py` | + +--- + +## ๐Ÿ“จ Anthropic Messages (/v1/messages) Dispatch Path + +| What's tested | Key files | +|---|---| +| Litellm-based dispatch for non-native providers, stripping Anthropic-only fields (`thinking`, `output_config`, `cache_control`, etc.) | `test_messages_litellm_dispatch.py` | +| Non-streaming via litellm โ†’ returns Anthropic Message response with `cost_sats` | `test_messages_litellm_dispatch.py` | +| Streaming via litellm โ†’ emits SSE events + `cost` event at end | `test_messages_litellm_dispatch.py` | +| SSE byte chunks from litellm parsed correctly (split mid-chunk, keepalive comments) | `test_messages_litellm_dispatch.py` | +| x-cashu messages dispatch: non-streaming refund, streaming refund, no-refund when fully consumed | `test_messages_litellm_dispatch.py` | +| Upstream-always-streams + aggregate-on-non-streaming for Fireworks (max_tokens > 4096 workaround) | `test_messages_litellm_dispatch.py` | +| Aggregator: text deltas โ†’ single message, tool_use `input_json_delta` concatenation, SSE byte chunk parsing | `test_messages_litellm_dispatch.py` | +| Cost accumulation across multiple message events (uses `+=` not `max()`) | `test_messages_dispatch_cost_accumulation.py` | +| Token/cost extraction, cache tokens, model name extraction/override, SSE encoding, malformed/negative cost clamping | `test_messages_dispatch_cost_accumulation.py` | + +--- + +## ๐Ÿท๏ธ Provider Field Injection + +| What's tested | Key files | +|---|---| +| Direct upstreams get bare `provider_type`, OpenRouter gets `openrouter:UpstreamProvider`, unknown/missing becomes `"unknown"` | `test_provider_field_injection.py` | +| Idempotency: double-stamping never nests prefix (`openrouter:openrouter:Google` โ†’ `openrouter:Google`) | `test_provider_field_injection.py` | +| Whitespace stripping, non-string/non-dict inputs skipped, `inject_cost_metadata` also stamps provider | `test_provider_field_injection.py` | + +--- + +## โš™๏ธ Upstream Providers + +| What's tested | Key files | +|---|---| +| Azure: `api-key` header instead of `Authorization`, BOM-stripped API version, deployment ID path construction, base URL stripping | `test_upstream_azure.py` | +| Gemini messages: `inject_thought_signatures` for tool calls, `_openai_chunks_to_anthropic_events` translator (text, tool_use, [DONE] sentinel, blank lines) | `test_upstream_gemini.py` | +| Routstr upstream: balance RPC (auth header omitted when api_key empty, connect timeout โ†’ None), `/v1` path preservation, native messages support | `test_upstream_routstr.py` | +| Error normalization: HTML/plaintext upstream errors โ†’ JSON envelope, JSON errors pass through unchanged, empty body handling | `test_upstream_error_response.py` | +| Litellm provider prefix detection: 40+ providers from URL patterns (Fireworks, Groq, xAI, DeepSeek, Together, Perplexity, Mistral, etc.) | `test_litellm_routing.py` | +| Azure ordering beats `api.openai.com`, Ollama localhost detection, casing/trailing slash normalization, custom defaults | `test_litellm_routing.py` | +| Subclass prefix override wins over URL detection, native messages support flags | `test_messages_litellm_dispatch.py` | + +--- + +## ๐Ÿ’ต Cost Calculation & Caching + +| What's tested | Key files | +|---|---| +| OpenAI vs Anthropic cache token formats (subtractive vs additive), cache_read exceeds prompt_tokens, malformed/boolean/float token coercion | `test_cost_calculation_caching.py` | +| Token field fallback order, missing/null usage blocks, both cache_read and cache_creation simultaneously | `test_cost_calculation_caching.py` | +| x-cashu cost injection in non-streaming/streaming responses, `cost_sats` rounding, existing usage fields preserved | `test_x_cashu_cost_sats.py` | + +--- + +## ๐Ÿงฎ Token Counting + +| What's tested | Key files | +|---|---| +| Local `count_tokens` shim: simple messages, litellm fallback, missing model object, empty body, malformed JSON, system prompts, Anthropic system block list, `forwarded_model_id` | `test_count_tokens_local.py` | +| Image token estimation: low/high/auto detail, small/large images, base64, multiple images, `input_image` type, no images | `test_image_tokens.py` | +| Invalid image data falls back to 512ร—512 defaults | `test_image_tokens.py` | + +--- + +## ๐Ÿง  Model Prioritization Algorithm + +| What's tested | Key files | +|---|---| +| Cost scores (basic, with request fee, expensive models), provider penalties (regular=1.0, OpenRouter=1.001) | `test_algorithm.py` | +| Model overrides for missing cached models, deduplication by provider identity (not provider type) | `test_algorithm.py` | + +--- + +## ๐Ÿ”„ Reactive Request Correction + +| What's tested | Key files | +|---|---| +| Stripping deprecated `temperature`, unsupported params from request body before retry | `test_request_correction.py` | +| No correction when param absent, label already applied, error message doesn't match, empty inputs, non-object body | `test_request_correction.py` | +| Deprecated model name NOT stripped as param, streaming 400 buffered error is correctable | `test_request_correction.py` | +| Sequential two-param correction (with `applied` set guard), immutability of input | `test_request_correction.py` | + +--- + +## ๐Ÿ—„๏ธ Database Consistency + +| What's tested | Key files | +|---|---| +| Transaction atomicity: balance update rollback on failure, top-up rollback on network error | `test_database_consistency.py` | +| Concurrent balance updates via direct DB operations, race condition prevention | `test_database_consistency.py` | +| Primary key uniqueness enforced, numeric field constraints | `test_database_consistency.py` | +| Connection pooling under load (50 concurrent requests), index usage (primary key lookup < 10ms) | `test_database_consistency.py` | + +--- + +## ๐Ÿ” Nostr Discovery & Analytics + +| What's tested | Key files | +|---|---| +| Provider discovery endpoint: default format, `include_json=true`, data structure validation, NIP-91-only parsing | `test_provider_management.py` | +| No-providers, offline providers, duplicate URLs, Nostr relay failures, malformed URLs, parameter validation | `test_provider_management.py` | +| Admin routstr top-up with transient upstream failure retry | `test_provider_management.py` | +| Analytics snapshot payload: top model usage aggregation, schema/shape, fingerprint ignores `generated_at` | `test_nostr_analytics.py` | +| Analytics disable/empty-nsec skip, deduplication of unchanged payloads | `test_nostr_analytics.py` | + +--- + +## โš™๏ธ Infrastructure & Settings + +| What's tested | Key files | +|---|---| +| Settings seed from env, DB precedence over env, unknown key discarding, payout settings defaults and validation | `test_settings.py` | +| Periodic upstream models refresh loop: picks up providers added after startup, disabled at non-positive interval | `test_models_refresh_loop.py` | +| Logging SecurityFilter: Bearer tokens, Cashu tokens, nsec keys, API keys โ€” redacted; case insensitivity, multiple secrets, non-sensitive messages left intact | `test_logging_securityfilter.py` | + +--- + +## ๐Ÿงช Integration Test Infrastructure + +The `conftest.py` (~400 lines) provides: + +- **`TestmintWallet`**: Simulated cashu mint with fallback token creation, secure token uniqueness via `secrets.token_hex` to prevent hash collisions in concurrent tests +- **`DatabaseSnapshot`**: Before/after diffing of API key state (added/modified/removed with field-level deltas) +- **App fixture**: Full FastAPI app with all wallet/proxy mocks patched in (credit_balance, send_token, recieve_token, etc.) +- **Authenticated client**: Client with persistent API key and 10k sat balance created automatically +- **WebSocket mock**: Nostr discovery patched to fail fast for performance +- **Docker vs mock mode**: Switches between real Docker services and in-memory mocks via `USE_LOCAL_SERVICES` env var diff --git a/routstr/upstream/ehbp.py b/routstr/upstream/ehbp.py index 3f643a74..d0874866 100644 --- a/routstr/upstream/ehbp.py +++ b/routstr/upstream/ehbp.py @@ -331,14 +331,24 @@ async def _compute_ehbp_actual_cost( actual_model: str | None = usage_dict.pop("model", None) # type: ignore[arg-type] pricing_model_id = model_obj.id expected_upstream_model = model_obj.forwarded_model_id or model_obj.id - if actual_model and actual_model != expected_upstream_model: + # Case-insensitive comparison: ``get_model_instance`` lowercases lookup + # keys, so a casing difference between the header and the configured + # ``forwarded_model_id`` (e.g. ``GLM-5-2`` vs ``glm-5-2``) should not + # be treated as a real mismatch. + if ( + actual_model + and actual_model.lower() != expected_upstream_model.lower() + ): from ..proxy import get_model_instance # ``forwarded_model_id`` values are registered as routable aliases in # the global model map, so ``get_model_instance`` will find a model - # whose upstream ID matches the actually-served model. + # whose upstream ID matches the actually-served model. It also strips + # date-version suffixes (e.g. ``glm-5-2-20260415`` -> ``glm-5-2``), + # so a resolved model that is actually the *same* as the requested + # one is treated as a non-mismatch. actual_model_obj = get_model_instance(actual_model) - if actual_model_obj: + if actual_model_obj and actual_model_obj.id != model_obj.id: logger.info( "EHBP served model differs from requested, using actual " "model for pricing", @@ -350,16 +360,21 @@ async def _compute_ehbp_actual_cost( ) pricing_model_id = actual_model_obj.id else: - logger.warning( - "EHBP served model not found in registry, falling back to " - "requested model for pricing", - extra={ - "requested_model": model_obj.id, - "expected_upstream_model": expected_upstream_model, - "actual_model": actual_model, - }, - ) - actual_model = None # do not propagate unknown model + # Either the served model is not in the registry (unknown), or + # it resolves back to the requested model (e.g. a date-versioned + # alias like ``glm-5-2-20260415``). In both cases use the + # requested model's pricing and do not propagate actual_model. + if actual_model_obj is None: + logger.warning( + "EHBP served model not found in registry, falling back " + "to requested model for pricing", + extra={ + "requested_model": model_obj.id, + "expected_upstream_model": expected_upstream_model, + "actual_model": actual_model, + }, + ) + actual_model = None # do not propagate unknown / same model else: # Models match or no model in header โ€” use requested model's pricing. actual_model = None diff --git a/tests/unit/test_tinfoil_integration.py b/tests/unit/test_tinfoil_integration.py index 3548b4f3..a5579504 100644 --- a/tests/unit/test_tinfoil_integration.py +++ b/tests/unit/test_tinfoil_integration.py @@ -483,6 +483,86 @@ class TestComputeEhbpActualCost: call_args = mock_calc.call_args assert call_args[0][0]["model"] == "llama3-3-70b" + @pytest.mark.asyncio + async def test_case_insensitive_model_match(self) -> None: + """Casing differences between the header and forwarded_model_id + should not trigger a spurious mismatch.""" + model_obj = MagicMock() + model_obj.id = "tinfoil-glm-5-2" + model_obj.forwarded_model_id = "glm-5-2" # lowercase + with patch( + "routstr.upstream.ehbp.calculate_cost", + new_callable=AsyncMock, + ) as mock_calc: + from routstr.payment.cost_calculation import CostData + + mock_calc.return_value = CostData( + base_msats=0, + input_msats=5, + output_msats=10, + total_msats=15, + total_usd=0.0, + input_tokens=42, + output_tokens=10, + ) + # Header returns uppercase โ€” same model, different casing + result = await _compute_ehbp_actual_cost( + "prompt=42,completion=10,total=52,model=GLM-5-2", + model_obj, + 100_000, + ) + assert "actual_model" not in result + # No mismatch: requested model pricing used + call_args = mock_calc.call_args + assert call_args[0][0]["model"] == "tinfoil-glm-5-2" + # get_model_instance must not be consulted for a casing-only diff + assert not any( + call[0] == ("GLM-5-2",) + for call in mock_calc.call_args_list + ) + + @pytest.mark.asyncio + async def test_date_versioned_alias_resolves_to_requested(self) -> None: + """When the served model is a date-versioned alias that resolves back + to the requested model, no mismatch is propagated.""" + model_obj = MagicMock() + model_obj.id = "tinfoil-glm-5-2" + model_obj.forwarded_model_id = "glm-5-2" + + # get_model_instance strips the date suffix and returns the SAME model + actual_model_obj = MagicMock() + actual_model_obj.id = "tinfoil-glm-5-2" # identical to requested + actual_model_obj.forwarded_model_id = "glm-5-2" + + with patch( + "routstr.proxy.get_model_instance", + return_value=actual_model_obj, + ), patch( + "routstr.upstream.ehbp.calculate_cost", + new_callable=AsyncMock, + ) as mock_calc: + from routstr.payment.cost_calculation import CostData + + mock_calc.return_value = CostData( + base_msats=0, + input_msats=5, + output_msats=10, + total_msats=15, + total_usd=0.0, + input_tokens=42, + output_tokens=10, + ) + # Tinfoil returns a date-versioned ID + result = await _compute_ehbp_actual_cost( + "prompt=42,completion=10,total=52,model=glm-5-2-20260415", + model_obj, + 100_000, + ) + # No mismatch โ€” resolves to the same model + assert "actual_model" not in result + call_args = mock_calc.call_args + assert call_args[0][0]["model"] == "tinfoil-glm-5-2" + # --------------------------------------------------------------------------- # TinfoilUpstreamProvider