From e70986b72a43c17684bc9832caaa0bea80129d1c Mon Sep 17 00:00:00 2001 From: GitHappens2Me Date: Sun, 25 May 2025 22:56:28 +0200 Subject: [PATCH 1/8] added Key-Expiry-Time and Refund-LNURL and automatic refunds to lnurl --- .gitignore | 4 ++-- router/admin.py | 7 ++++--- router/auth.py | 16 ++++++++++------ router/cashu.py | 48 +++++++++++++++++++++++++++++++++++++++--------- router/db.py | 4 ++++ router/main.py | 4 +++- router/proxy.py | 24 +++++++++++++++++++++++- 7 files changed, 85 insertions(+), 22 deletions(-) diff --git a/.gitignore b/.gitignore index c6414db9..39f0e554 100644 --- a/.gitignore +++ b/.gitignore @@ -6,5 +6,5 @@ wallet.sqlite3 # Development .notes -.keys.db -.wallet.sqlite3 \ No newline at end of file +.*keys.db +.*wallet.sqlite3 \ No newline at end of file diff --git a/router/admin.py b/router/admin.py index b9f6be2d..7d33cd03 100644 --- a/router/admin.py +++ b/router/admin.py @@ -91,7 +91,7 @@ async def dashboard(request: Request) -> str: result = await session.exec(select(ApiKey)) api_keys = result.all() api_keys_table_rows = "".join( - f"{key.hashed_key}{key.balance}{key.refund_address}{key.total_spent}{key.total_requests}" + f"{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{key.key_expiry_time}" for key in api_keys ) @@ -124,9 +124,10 @@ async def dashboard(request: Request) -> str: Hashed Key Balance (mSats) - Refund Address - Total Spent(mSats) + Total Spent (mSats) Total Requests + Refund Address + Refund Time {api_keys_table_rows} diff --git a/router/auth.py b/router/auth.py index 32428ded..b8f47e3b 100644 --- a/router/auth.py +++ b/router/auth.py @@ -1,5 +1,6 @@ import hashlib import os +from typing import Optional from fastapi import HTTPException @@ -19,7 +20,7 @@ COST_PER_1K_OUTPUT_TOKENS = ( MODEL_BASED_PRICING = os.environ.get("MODEL_BASED_PRICING", "false").lower() == "true" -async def validate_bearer_key(bearer_key: str, session: AsyncSession) -> ApiKey: +async def validate_bearer_key(bearer_key: str, session: AsyncSession, refund_address: Optional[str] = None, key_expiry_time: Optional[int] = None) -> ApiKey: """ Validates the provided API key using SQLModel. If it's a cashu key, it redeems it and stores its hash and balance. @@ -29,15 +30,18 @@ async def validate_bearer_key(bearer_key: str, session: AsyncSession) -> ApiKey: raise HTTPException(status_code=401, detail="api-key or cashu-token required") if bearer_key.startswith("sk-"): - if exsisting_key := await session.get(ApiKey, bearer_key[3:]): - return exsisting_key + if existing_key := await session.get(ApiKey, bearer_key[3:]): + existing_key.key_expiry_time, existing_key.refund_address = key_expiry_time, refund_address + return existing_key if bearer_key.startswith("cashu"): try: hashed_key = hashlib.sha256(bearer_key.encode()).hexdigest() - if exsisting_key := await session.get(ApiKey, hashed_key): - return exsisting_key - new_key = ApiKey(hashed_key=hashed_key, balance=0) + if existing_key := await session.get(ApiKey, hashed_key): + existing_key.key_expiry_time, existing_key.refund_address = key_expiry_time, refund_address + return existing_key + + new_key = ApiKey(hashed_key=hashed_key, balance=0, refund_address = refund_address, key_expiry_time = key_expiry_time) await credit_balance(bearer_key, new_key, session) #TODO: see cashu.py "_initialize_wallet" await session.refresh(new_key) return new_key diff --git a/router/cashu.py b/router/cashu.py index 6d49f4af..ef2005f2 100644 --- a/router/cashu.py +++ b/router/cashu.py @@ -1,11 +1,13 @@ import os import httpx +import asyncio +import time from cashu.core.base import Token # type: ignore from cashu.wallet.wallet import Wallet # type: ignore from cashu.wallet.helpers import deserialize_token_from_string, receive # type: ignore from sqlmodel import select, func, col -from .db import ApiKey, AsyncSession +from .db import ApiKey, AsyncSession, get_session RECEIVE_LN_ADDRESS = os.environ["RECEIVE_LN_ADDRESS"] MINT = os.environ.get("MINT", "https://mint.minibits.cash/Bitcoin") @@ -19,7 +21,7 @@ WALLET = None async def _initialize_wallet(mint_url: str | None = None) -> Wallet: """Initializes and loads a Cashu wallet.""" global WALLET - if WALLET is not None and WALLET.mint_url == mint_url: # only return existing wallet if the mint_url matches the one of current WALLET + if WALLET is not None: return WALLET if mint_url is None: mint_url = MINT @@ -85,6 +87,25 @@ async def _pay_invoice_with_cashu( return quote.amount +async def check_for_refunds() -> None: + while True: + try: + async for session in get_session(): + result = await session.exec(select(ApiKey)) + keys = result.all() + + for key in keys: + if(key.balance > 0 and key.refund_address and key.key_expiry_time and key.key_expiry_time < time.time()): + print(f"Refunding key {key.hashed_key}, {time.time()=}, {key.key_expiry_time=}") + await refund_balance(key.balance, key, session) + #TODO Error balance to low + + except Exception as e: + print(f"Error during refund check: {e}") + #TODO Define time to sleep # hour: 3600 + await asyncio.sleep(10) + + async def pay_out(session: AsyncSession) -> None: """ Calculates the pay-out amount based on the spent balance, profit, and donation rate. @@ -123,13 +144,22 @@ async def pay_out(session: AsyncSession) -> None: async def credit_balance(cashu_token: str, key: ApiKey, session: AsyncSession) -> int: token_obj: Token = deserialize_token_from_string(cashu_token) - token_mint = await _initialize_wallet(token_obj.mint) + # Initialize the wallet with the mint specified in the token + print(f"Trying to credit token from mint: {token_obj.mint}", flush=True) wallet: Wallet = await _initialize_wallet(token_obj.mint) - amount_msats = await _handle_token_receive(wallet, token_obj) - key.balance += amount_msats - session.add(key) - await session.commit() - return amount_msats + if token_obj.mint == MINT: + # crediting a token created using the same mint as specified in .env + print("Received a token from the same mint", flush=True) + + amount_msats = await _handle_token_receive(wallet, token_obj) + key.balance += amount_msats + session.add(key) + await session.commit() + return amount_msats + else: + # crediting a token created using a different mint as specified in .env + print("Received a token from a different mint", flush=True) + #TODO async def refund_balance(amount: int, key: ApiKey, session: AsyncSession) -> int: @@ -143,7 +173,7 @@ async def refund_balance(amount: int, key: ApiKey, session: AsyncSession) -> int await session.commit() if key.refund_address is None: raise ValueError("Refund address not set.") - return await send_to_lnurl(wallet, key.refund_address, amount_msat=amount) # todo msats / sats conversion error? + return await send_to_lnurl(wallet, key.refund_address, amount_msat=amount * 1000) # todo msats / sats conversion error? async def create_token( diff --git a/router/db.py b/router/db.py index dec53645..10e45d8b 100644 --- a/router/db.py +++ b/router/db.py @@ -21,6 +21,10 @@ class ApiKey(SQLModel, table=True): # type: ignore default=None, description="Lightning address to refund remaining balance after key expires", ) + key_expiry_time: int | None = Field( + default=None, + description="Unix-timestamp after which the cashu-token's balance gets refunded to the refund_address", + ) total_spent: int = Field( default=0, description="Total spent in millisatoshis (msats)" ) diff --git a/router/main.py b/router/main.py index 7ae8e71d..39c53bc5 100644 --- a/router/main.py +++ b/router/main.py @@ -7,9 +7,10 @@ from .db import init_db from .admin import admin_router from .proxy import proxy_router from .account import account_router -from .cashu import _initialize_wallet +from .cashu import _initialize_wallet, check_for_refunds from .models import MODELS, update_sats_pricing + __version__ = "0.0.1" app = FastAPI( @@ -53,3 +54,4 @@ async def startup_event(): await init_db() await _initialize_wallet() asyncio.create_task(update_sats_pricing()) + asyncio.create_task(check_for_refunds()) diff --git a/router/proxy.py b/router/proxy.py index c70d3faf..e8f610de 100644 --- a/router/proxy.py +++ b/router/proxy.py @@ -25,14 +25,36 @@ async def proxy( ): auth = request.headers.get("Authorization", "") bearer_key = auth.replace("Bearer ", "") if auth.startswith("Bearer ") else "" + refund_address = request.headers.get("Refund-LNURL", None) + key_expiry_time = request.headers.get("Key-Expiry-Time", None) - key = await validate_bearer_key(bearer_key, session) + if key_expiry_time: + try: + key_expiry_time = int(key_expiry_time) + except ValueError: + return Response( + content="Invalid Key-Expiry-Time: must be a valid Unix timestamp", + status_code=400, + ) + else: + key_expiry_time = None + + if(key_expiry_time and not refund_address): + return Response( + content=f"Error: Refund-LNURL header required when using Key-Expiry-Time", + status_code=400, + ) + + + key = await validate_bearer_key(bearer_key, session, refund_address, key_expiry_time) await pay_for_request(key, session) # Prepare headers, removing sensitive/problematic ones headers = dict(request.headers) headers.pop("host", None) headers.pop("content-length", None) + headers.pop("refund-lnurl", None) + headers.pop("key-expiry-time", None) if UPSTREAM_API_KEY: headers["Authorization"] = f"Bearer {UPSTREAM_API_KEY}" From 9503c590228a159a0f7d371ce542af5dfa74dbcd Mon Sep 17 00:00:00 2001 From: GitHappens2Me Date: Tue, 27 May 2025 13:47:02 +0200 Subject: [PATCH 2/8] fixed conversion error with automatic refunds, improved admin interface (human-readable refund date and correct owner balance) --- router/admin.py | 40 ++++++++++++++++++++++----- router/cashu.py | 73 +++++++++++++++++++++++++++++-------------------- 2 files changed, 77 insertions(+), 36 deletions(-) diff --git a/router/admin.py b/router/admin.py index 7d33cd03..ca9a20db 100644 --- a/router/admin.py +++ b/router/admin.py @@ -1,6 +1,10 @@ import os +from datetime import datetime, timezone + + from fastapi import APIRouter, Request from fastapi.responses import HTMLResponse + from .cashu import _initialize_wallet admin_router = APIRouter(prefix="/admin") @@ -90,14 +94,33 @@ async def dashboard(request: Request) -> str: async with create_session() as session: result = await session.exec(select(ApiKey)) api_keys = result.all() - api_keys_table_rows = "".join( - f"{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{key.key_expiry_time}" - for key in api_keys - ) + + api_keys_table_rows = [] + for key in api_keys: + if key.key_expiry_time is not None: + expiry_time_utc = datetime.fromtimestamp(key.key_expiry_time, tz=timezone.utc) + expiry_time_human_readable = expiry_time_utc.strftime('%Y-%m-%d %H:%M:%S') + api_keys_table_rows.append( + f"{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{key.key_expiry_time} ({expiry_time_human_readable} UTC)" + ) + else: + expiry_time_human_readable = "" + api_keys_table_rows.append( + f"{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{key.key_expiry_time}" + ) + + api_keys_table_rows = "".join(api_keys_table_rows) + + + # Calculate the total balance of all API keys + total_user_balance = int(sum(key.balance / 1000 for key in api_keys)) # Fetch balance from cashu wallet = await _initialize_wallet() - current_balance = wallet.balance # Not awaited, assuming it's a property + wallet_balance = wallet.balance + + # calculate owner balance + owner_balance = wallet_balance - total_user_balance return f""" @@ -117,8 +140,11 @@ async def dashboard(request: Request) -> str:

Admin Dashboard

-

Current Cashu Balance (including user balances)

-

{current_balance} sats

+

Current Cashu Balance

+

Your Balance: {owner_balance} sats

+

The balance is calculated by subtracting the combined user balance from the total Cashu wallet balance.

+

Total Cashu Balance: {wallet_balance} sats

+

User Balance: {total_user_balance} sats

User's API Keys

diff --git a/router/cashu.py b/router/cashu.py index ef2005f2..b86f4353 100644 --- a/router/cashu.py +++ b/router/cashu.py @@ -87,25 +87,6 @@ async def _pay_invoice_with_cashu( return quote.amount -async def check_for_refunds() -> None: - while True: - try: - async for session in get_session(): - result = await session.exec(select(ApiKey)) - keys = result.all() - - for key in keys: - if(key.balance > 0 and key.refund_address and key.key_expiry_time and key.key_expiry_time < time.time()): - print(f"Refunding key {key.hashed_key}, {time.time()=}, {key.key_expiry_time=}") - await refund_balance(key.balance, key, session) - #TODO Error balance to low - - except Exception as e: - print(f"Error during refund check: {e}") - #TODO Define time to sleep # hour: 3600 - await asyncio.sleep(10) - - async def pay_out(session: AsyncSession) -> None: """ Calculates the pay-out amount based on the spent balance, profit, and donation rate. @@ -121,10 +102,8 @@ async def pay_out(session: AsyncSession) -> None: wallet = await _initialize_wallet() wallet_balance = wallet.available_balance - print(f"Wallet-balance: {wallet_balance}, User-balance: {user_balance}, Revenue: {wallet_balance - user_balance}, MinPayout:{MINIMUM_PAYOUT}", flush=True) - # Why is that bad? - #assert wallet_balance <= user_balance, f"Something went deeply wrong. Wallet-balance: {wallet_balance}, User-Balance: {user_balance}" + assert wallet_balance >= user_balance, f"Something went deeply wrong. Wallet-balance: {wallet_balance}, User-Balance: {user_balance}" if (revenue := wallet_balance - user_balance) <= MINIMUM_PAYOUT: return @@ -132,6 +111,7 @@ async def pay_out(session: AsyncSession) -> None: owners_draw = revenue - devs_donation + print(f" DEBUG Revenue > Minimum Payout: paying {owners_draw} sats to {RECEIVE_LN_ADDRESS}", flush=True) await send_to_lnurl(wallet, RECEIVE_LN_ADDRESS, owners_draw * 1000) # conversion to msats for send_to_lnurl if devs_donation > 0: @@ -162,7 +142,42 @@ async def credit_balance(cashu_token: str, key: ApiKey, session: AsyncSession) - #TODO +# TODO i think the wallet.balance is not updated correctly after refunds +# TODO what happens if minimum payout gets reached by a request that needs to be refunded +async def check_for_refunds() -> None: + while True: + try: + async for session in get_session(): + result = await session.exec(select(ApiKey)) + keys = result.all() + + for key in keys: + if(key.balance > 0 and key.refund_address and key.key_expiry_time and key.key_expiry_time < time.time()): + print(f" DEBUG Refunding key {key.hashed_key[:3] + '[...]' + key.hashed_key[-3:]}, Current Time: {int(time.time())}, Expirary Time: {key.key_expiry_time}", flush = True) + await refund_balance(key.balance, key, session) + #TODO Error balance to low + await asyncio.sleep(30) + #TODO Define time to sleep # hour: 3600 + except Exception as e: + print(f"Error during refund check: {e}") + + + async def refund_balance(amount: int, key: ApiKey, session: AsyncSession) -> int: + """ + Refunds the specified amount from an API key's balance to the key's refund address. + + Args: + amount (int): The amount to refund in millisatoshis. + key (ApiKey): The API key object containing balance and refund address. + session (AsyncSession): The database session for committing changes. + + Returns: + int: The amount in millisatoshis that was successfully sent. + + Raises: + ValueError: If balance is insufficient or refund address is not set. + """ wallet = await _initialize_wallet() if key.balance < amount: raise ValueError("Insufficient balance.") @@ -173,8 +188,7 @@ async def refund_balance(amount: int, key: ApiKey, session: AsyncSession) -> int await session.commit() if key.refund_address is None: raise ValueError("Refund address not set.") - return await send_to_lnurl(wallet, key.refund_address, amount_msat=amount * 1000) # todo msats / sats conversion error? - + return await send_to_lnurl(wallet, key.refund_address, amount_msat=amount) async def create_token( amount_msats: int, mint: str = MINT @@ -244,19 +258,20 @@ async def send_to_lnurl(wallet: Wallet, lnurl: str, amount_msat: int) -> int: f"({min_sendable / 1000} - {max_sendable / 1000} sat)." ) # subtract estimated fees - amount_to_send = amount_msat - int(max(5000, amount_msat * 0.01)) + # TODO: Is a static fee calculation working well? + # moving the 2000 and 0.01 to optional enviroment variables gives more control to users + amount_to_send = amount_msat - int(max(2000, amount_msat * 0.01)) + print(f" DEBUG Trying to pay {amount_to_send} msats to {lnurl}, with Wallet balance = {wallet.balance}", flush = True) - print(f"trying to pay {amount_to_send} msats to {lnurl}", flush=True) - print(f"Available balance: {wallet.balance}", flush = True ) # Note: We pass amount_msat directly. The actual amount paid might be adjusted # slightly by the melt quote based on the invoice details. bolt11_invoice, _ = await _get_lnurl_invoice(callback_url, amount_to_send) - # Conversion to Sats (/ 1000 necessary for cashu payments) + # Conversion to Sats (/ 1000) necessary for cashu payments amount_paid = await _pay_invoice_with_cashu(wallet, bolt11_invoice, amount_to_send / 1000) - print(f"{amount_paid} sats paid to lnurl", flush=True) + print(f" DEBUG {amount_paid} sats paid to lnurl", flush=True) return amount_paid From 6c0ca7f697b94369b26da82f62e19f69f15b0ecc Mon Sep 17 00:00:00 2001 From: shroominic Date: Wed, 28 May 2025 11:57:01 +0000 Subject: [PATCH 3/8] ai fix tests --- tests/conftest.py | 2 +- tests/test_account.py | 1 - tests/test_main.py | 24 ++++++++++-------------- tests/test_models.py | 39 ++++++++++++++++++++++++++++++--------- tests/test_proxy.py | 21 +++++++++++++++++++-- 5 files changed, 60 insertions(+), 27 deletions(-) diff --git a/tests/conftest.py b/tests/conftest.py index ed98bc35..c5ca43ff 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -97,7 +97,7 @@ def test_client() -> TestClient: with patch("router.models.update_sats_pricing") as mock_update: mock_update.return_value = None - return TestClient(app) + yield TestClient(app) @pytest_asyncio.fixture diff --git a/tests/test_account.py b/tests/test_account.py index bdcb8f94..9105db3f 100644 --- a/tests/test_account.py +++ b/tests/test_account.py @@ -188,7 +188,6 @@ async def test_account_with_cashu_token( ): """Test authentication with a cashu token creates a new account.""" cashu_token = "cashuBqQSEQ123456" - hashed = hash_api_key(cashu_token) with patch("router.cashu.credit_balance", new_callable=AsyncMock) as mock_credit: # Mock successful token redemption diff --git a/tests/test_main.py b/tests/test_main.py index 0a66632a..cee18b37 100644 --- a/tests/test_main.py +++ b/tests/test_main.py @@ -7,20 +7,16 @@ from unittest.mock import patch async def test_root_endpoint(async_client: AsyncClient): """Test the root endpoint returns expected information.""" # Mock the environment variables for this specific test - with patch("os.environ.get") as mock_env_get: - def env_side_effect(key, default=None): - env_map = { - "NAME": "TestRoutstrNode", - "DESCRIPTION": "Test Node", - "NPUB": "npub1test", - "MINT": "https://test.mint.com", - "HTTP_URL": "http://test.example.com", - "ONION_URL": "http://test.onion", - } - return env_map.get(key, default) - - mock_env_get.side_effect = env_side_effect - + env_vars = { + "NAME": "TestRoutstrNode", + "DESCRIPTION": "Test Node", + "NPUB": "npub1test", + "MINT": "https://test.mint.com", + "HTTP_URL": "http://test.example.com", + "ONION_URL": "http://test.onion", + } + + with patch.dict("os.environ", env_vars, clear=False): response = await async_client.get("/") assert response.status_code == 200 diff --git a/tests/test_models.py b/tests/test_models.py index d28a2f2f..c4a8494e 100644 --- a/tests/test_models.py +++ b/tests/test_models.py @@ -67,14 +67,21 @@ async def test_update_sats_pricing_calculation(sample_model: Model): assert sample_model.sats_pricing is not None # Verify calculations (prices in USD / sats_to_usd) - assert sample_model.sats_pricing.prompt == 0.01 / 0.0001 # 100 sats - assert sample_model.sats_pricing.completion == 0.02 / 0.0001 # 200 sats - assert sample_model.sats_pricing.request == 0.001 / 0.0001 # 10 sats + assert sample_model.sats_pricing.prompt == pytest.approx(0.01 / 0.0001) # 100 sats + assert sample_model.sats_pricing.completion == pytest.approx(0.02 / 0.0001) # 200 sats + assert sample_model.sats_pricing.request == pytest.approx(0.001 / 0.0001) # 10 sats # Verify max_cost calculation for model with top_provider expected_max_context = 4096 * sample_model.sats_pricing.prompt expected_max_completion = 2048 * sample_model.sats_pricing.completion - assert sample_model.sats_pricing.max_cost == expected_max_context + expected_max_completion + assert sample_model.sats_pricing.max_cost == pytest.approx(expected_max_context + expected_max_completion) + + # Cancel and await the task + task.cancel() + try: + await task + except asyncio.CancelledError: + pass except asyncio.CancelledError: pass @@ -140,7 +147,14 @@ async def test_update_sats_pricing_without_top_provider(): ir = model_without_top.sats_pricing.internal_reasoning * 100 expected_max = p + c + r + i + w + ir - assert model_without_top.sats_pricing.max_cost == expected_max + assert model_without_top.sats_pricing.max_cost == pytest.approx(expected_max) + + # Cancel and await the task + task.cancel() + try: + await task + except asyncio.CancelledError: + pass except asyncio.CancelledError: pass @@ -158,11 +172,11 @@ async def test_update_sats_pricing_handles_errors(): error_printed = False original_print = print - def mock_print(msg): + def mock_print(*args, **kwargs): nonlocal error_printed - if isinstance(msg, Exception) and str(msg) == "API Error": + if args and isinstance(args[0], Exception) and str(args[0]) == "API Error": error_printed = True - original_print(msg) + original_print(*args, **kwargs) with patch("builtins.print", side_effect=mock_print): sleep_called = asyncio.Event() @@ -179,6 +193,13 @@ async def test_update_sats_pricing_handles_errors(): # Verify error was printed assert error_printed + # Cancel and await the task + task.cancel() + try: + await task + except asyncio.CancelledError: + pass + except asyncio.CancelledError: pass @@ -197,4 +218,4 @@ def test_model_serialization(sample_model: Model): # Test deserialization new_model = Model(**model_dict) assert new_model.id == sample_model.id - assert new_model.pricing.prompt == sample_model.pricing.prompt \ No newline at end of file + assert new_model.pricing.prompt == pytest.approx(sample_model.pricing.prompt) \ No newline at end of file diff --git a/tests/test_proxy.py b/tests/test_proxy.py index 603eb522..7386b5c9 100644 --- a/tests/test_proxy.py +++ b/tests/test_proxy.py @@ -113,6 +113,10 @@ async def test_proxy_successful_request_mock( mock_client = AsyncMock() mock_client_class.return_value = mock_client + # Add async context manager methods + mock_client.__aenter__ = AsyncMock(return_value=mock_client) + mock_client.__aexit__ = AsyncMock() + # Create a mock response mock_response = AsyncMock() mock_response.status_code = 200 @@ -175,10 +179,14 @@ async def test_proxy_streaming_response( mock_client = AsyncMock() mock_client_class.return_value = mock_client + # Add async context manager methods + mock_client.__aenter__ = AsyncMock(return_value=mock_client) + mock_client.__aexit__ = AsyncMock() + mock_response = AsyncMock() mock_response.status_code = 200 mock_response.headers = {"content-type": "text/event-stream"} - mock_response.aiter_bytes = mock_aiter_bytes + mock_response.aiter_bytes = lambda: mock_aiter_bytes() mock_response.aclose = AsyncMock() mock_client.send = AsyncMock(return_value=mock_response) @@ -213,6 +221,10 @@ async def test_proxy_handles_upstream_errors( mock_client = AsyncMock() mock_client_class.return_value = mock_client + # Add async context manager methods + mock_client.__aenter__ = AsyncMock(return_value=mock_client) + mock_client.__aexit__ = AsyncMock() + # Simulate connection error mock_client.send.side_effect = Exception("Connection refused") mock_client.build_request = AsyncMock() @@ -252,7 +264,8 @@ async def test_proxy_with_model_based_pricing( test_session.add(key) await test_session.commit() - with patch.dict(os.environ, {"MODEL_BASED_PRICING": "true"}): + # Patch the MODEL_BASED_PRICING constant directly + with patch("router.auth.MODEL_BASED_PRICING", True): with patch("os.path.exists", return_value=True): # Mock a model with pricing from router.models import MODELS, Model, Pricing, Architecture, TopProvider @@ -304,6 +317,10 @@ async def test_proxy_with_model_based_pricing( mock_client = AsyncMock() mock_client_class.return_value = mock_client + # Add async context manager methods + mock_client.__aenter__ = AsyncMock(return_value=mock_client) + mock_client.__aexit__ = AsyncMock() + # Create a mock response mock_response = AsyncMock() mock_response.status_code = 200 From 0102aefd96d621e9ed32dfdadb9a22cb7c592eca Mon Sep 17 00:00:00 2001 From: GitHappens2Me Date: Wed, 28 May 2025 18:58:40 +0200 Subject: [PATCH 4/8] simplified key_expiry_time header logic --- .gitignore | 2 +- router/proxy.py | 10 ++++------ 2 files changed, 5 insertions(+), 7 deletions(-) diff --git a/.gitignore b/.gitignore index b7f6d083..2a5e2170 100644 --- a/.gitignore +++ b/.gitignore @@ -7,5 +7,5 @@ wallet.sqlite3 .notes .*keys.db .*wallet.sqlite3 -.models.json +.*models.json compose.override.yml \ No newline at end of file diff --git a/router/proxy.py b/router/proxy.py index 84ae24b9..9b06125e 100644 --- a/router/proxy.py +++ b/router/proxy.py @@ -37,16 +37,14 @@ async def proxy( content="Invalid Key-Expiry-Time: must be a valid Unix timestamp", status_code=400, ) - else: - key_expiry_time = None - - if(key_expiry_time and not refund_address): - return Response( + if(not refund_address): + return Response( content=f"Error: Refund-LNURL header required when using Key-Expiry-Time", status_code=400, ) + else: + key_expiry_time = None - key = await validate_bearer_key(bearer_key, session, refund_address, key_expiry_time) # Pre-validate JSON for requests that require it From b76414dc00359196c638fed68362f084c67b54e7 Mon Sep 17 00:00:00 2001 From: GitHappens2Me Date: Wed, 28 May 2025 20:30:24 +0200 Subject: [PATCH 5/8] clean up and .env variables --- .env.example | 8 ++++++++ router/admin.py | 23 ++++++----------------- router/cashu.py | 35 +++++++++++++++++++++++------------ router/proxy.py | 1 - 4 files changed, 37 insertions(+), 30 deletions(-) diff --git a/.env.example b/.env.example index 9e402b08..3dcf8ec9 100644 --- a/.env.example +++ b/.env.example @@ -9,6 +9,9 @@ UPSTREAM_API_KEY="sk-21212121212121212121212121212121" # Lightning address used to receive funds RECEIVE_LN_ADDRESS="shroominic@walletofsatoshi.com" +# When your cashu balance reaches this number of sats, send the funds to RECEIVE_LN_ADDRESS. +MINIMUM_PAYOUT = "100" + # Costs in Sats, if MODEL_BASED_PRICING is set to false COST_PER_REQUEST="10" COST_PER_1K_INPUT_TOKENS = "0" @@ -17,6 +20,11 @@ COST_PER_1K_OUTPUT_TOKENS = "0" # If set to true, make sure model pricings are defined in models.json MODEL_BASED_PRICING = "false" +# Time in seconds between each automatically refunding funds to users whose API keys have expired +# Setting this to "0" disables automatic refunds +REFUND_PROCESSING_INTERVAL = "3600" + + # password used to log into admin interface ADMIN_PASSWORD="XXX" diff --git a/router/admin.py b/router/admin.py index ca9a20db..5cfb7a5d 100644 --- a/router/admin.py +++ b/router/admin.py @@ -1,7 +1,6 @@ import os from datetime import datetime, timezone - from fastapi import APIRouter, Request from fastapi.responses import HTMLResponse @@ -9,7 +8,6 @@ from .cashu import _initialize_wallet admin_router = APIRouter(prefix="/admin") - def login_form() -> str: return """ @@ -97,32 +95,23 @@ async def dashboard(request: Request) -> str: api_keys_table_rows = [] for key in api_keys: - if key.key_expiry_time is not None: - expiry_time_utc = datetime.fromtimestamp(key.key_expiry_time, tz=timezone.utc) - expiry_time_human_readable = expiry_time_utc.strftime('%Y-%m-%d %H:%M:%S') - api_keys_table_rows.append( - f"" - ) - else: - expiry_time_human_readable = "" - api_keys_table_rows.append( - f"" - ) + expiry_time_utc = datetime.fromtimestamp(key.key_expiry_time, tz=timezone.utc) if key.key_expiry_time is not None else None + expiry_time_human_readable = expiry_time_utc.strftime('%Y-%m-%d %H:%M:%S') if expiry_time_utc else "" + + api_keys_table_rows.append( + f"" + ) api_keys_table_rows = "".join(api_keys_table_rows) - # Calculate the total balance of all API keys total_user_balance = int(sum(key.balance / 1000 for key in api_keys)) - # Fetch balance from cashu wallet = await _initialize_wallet() wallet_balance = wallet.balance - # calculate owner balance owner_balance = wallet_balance - total_user_balance - return f""" diff --git a/router/cashu.py b/router/cashu.py index 6d93c47b..41f8e580 100644 --- a/router/cashu.py +++ b/router/cashu.py @@ -12,6 +12,7 @@ from .db import ApiKey, AsyncSession, get_session RECEIVE_LN_ADDRESS = os.environ["RECEIVE_LN_ADDRESS"] MINT = os.environ.get("MINT", "https://mint.minibits.cash/Bitcoin") MINIMUM_PAYOUT = int(os.environ.get("MINIMUM_PAYOUT", 100)) +REFUND_PROCESSING_INTERVAL = int(os.environ.get("REFUND_PROCESSING_INTERVAL", 3600)) DEV_LN_ADDRESS = "routstr@minibits.cash" DEVS_DONATION_RATE = 0.021 # 2.1% WALLET = None @@ -81,7 +82,9 @@ async def _pay_invoice_with_cashu( proofs_to_melt, _ = await wallet.select_to_send( wallet.proofs, quote.amount + quote.fee_reserve ) - print(f"Proofs to melt: {proofs_to_melt}") + + # Debugging Cashu Proofs + #print(f"Proofs to melt: {proofs_to_melt}") _ = await wallet.melt( proofs_to_melt, bolt11_invoice, quote.fee_reserve, quote.quote @@ -164,25 +167,35 @@ async def credit_balance(cashu_token: str, key: ApiKey, session: AsyncSession) - else: # crediting a token created using a different mint as specified in .env print("Received a token from a different mint", flush=True) - #TODO + #TODO This fails, and needs to be fixed -# TODO i think the wallet.balance is not updated correctly after refunds -# TODO what happens if minimum payout gets reached by a request that needs to be refunded async def check_for_refunds() -> None: + """ + Periodically checks for API keys that are eligible for refunds and processes them. + + Raises: + Exception: If an error occurs during the refund check process. + """ + + # Setting REFUND_PROCESSING_INTERVAL to 0 disables it + if REFUND_PROCESSING_INTERVAL == 0: + print("Automatic refund processing is disabled.") + return + while True: try: async for session in get_session(): result = await session.exec(select(ApiKey)) keys = result.all() - + current_time = int(time.time()) for key in keys: - if(key.balance > 0 and key.refund_address and key.key_expiry_time and key.key_expiry_time < time.time()): - print(f" DEBUG Refunding key {key.hashed_key[:3] + '[...]' + key.hashed_key[-3:]}, Current Time: {int(time.time())}, Expirary Time: {key.key_expiry_time}", flush = True) + if(key.balance > 0 and key.refund_address and key.key_expiry_time and key.key_expiry_time < current_time): + print(f" DEBUG Refunding key {key.hashed_key[:3] + '[...]' + key.hashed_key[-3:]}, Current Time: {current_time}, Expirary Time: {key.key_expiry_time}", flush = True) await refund_balance(key.balance, key, session) - #TODO Error balance to low - await asyncio.sleep(30) - #TODO Define time to sleep # hour: 3600 + + # Sleep for the specified interval before checking again + await asyncio.sleep(REFUND_PROCESSING_INTERVAL) except Exception as e: print(f"Error during refund check: {e}") @@ -290,7 +303,6 @@ async def send_to_lnurl(wallet: Wallet, lnurl: str, amount_msat: int) -> int: print(f" DEBUG Trying to pay {amount_to_send} msats to {lnurl}, with Wallet balance = {wallet.balance}", flush = True) - print(f"trying to pay {amount_to_send} msats to {lnurl}. Available balance: {wallet.balance}", flush=True) # Note: We pass amount_msat directly. The actual amount paid might be adjusted # slightly by the melt quote based on the invoice details. bolt11_invoice, _ = await _get_lnurl_invoice(callback_url, amount_to_send) @@ -300,7 +312,6 @@ async def send_to_lnurl(wallet: Wallet, lnurl: str, amount_msat: int) -> int: print(f" DEBUG {amount_paid} sats paid to lnurl", flush=True) - print(f"Amount paid: {amount_paid / 1000} sat") return amount_paid diff --git a/router/proxy.py b/router/proxy.py index 9b06125e..99f4fe7f 100644 --- a/router/proxy.py +++ b/router/proxy.py @@ -193,7 +193,6 @@ async def proxy( background_tasks = BackgroundTasks() background_tasks.add_task(response.aclose) background_tasks.add_task(client.aclose) - return StreamingResponse( stream_with_cost(), status_code=response.status_code, From 7f578cb980c85a840347130eccdc5475fd29cbf8 Mon Sep 17 00:00:00 2001 From: shroominic Date: Thu, 29 May 2025 12:42:59 +0000 Subject: [PATCH 6/8] update models --- models.json | 588 +++++++++++++++++++++++++--------------------------- 1 file changed, 287 insertions(+), 301 deletions(-) diff --git a/models.json b/models.json index acca2523..6f05d95b 100644 --- a/models.json +++ b/models.json @@ -1,5 +1,103 @@ { "models": [ + { + "id": "google/gemma-2b-it", + "hugging_face_id": "google/gemma-2b-it", + "name": "Google: Gemma 2 2B", + "created": 1748460815, + "description": "Gemma 2 2B by Google is an open model built from the same research and technology used to create the [Gemini models](/models?q=gemini).\n\nGemma models are well-suited for a variety of text generation tasks, including question answering, summarization, and reasoning.\n\nSee the [launch announcement](https://blog.google/technology/developers/google-gemma-2/) for more details. Usage of Gemma is subject to Google's [Gemma Terms of Use](https://ai.google.dev/gemma/terms).", + "context_length": 8192, + "architecture": { + "modality": "text->text", + "input_modalities": [ + "text" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "Gemini", + "instruct_type": "gemma" + }, + "pricing": { + "prompt": "0.0000001", + "completion": "0.0000001", + "request": "0", + "image": "0", + "web_search": "0", + "internal_reasoning": "0" + }, + "top_provider": { + "context_length": 8192, + "max_completion_tokens": null, + "is_moderated": false + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "top_p", + "stop", + "frequency_penalty", + "presence_penalty", + "top_k", + "repetition_penalty", + "logit_bias", + "min_p", + "response_format" + ] + }, + { + "id": "deepseek/deepseek-r1-0528", + "hugging_face_id": "deepseek-ai/DeepSeek-R1-0528", + "name": "DeepSeek: R1 0528", + "created": 1748455170, + "description": "May 28th update to the [original DeepSeek R1](/deepseek/deepseek-r1) Performance on par with [OpenAI o1](/openai/o1), but open-sourced and with fully open reasoning tokens. It's 671B parameters in size, with 37B active in an inference pass.\n\nFully open-source model.", + "context_length": 163840, + "architecture": { + "modality": "text->text", + "input_modalities": [ + "text" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "DeepSeek", + "instruct_type": "deepseek-r1" + }, + "pricing": { + "prompt": "0.0000005", + "completion": "0.00000218", + "request": "0", + "image": "0", + "web_search": "0", + "internal_reasoning": "0" + }, + "top_provider": { + "context_length": 163840, + "max_completion_tokens": null, + "is_moderated": false + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "top_p", + "reasoning", + "include_reasoning", + "presence_penalty", + "frequency_penalty", + "repetition_penalty", + "top_k", + "stop", + "seed", + "min_p", + "logit_bias", + "top_logprobs", + "logprobs", + "response_format", + "structured_outputs" + ] + }, { "id": "sarvamai/sarvam-m", "hugging_face_id": "sarvamai/sarvam-m", @@ -165,7 +263,7 @@ "top_provider": { "context_length": 200000, "max_completion_tokens": 64000, - "is_moderated": true + "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ @@ -1069,7 +1167,7 @@ }, "top_provider": { "context_length": 128000, - "max_completion_tokens": null, + "max_completion_tokens": 20000, "is_moderated": false }, "per_request_limits": null, @@ -1175,20 +1273,20 @@ }, "per_request_limits": null, "supported_parameters": [ - "tools", - "tool_choice", "max_tokens", "temperature", "top_p", "reasoning", "include_reasoning", + "seed", + "tools", + "tool_choice", "stop", "frequency_penalty", "presence_penalty", "repetition_penalty", "response_format", "top_k", - "seed", "min_p", "structured_outputs", "logprobs", @@ -1234,20 +1332,20 @@ "top_p", "reasoning", "include_reasoning", - "presence_penalty", + "stop", "frequency_penalty", - "repetition_penalty", + "presence_penalty", "top_k", + "repetition_penalty", + "logit_bias", + "min_p", + "response_format", + "seed", "tools", "tool_choice", - "stop", - "response_format", "structured_outputs", - "logit_bias", "logprobs", - "top_logprobs", - "seed", - "min_p" + "top_logprobs" ] }, { @@ -1278,7 +1376,7 @@ }, "top_provider": { "context_length": 32000, - "max_completion_tokens": null, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -1326,7 +1424,7 @@ }, "top_provider": { "context_length": 32000, - "max_completion_tokens": null, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -1374,7 +1472,7 @@ }, "top_provider": { "context_length": 32000, - "max_completion_tokens": null, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -1568,10 +1666,18 @@ "supported_parameters": [ "tools", "tool_choice", - "seed", "max_tokens", + "reasoning", + "include_reasoning", + "structured_outputs", "response_format", - "structured_outputs" + "stop", + "frequency_penalty", + "presence_penalty", + "seed", + "logit_bias", + "logprobs", + "top_logprobs" ] }, { @@ -1617,52 +1723,6 @@ "structured_outputs" ] }, - { - "id": "qwen/qwen2.5-coder-7b-instruct", - "hugging_face_id": "Qwen/Qwen2.5-Coder-7B-Instruct", - "name": "Qwen: Qwen2.5 Coder 7B Instruct", - "created": 1744734887, - "description": "Qwen2.5-Coder-7B-Instruct is a 7B parameter instruction-tuned language model optimized for code-related tasks such as code generation, reasoning, and bug fixing. Based on the Qwen2.5 architecture, it incorporates enhancements like RoPE, SwiGLU, RMSNorm, and GQA attention with support for up to 128K tokens using YaRN-based extrapolation. It is trained on a large corpus of source code, synthetic data, and text-code grounding, providing robust performance across programming languages and agentic coding workflows.\n\nThis model is part of the Qwen2.5-Coder family and offers strong compatibility with tools like vLLM for efficient deployment. Released under the Apache 2.0 license.", - "context_length": 32768, - "architecture": { - "modality": "text->text", - "input_modalities": [ - "text" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Qwen", - "instruct_type": null - }, - "pricing": { - "prompt": "0.00000001", - "completion": "0.00000003", - "request": "0", - "image": "0", - "web_search": "0", - "internal_reasoning": "0" - }, - "top_provider": { - "context_length": 32768, - "max_completion_tokens": null, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ - "max_tokens", - "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" - ] - }, { "id": "openai/gpt-4.1", "hugging_face_id": "", @@ -2051,6 +2111,54 @@ "top_logprobs" ] }, + { + "id": "nvidia/llama-3.1-nemotron-ultra-253b-v1", + "hugging_face_id": "nvidia/Llama-3_1-Nemotron-Ultra-253B-v1", + "name": "NVIDIA: Llama 3.1 Nemotron Ultra 253B v1", + "created": 1744115059, + "description": "Llama-3.1-Nemotron-Ultra-253B-v1 is a large language model (LLM) optimized for advanced reasoning, human-interactive chat, retrieval-augmented generation (RAG), and tool-calling tasks. Derived from Meta\u2019s Llama-3.1-405B-Instruct, it has been significantly customized using Neural Architecture Search (NAS), resulting in enhanced efficiency, reduced memory usage, and improved inference latency. The model supports a context length of up to 128K tokens and can operate efficiently on an 8x NVIDIA H100 node.\n\nNote: you must include `detailed thinking on` in the system prompt to enable reasoning. Please see [Usage Recommendations](https://huggingface.co/nvidia/Llama-3_1-Nemotron-Ultra-253B-v1#quick-start-and-usage-recommendations) for more.", + "context_length": 131072, + "architecture": { + "modality": "text->text", + "input_modalities": [ + "text" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "Llama3", + "instruct_type": null + }, + "pricing": { + "prompt": "0.0000006", + "completion": "0.0000018", + "request": "0", + "image": "0", + "web_search": "0", + "internal_reasoning": "0" + }, + "top_provider": { + "context_length": 131072, + "max_completion_tokens": null, + "is_moderated": false + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "top_p", + "reasoning", + "include_reasoning", + "stop", + "frequency_penalty", + "presence_penalty", + "seed", + "top_k", + "logit_bias", + "logprobs", + "top_logprobs" + ] + }, { "id": "meta-llama/llama-4-maverick", "hugging_face_id": "meta-llama/Llama-4-Maverick-17B-128E-Instruct", @@ -2362,8 +2470,8 @@ "instruct_type": null }, "pricing": { - "prompt": "0.0000008", - "completion": "0.0000008", + "prompt": "0.0000009", + "completion": "0.0000009", "request": "0", "image": "0", "web_search": "0", @@ -2371,7 +2479,7 @@ }, "top_provider": { "context_length": 128000, - "max_completion_tokens": 128000, + "max_completion_tokens": null, "is_moderated": false }, "per_request_limits": null, @@ -3204,8 +3312,6 @@ "logprobs", "top_logprobs", "seed", - "tools", - "tool_choice", "structured_outputs" ] }, @@ -3344,62 +3450,15 @@ }, "per_request_limits": null, "supported_parameters": [ - "tools", - "tool_choice", "max_tokens", "temperature", - "top_p", + "stop", "reasoning", "include_reasoning", - "top_k", - "stop" - ] - }, - { - "id": "anthropic/claude-3.7-sonnet:thinking", - "hugging_face_id": "", - "name": "Anthropic: Claude 3.7 Sonnet (thinking)", - "created": 1740422110, - "description": "Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and extended, step-by-step processing for complex tasks. The model demonstrates notable improvements in coding, particularly in front-end development and full-stack updates, and excels in agentic workflows, where it can autonomously navigate multi-step processes. \n\nClaude 3.7 Sonnet maintains performance parity with its predecessor in standard mode while offering an extended reasoning mode for enhanced accuracy in math, coding, and instruction-following tasks.\n\nRead more at the [blog post here](https://www.anthropic.com/news/claude-3-7-sonnet)", - "context_length": 200000, - "architecture": { - "modality": "text+image->text", - "input_modalities": [ - "text", - "image" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Claude", - "instruct_type": null - }, - "pricing": { - "prompt": "0.000003", - "completion": "0.000015", - "request": "0", - "image": "0.0048", - "web_search": "0", - "internal_reasoning": "0", - "input_cache_read": "0.0000003", - "input_cache_write": "0.00000375" - }, - "top_provider": { - "context_length": 200000, - "max_completion_tokens": 64000, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ "tools", "tool_choice", - "max_tokens", - "temperature", "top_p", - "reasoning", - "include_reasoning", - "top_k", - "stop" + "top_k" ] }, { @@ -3447,6 +3506,51 @@ "tool_choice" ] }, + { + "id": "anthropic/claude-3.7-sonnet:thinking", + "hugging_face_id": "", + "name": "Anthropic: Claude 3.7 Sonnet (thinking)", + "created": 1740422110, + "description": "Claude 3.7 Sonnet is an advanced large language model with improved reasoning, coding, and problem-solving capabilities. It introduces a hybrid reasoning approach, allowing users to choose between rapid responses and extended, step-by-step processing for complex tasks. The model demonstrates notable improvements in coding, particularly in front-end development and full-stack updates, and excels in agentic workflows, where it can autonomously navigate multi-step processes. \n\nClaude 3.7 Sonnet maintains performance parity with its predecessor in standard mode while offering an extended reasoning mode for enhanced accuracy in math, coding, and instruction-following tasks.\n\nRead more at the [blog post here](https://www.anthropic.com/news/claude-3-7-sonnet)", + "context_length": 200000, + "architecture": { + "modality": "text+image->text", + "input_modalities": [ + "text", + "image" + ], + "output_modalities": [ + "text" + ], + "tokenizer": "Claude", + "instruct_type": null + }, + "pricing": { + "prompt": "0.000003", + "completion": "0.000015", + "request": "0", + "image": "0.0048", + "web_search": "0", + "internal_reasoning": "0", + "input_cache_read": "0.0000003", + "input_cache_write": "0.00000375" + }, + "top_provider": { + "context_length": 200000, + "max_completion_tokens": 128000, + "is_moderated": true + }, + "per_request_limits": null, + "supported_parameters": [ + "max_tokens", + "temperature", + "stop", + "reasoning", + "include_reasoning", + "tools", + "tool_choice" + ] + }, { "id": "perplexity/r1-1776", "hugging_face_id": "perplexity-ai/r1-1776", @@ -4019,12 +4123,12 @@ "stop", "frequency_penalty", "presence_penalty", + "seed", "top_k", + "min_p", "repetition_penalty", "logit_bias", - "min_p", "response_format", - "seed", "logprobs", "top_logprobs" ] @@ -4299,12 +4403,12 @@ "stop", "frequency_penalty", "presence_penalty", - "repetition_penalty", - "response_format", - "top_k", "seed", + "top_k", "min_p", - "logit_bias" + "repetition_penalty", + "logit_bias", + "response_format" ] }, { @@ -4335,7 +4439,7 @@ }, "top_provider": { "context_length": 64000, - "max_completion_tokens": 64000, + "max_completion_tokens": 32000, "is_moderated": false }, "per_request_limits": null, @@ -4348,12 +4452,12 @@ "stop", "frequency_penalty", "presence_penalty", + "seed", "top_k", + "min_p", "repetition_penalty", "logit_bias", - "min_p", - "response_format", - "seed" + "response_format" ] }, { @@ -4610,7 +4714,7 @@ "instruct_type": "deepseek-r1" }, "pricing": { - "prompt": "0.0000005", + "prompt": "0.00000045", "completion": "0.00000218", "request": "0", "image": "0", @@ -4634,13 +4738,13 @@ "presence_penalty", "seed", "top_k", + "min_p", "logit_bias", - "logprobs", "top_logprobs", - "repetition_penalty", "response_format", "structured_outputs", - "min_p", + "logprobs", + "repetition_penalty", "tools", "tool_choice" ] @@ -5146,13 +5250,13 @@ "frequency_penalty", "presence_penalty", "seed", + "top_k", + "min_p", + "repetition_penalty", "logit_bias", "logprobs", "top_logprobs", "response_format", - "top_k", - "min_p", - "repetition_penalty", "structured_outputs" ] }, @@ -5300,8 +5404,8 @@ "instruct_type": "deepseek-r1" }, "pricing": { - "prompt": "0.00000009", - "completion": "0.00000027", + "prompt": "0.0000002", + "completion": "0.0000002", "request": "0", "image": "0", "web_search": "0", @@ -6810,8 +6914,7 @@ "seed", "min_p", "logit_bias", - "top_logprobs", - "logprobs" + "top_logprobs" ] }, { @@ -6899,14 +7002,14 @@ "max_tokens", "temperature", "top_p", - "stop", + "top_k", + "seed", + "repetition_penalty", "frequency_penalty", "presence_penalty", - "seed", - "top_k", - "min_p", - "repetition_penalty", + "stop", "logit_bias", + "min_p", "response_format", "top_logprobs", "tools", @@ -7412,7 +7515,7 @@ "name": "Microsoft: Phi-3.5 Mini 128K Instruct", "created": 1724198400, "description": "Phi-3.5 models are lightweight, state-of-the-art open models. These models were trained with Phi-3 datasets that include both synthetic data and the filtered, publicly available websites data, with a focus on high quality and reasoning-dense properties. Phi-3.5 Mini uses 3.8B parameters, and is a dense decoder-only transformer model using the same tokenizer as [Phi-3 Mini](/models/microsoft/phi-3-mini-128k-instruct).\n\nThe models underwent a rigorous enhancement process, incorporating both supervised fine-tuning, proximal policy optimization, and direct preference optimization to ensure precise instruction adherence and robust safety measures. When assessed against benchmarks that test common sense, language understanding, math, code, long context and logical reasoning, Phi-3.5 models showcased robust and state-of-the-art performance among models with less than 13 billion parameters.", - "context_length": 131072, + "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ @@ -7425,15 +7528,15 @@ "instruct_type": "phi3" }, "pricing": { - "prompt": "0.00000003", - "completion": "0.00000009", + "prompt": "0.0000001", + "completion": "0.0000001", "request": "0", "image": "0", "web_search": "0", "internal_reasoning": "0" }, "top_provider": { - "context_length": 131072, + "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, @@ -7443,15 +7546,7 @@ "tool_choice", "max_tokens", "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" + "top_p" ] }, { @@ -7720,13 +7815,12 @@ "request": "0", "image": "0.003613", "web_search": "0", - "internal_reasoning": "0", - "input_cache_read": "0.00000125" + "internal_reasoning": "0" }, "top_provider": { "context_length": 128000, "max_completion_tokens": 16384, - "is_moderated": true + "is_moderated": false }, "per_request_limits": null, "supported_parameters": [ @@ -8151,7 +8245,7 @@ }, "top_provider": { "context_length": 131072, - "max_completion_tokens": 131072, + "max_completion_tokens": 16384, "is_moderated": false }, "per_request_limits": null, @@ -8162,17 +8256,17 @@ "stop", "frequency_penalty", "presence_penalty", - "seed", + "repetition_penalty", + "response_format", "top_k", + "seed", + "min_p", "logit_bias", "logprobs", "top_logprobs", "tools", "tool_choice", - "response_format", - "structured_outputs", - "repetition_penalty", - "min_p" + "structured_outputs" ] }, { @@ -8300,8 +8394,8 @@ "instruct_type": "gemma" }, "pricing": { - "prompt": "0.0000001", - "completion": "0.0000003", + "prompt": "0.0000008", + "completion": "0.0000008", "request": "0", "image": "0", "web_search": "0", @@ -8309,7 +8403,7 @@ }, "top_provider": { "context_length": 8192, - "max_completion_tokens": null, + "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, @@ -8324,10 +8418,7 @@ "repetition_penalty", "logit_bias", "min_p", - "response_format", - "seed", - "logprobs", - "top_logprobs" + "response_format" ] }, { @@ -8394,8 +8485,8 @@ "instruct_type": "gemma" }, "pricing": { - "prompt": "0.00000002", - "completion": "0.00000006", + "prompt": "0.0000002", + "completion": "0.0000002", "request": "0", "image": "0", "web_search": "0", @@ -8403,7 +8494,7 @@ }, "top_provider": { "context_length": 8192, - "max_completion_tokens": null, + "max_completion_tokens": 8192, "is_moderated": false }, "per_request_limits": null, @@ -8414,14 +8505,11 @@ "stop", "frequency_penalty", "presence_penalty", - "seed", - "top_k", - "min_p", - "repetition_penalty", - "logit_bias", "response_format", "top_logprobs", - "logprobs" + "logprobs", + "logit_bias", + "seed" ] }, { @@ -8897,7 +8985,7 @@ "name": "Microsoft: Phi-3 Medium 128K Instruct", "created": 1716508800, "description": "Phi-3 128K Medium is a powerful 14-billion parameter model designed for advanced language understanding, reasoning, and instruction following. Optimized through supervised fine-tuning and preference adjustments, it excels in tasks involving common sense, mathematics, logical reasoning, and code processing.\n\nAt time of release, Phi-3 Medium demonstrated state-of-the-art performance among lightweight models. In the MMLU-Pro eval, the model even comes close to a Llama3 70B level of performance.\n\nFor 4k context length, try [Phi-3 Medium 4K](/models/microsoft/phi-3-medium-4k-instruct).", - "context_length": 131072, + "context_length": 128000, "architecture": { "modality": "text->text", "input_modalities": [ @@ -8910,15 +8998,15 @@ "instruct_type": "phi3" }, "pricing": { - "prompt": "0.0000001", - "completion": "0.0000003", + "prompt": "0.000001", + "completion": "0.000001", "request": "0", "image": "0", "web_search": "0", "internal_reasoning": "0" }, "top_provider": { - "context_length": 131072, + "context_length": 128000, "max_completion_tokens": null, "is_moderated": false }, @@ -8928,15 +9016,7 @@ "tool_choice", "max_tokens", "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" + "top_p" ] }, { @@ -8984,52 +9064,6 @@ "seed" ] }, - { - "id": "deepseek/deepseek-coder", - "hugging_face_id": "deepseek-ai/DeepSeek-Coder-V2-Instruct", - "name": "DeepSeek-Coder-V2", - "created": 1715644800, - "description": "DeepSeek-Coder-V2, an open-source Mixture-of-Experts (MoE) code language model. It is further pre-trained from an intermediate checkpoint of DeepSeek-V2 with additional 6 trillion tokens.\n\nThe original V1 model was trained from scratch on 2T tokens, with a composition of 87% code and 13% natural language in both English and Chinese. It was pre-trained on project-level code corpus by employing a extra fill-in-the-blank task.", - "context_length": 128000, - "architecture": { - "modality": "text->text", - "input_modalities": [ - "text" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Other", - "instruct_type": null - }, - "pricing": { - "prompt": "0.00000004", - "completion": "0.00000012", - "request": "0", - "image": "0", - "web_search": "0", - "internal_reasoning": "0" - }, - "top_provider": { - "context_length": 128000, - "max_completion_tokens": null, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ - "max_tokens", - "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" - ] - }, { "id": "google/gemini-flash-1.5", "hugging_face_id": null, @@ -9282,52 +9316,6 @@ "structured_outputs" ] }, - { - "id": "allenai/olmo-7b-instruct", - "hugging_face_id": "allenai/OLMo-7B-Instruct", - "name": "OLMo 7B Instruct", - "created": 1715299200, - "description": "OLMo 7B Instruct by the Allen Institute for AI is a model finetuned for question answering. It demonstrates **notable performance** across multiple benchmarks including TruthfulQA and ToxiGen.\n\n**Open Source**: The model, its code, checkpoints, logs are released under the [Apache 2.0 license](https://choosealicense.com/licenses/apache-2.0).\n\n- [Core repo (training, inference, fine-tuning etc.)](https://github.com/allenai/OLMo)\n- [Evaluation code](https://github.com/allenai/OLMo-Eval)\n- [Further fine-tuning code](https://github.com/allenai/open-instruct)\n- [Paper](https://arxiv.org/abs/2402.00838)\n- [Technical blog post](https://blog.allenai.org/olmo-open-language-model-87ccfc95f580)\n- [W&B Logs](https://wandb.ai/ai2-llm/OLMo-7B/reports/OLMo-7B--Vmlldzo2NzQyMzk5)", - "context_length": 2048, - "architecture": { - "modality": "text->text", - "input_modalities": [ - "text" - ], - "output_modalities": [ - "text" - ], - "tokenizer": "Other", - "instruct_type": "zephyr" - }, - "pricing": { - "prompt": "0.00000008", - "completion": "0.00000024", - "request": "0", - "image": "0", - "web_search": "0", - "internal_reasoning": "0" - }, - "top_provider": { - "context_length": 2048, - "max_completion_tokens": null, - "is_moderated": false - }, - "per_request_limits": null, - "supported_parameters": [ - "max_tokens", - "temperature", - "top_p", - "stop", - "frequency_penalty", - "presence_penalty", - "seed", - "top_k", - "logit_bias", - "logprobs", - "top_logprobs" - ] - }, { "id": "neversleep/llama-3-lumimaid-8b", "hugging_face_id": "NeverSleep/Llama-3-Lumimaid-8B-v0.1", @@ -9542,8 +9530,8 @@ "instruct_type": "mistral" }, "pricing": { - "prompt": "0.0000004", - "completion": "0.0000012", + "prompt": "0.0000009", + "completion": "0.0000009", "request": "0", "image": "0", "web_search": "0", @@ -9610,13 +9598,13 @@ "max_tokens", "temperature", "top_p", - "presence_penalty", - "frequency_penalty", - "repetition_penalty", - "top_k", "stop", + "frequency_penalty", + "presence_penalty", "seed", + "top_k", "min_p", + "repetition_penalty", "logit_bias", "response_format" ] @@ -9837,7 +9825,7 @@ }, "top_provider": { "context_length": 4096, - "max_completion_tokens": null, + "max_completion_tokens": 2048, "is_moderated": false }, "per_request_limits": null, @@ -10682,9 +10670,7 @@ "logit_bias", "min_p", "response_format", - "seed", - "logprobs", - "top_logprobs" + "seed" ] }, { @@ -11579,13 +11565,13 @@ "max_tokens", "temperature", "top_p", - "presence_penalty", - "frequency_penalty", - "repetition_penalty", - "top_k", "stop", + "frequency_penalty", + "presence_penalty", "seed", + "top_k", "min_p", + "repetition_penalty", "logit_bias", "response_format", "top_a" From 559fcbf3b9b41a09ea8bd0a4818628ca707140e8 Mon Sep 17 00:00:00 2001 From: GitHappens2Me Date: Thu, 29 May 2025 19:42:49 +0200 Subject: [PATCH 7/8] fixed non-streaming responses --- .gitignore | 6 +++--- router/proxy.py | 9 ++++++++- 2 files changed, 11 insertions(+), 4 deletions(-) diff --git a/.gitignore b/.gitignore index 359006b7..51b3e357 100644 --- a/.gitignore +++ b/.gitignore @@ -5,7 +5,7 @@ wallet.sqlite3 # Development .notes -.keys.db -.wallet.sqlite3 -.models.json +.*keys.db +.*wallet.sqlite3 +.*models.json compose.override.yml diff --git a/router/proxy.py b/router/proxy.py index 91f0c0c2..6552bfb3 100644 --- a/router/proxy.py +++ b/router/proxy.py @@ -189,10 +189,17 @@ async def proxy( key, response_json, session ) response_json["cost"] = cost_data + + response_headers = dict(response.headers) + + # Remove Transfer-Encoding header to avoid conflict with Content-Length header in common nginx setups + if "transfer-encoding" in response_headers: + del response_headers["transfer-encoding"] + return Response( content=json.dumps(response_json).encode(), status_code=response.status_code, - headers=dict(response.headers), + headers=response_headers, media_type="application/json", ) except json.JSONDecodeError as e: From 4b9974b0378ff9522a5bd5783543b8ee6aca3de9 Mon Sep 17 00:00:00 2001 From: shroominic Date: Tue, 3 Jun 2025 12:32:07 +0000 Subject: [PATCH 8/8] make donation rate adjustable --- router/cashu.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/router/cashu.py b/router/cashu.py index 41f8e580..65cacc29 100644 --- a/router/cashu.py +++ b/router/cashu.py @@ -14,7 +14,7 @@ MINT = os.environ.get("MINT", "https://mint.minibits.cash/Bitcoin") MINIMUM_PAYOUT = int(os.environ.get("MINIMUM_PAYOUT", 100)) REFUND_PROCESSING_INTERVAL = int(os.environ.get("REFUND_PROCESSING_INTERVAL", 3600)) DEV_LN_ADDRESS = "routstr@minibits.cash" -DEVS_DONATION_RATE = 0.021 # 2.1% +DEVS_DONATION_RATE = float(os.environ.get("DEVS_DONATION_RATE", 0.021)) # 2.1% WALLET = None #TODO
{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{key.key_expiry_time} ({expiry_time_human_readable} UTC)
{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{key.key_expiry_time}
{key.hashed_key}{key.balance}{key.total_spent}{key.total_requests}{key.refund_address}{'{} ({} UTC)'.format(key.key_expiry_time, expiry_time_human_readable) if key.key_expiry_time else key.key_expiry_time}